Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
212634ab5b | ||
|
|
761f73ebd9 | ||
|
|
c6fdeedf51 | ||
|
|
7fa27c6cbc |
@@ -1,2 +0,0 @@
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
@@ -1,166 +1 @@
|
||||
# Byte-compiled / optimized / DLL files
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
*$py.class
|
||||
|
||||
# C extensions
|
||||
*.so
|
||||
|
||||
# Distribution / packaging
|
||||
.Python
|
||||
build/
|
||||
develop-eggs/
|
||||
dist/
|
||||
downloads/
|
||||
eggs/
|
||||
.eggs/
|
||||
lib/
|
||||
lib64/
|
||||
parts/
|
||||
sdist/
|
||||
var/
|
||||
wheels/
|
||||
share/python-wheels/
|
||||
*.egg-info/
|
||||
.installed.cfg
|
||||
*.egg
|
||||
MANIFEST
|
||||
|
||||
# PyInstaller
|
||||
# Usually these files are written by a python script from a template
|
||||
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
||||
*.manifest
|
||||
*.spec
|
||||
|
||||
# Installer logs
|
||||
pip-log.txt
|
||||
pip-delete-this-directory.txt
|
||||
|
||||
# Unit test / coverage reports
|
||||
htmlcov/
|
||||
.tox/
|
||||
.nox/
|
||||
.coverage
|
||||
.coverage.*
|
||||
.cache
|
||||
nosetests.xml
|
||||
coverage.xml
|
||||
*.cover
|
||||
*.py,cover
|
||||
.hypothesis/
|
||||
.pytest_cache/
|
||||
cover/
|
||||
|
||||
# Translations
|
||||
*.mo
|
||||
*.pot
|
||||
|
||||
# Django stuff:
|
||||
*.log
|
||||
local_settings.py
|
||||
db.sqlite3
|
||||
db.sqlite3-journal
|
||||
|
||||
# Flask stuff:
|
||||
instance/
|
||||
.webassets-cache
|
||||
|
||||
# Scrapy stuff:
|
||||
.scrapy
|
||||
|
||||
# Sphinx documentation
|
||||
docs/_build/
|
||||
|
||||
# PyBuilder
|
||||
.pybuilder/
|
||||
target/
|
||||
|
||||
# Jupyter Notebook
|
||||
.ipynb_checkpoints
|
||||
|
||||
# IPython
|
||||
profile_default/
|
||||
ipython_config.py
|
||||
|
||||
# pyenv
|
||||
# For a library or package, you might want to ignore these files since the code is
|
||||
# intended to run in multiple environments; otherwise, check them in:
|
||||
# .python-version
|
||||
|
||||
# pipenv
|
||||
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
|
||||
# However, in case of collaboration, if having platform-specific dependencies or dependencies
|
||||
# having no cross-platform support, pipenv may install dependencies that don't work, or not
|
||||
# install all needed dependencies.
|
||||
#Pipfile.lock
|
||||
|
||||
# poetry
|
||||
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
|
||||
# This is especially recommended for binary packages to ensure reproducibility, and is more
|
||||
# commonly ignored for libraries.
|
||||
# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
|
||||
#poetry.lock
|
||||
|
||||
# pdm
|
||||
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
|
||||
#pdm.lock
|
||||
# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
|
||||
# in version control.
|
||||
# https://pdm.fming.dev/latest/usage/project/#working-with-version-control
|
||||
.pdm.toml
|
||||
.pdm-python
|
||||
.pdm-build/
|
||||
|
||||
# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
|
||||
__pypackages__/
|
||||
|
||||
# Celery stuff
|
||||
celerybeat-schedule
|
||||
celerybeat.pid
|
||||
|
||||
# SageMath parsed files
|
||||
*.sage.py
|
||||
|
||||
# Environments
|
||||
.env
|
||||
.venv
|
||||
env/
|
||||
venv/
|
||||
ENV/
|
||||
env.bak/
|
||||
venv.bak/
|
||||
|
||||
# Spyder project settings
|
||||
.spyderproject
|
||||
.spyproject
|
||||
|
||||
# Rope project settings
|
||||
.ropeproject
|
||||
|
||||
# mkdocs documentation
|
||||
/site
|
||||
|
||||
# mypy
|
||||
.mypy_cache/
|
||||
.dmypy.json
|
||||
dmypy.json
|
||||
|
||||
# Pyre type checker
|
||||
.pyre/
|
||||
|
||||
# pytype static type analyzer
|
||||
.pytype/
|
||||
|
||||
# Cython debug symbols
|
||||
cython_debug/
|
||||
|
||||
# PyCharm
|
||||
# JetBrains specific template is maintained in a separate JetBrains.gitignore that can
|
||||
# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
|
||||
# and can be added to the global gitignore or merged into this file. For a more nuclear
|
||||
# option (not recommended) you can uncomment the following to ignore the entire idea folder.
|
||||
#.idea/
|
||||
|
||||
output
|
||||
inference/xcodec_mini_infer
|
||||
inference/run_infer.sh
|
||||
__pycache__/
|
||||
@@ -1,276 +0,0 @@
|
||||
{
|
||||
"last_node_id": 16,
|
||||
"last_link_id": 23,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 9,
|
||||
"type": "YUE_Stage_B_Sampler",
|
||||
"pos": [
|
||||
24645.052734375,
|
||||
-4632.9755859375
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
150
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "stage1_set",
|
||||
"type": "STAGE_SET",
|
||||
"link": 10
|
||||
},
|
||||
{
|
||||
"name": "model",
|
||||
"type": "MODEL_YUE_B",
|
||||
"link": 23
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"links": [
|
||||
13
|
||||
],
|
||||
"slot_index": 0
|
||||
},
|
||||
{
|
||||
"name": "string",
|
||||
"type": "STRING",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "YUE_Stage_B_Sampler"
|
||||
},
|
||||
"widgets_values": [
|
||||
"decoder_131000.pth",
|
||||
"decoder_151000.pth",
|
||||
4
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 8,
|
||||
"type": "YUE_Stage_A_Sampler",
|
||||
"pos": [
|
||||
24111.734375,
|
||||
-4703.33349609375
|
||||
],
|
||||
"size": [
|
||||
449.79949951171875,
|
||||
795.2402954101562
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "model",
|
||||
"type": "MODEL_YUE_A",
|
||||
"link": 21
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "stage1_set",
|
||||
"type": "STAGE_SET",
|
||||
"links": [
|
||||
10
|
||||
],
|
||||
"slot_index": 0
|
||||
},
|
||||
{
|
||||
"name": "info",
|
||||
"type": "quantization_model",
|
||||
"links": [
|
||||
22
|
||||
],
|
||||
"slot_index": 1
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "YUE_Stage_A_Sampler"
|
||||
},
|
||||
"widgets_values": [
|
||||
"inspiring female uplifting pop airy vocal electronic bright vocal vocal.",
|
||||
"[verse]\nStaring at the sunset, colors paint the sky.\nThoughts of you keep swirling, can't deny.\nI know I let you down, I made mistakes.\nBut I'm here to mend the heart I didn't break.\n\n[chorus]\nEvery road you take, I'll be one step behind.\nEvery dream you chase, I'm reaching for the light.\nYou can't fight this feeling now.\nI won't back down.\nYou know you can't deny it now.\n I won't back down \n\n[verse]\nThey might say I'm foolish, chasing after you.\nBut they don't feel this love the way we do.\nMy heart beats only for you, can't you see?\nI won't let you slip away from me. \n\n[chorus]\nEvery road you take, I'll be one step behind.\nEvery dream you chase, I'm reaching for the light.\nYou can't fight this feeling now.\nI won't back down.\nYou know you can't deny it now.\n I won't back down \n\n[bridge]\nNo, I won't back down, won't turn around.\nUntil you're back where you belong.\nI'll cross the oceans wide, stand by your side.\nTogether we are strong. \n\n[outro]\nEvery road you take, I'll be one step behind.\nEvery dream you chase, love's the tie that binds.\nYou can't fight this feeling now.\nI won't back down.",
|
||||
1698056329,
|
||||
"randomize",
|
||||
2,
|
||||
1.1,
|
||||
0,
|
||||
30,
|
||||
3000,
|
||||
true,
|
||||
true,
|
||||
true,
|
||||
true
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 15,
|
||||
"type": "YUE_Stage_A_Loader",
|
||||
"pos": [
|
||||
23737.6640625,
|
||||
-4575.09814453125
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
202
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "model",
|
||||
"type": "MODEL_YUE_A",
|
||||
"links": [
|
||||
21
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "YUE_Stage_A_Loader"
|
||||
},
|
||||
"widgets_values": [
|
||||
"F:/test/ComfyUI/models/diffusers/Alissonerdx/YuE-s1-7B-anneal-zh-icl-int8",
|
||||
"ckpt_00360000.pth",
|
||||
"int8",
|
||||
false,
|
||||
16384,
|
||||
"FP16",
|
||||
2
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 16,
|
||||
"type": "YUE_Stage_B_Loader",
|
||||
"pos": [
|
||||
24641.8046875,
|
||||
-4368.71630859375
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
154
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "info",
|
||||
"type": "quantization_model",
|
||||
"link": 22
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "model",
|
||||
"type": "MODEL_YUE_B",
|
||||
"links": [
|
||||
23
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "YUE_Stage_B_Loader"
|
||||
},
|
||||
"widgets_values": [
|
||||
"F:/test/ComfyUI/models/diffusers/Alissonerdx/YuE-s2-1B-general-int8",
|
||||
8192,
|
||||
2,
|
||||
"FP16",
|
||||
false
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 5,
|
||||
"type": "PreviewAudio",
|
||||
"pos": [
|
||||
25016.130859375,
|
||||
-4446.78271484375
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
76
|
||||
],
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"link": 13
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "PreviewAudio"
|
||||
},
|
||||
"widgets_values": [
|
||||
null
|
||||
]
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
10,
|
||||
8,
|
||||
0,
|
||||
9,
|
||||
0,
|
||||
"STAGE_SET"
|
||||
],
|
||||
[
|
||||
13,
|
||||
9,
|
||||
0,
|
||||
5,
|
||||
0,
|
||||
"AUDIO"
|
||||
],
|
||||
[
|
||||
21,
|
||||
15,
|
||||
0,
|
||||
8,
|
||||
0,
|
||||
"MODEL_YUE_A"
|
||||
],
|
||||
[
|
||||
22,
|
||||
8,
|
||||
1,
|
||||
16,
|
||||
0,
|
||||
"quantization_model"
|
||||
],
|
||||
[
|
||||
23,
|
||||
16,
|
||||
0,
|
||||
9,
|
||||
1,
|
||||
"MODEL_YUE_B"
|
||||
]
|
||||
],
|
||||
"groups": [],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 0.8140274938684142,
|
||||
"offset": [
|
||||
-23442.787805256274,
|
||||
4871.586176223436
|
||||
]
|
||||
}
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
@@ -1,3 +1,4 @@
|
||||
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
@@ -0,0 +1,185 @@
|
||||
---
|
||||
license: cc-by-nc-4.0
|
||||
library_name: transformers
|
||||
pipeline_tag: feature-extraction
|
||||
tags:
|
||||
- audio
|
||||
- music
|
||||
- music-understanding
|
||||
- representation-learning
|
||||
- mert2
|
||||
- custom_code
|
||||
---
|
||||
<h1 align="center">🤗 MERT-v2-FullSong</h1>
|
||||
<p align="center"><strong>Music representations with full-song context</strong></p>
|
||||
<p align="center">24 kHz mono · 632M parameters · 24 layers · 1,024 dimensions · 25 Hz</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="https://map-yue2.github.io/">🎵 YuE2 project</a>
|
||||
·
|
||||
<a href="#quick-start">🚀 Quick start</a>
|
||||
·
|
||||
<a href="#marble">📊 MARBLE</a>
|
||||
·
|
||||
<a href="#layer-guide">🎛️ Layer guide</a>
|
||||
·
|
||||
<a href="#citation">📚 Citation</a>
|
||||
</p>
|
||||
<p align="center">
|
||||
<a href="https://huggingface.co/m-a-p/MERT-v2-30s"><img alt="MERT-v2-30s" src="https://img.shields.io/badge/MERT--v2--30s-374151?logo=huggingface&logoColor=FFD21E" height="20" /></a>
|
||||
|
||||
<a href="https://huggingface.co/m-a-p/MERT-v2-FullSong"><img alt="MERT-v2-FullSong" src="https://img.shields.io/badge/MERT--v2--FullSong-374151?logo=huggingface&logoColor=FFD21E" height="20" /></a>
|
||||
|
||||
<a href="https://huggingface.co/m-a-p/YuE2-3B"><img alt="YuE2-3B" src="https://img.shields.io/badge/YuE2--3B-374151?logo=huggingface&logoColor=FFD21E" height="20" /></a>
|
||||
|
||||
<a href="https://huggingface.co/m-a-p/YuE2-Vae"><img alt="🤗 YuE2-Vae" src="https://img.shields.io/badge/YuE2--Vae-374151?logo=huggingface&logoColor=FFD21E" height="20" /></a>
|
||||
|
||||
<a href="https://huggingface.co/m-a-p/YuE2-Vae-legacy"><img alt="🤗 YuE2-Vae-legacy" src="https://img.shields.io/badge/YuE2--Vae--legacy-374151?logo=huggingface&logoColor=FFD21E" height="20" /></a>
|
||||
|
||||
<a href="https://huggingface.co/datasets/m-a-p/WildSongBench"><img alt="🤗 WildSongBench" src="https://img.shields.io/badge/WildSongBench-374151?logo=huggingface&logoColor=FFD21E" height="20" /></a>
|
||||
|
||||
<a href="https://huggingface.co/m-a-p/SheetSage2"><img alt="SheetSage2" src="https://img.shields.io/badge/SheetSage2-374151?logo=huggingface&logoColor=FFD21E" height="20" /></a>
|
||||
</p>
|
||||
|
||||
**MERT-v2-FullSong is a bidirectional music encoder adapted to complete songs lasting 30–360 seconds.** Extract general-purpose music representations at the frame or recording level. Load with standard Hugging Face Transformers, using the familiar [MERT](https://huggingface.co/m-a-p/MERT-v1-330M) workflow.
|
||||
|
||||
It continues pretraining from [MERT-v2-30s](https://huggingface.co/m-a-p/MERT-v2-30s) and preserves the same feature interface.
|
||||
|
||||

|
||||
|
||||
*MERT-v2-30s uses the bidirectional backbone in (B); MERT-v2-FullSong continues through the full-song branch in (C). The causal branch is used for YuE2 tokenization.*
|
||||
|
||||
<a id="quick-start"></a>
|
||||
|
||||
## 🚀 Quick start
|
||||
|
||||
Install the matching PyTorch packages:
|
||||
|
||||
```bash
|
||||
python -m pip install torch==2.6.0 torchaudio==2.6.0 transformers==4.53.2 huggingface-hub safetensors soundfile
|
||||
```
|
||||
|
||||
```python
|
||||
import soundfile as sf
|
||||
import torch
|
||||
import torchaudio.functional as AF
|
||||
from transformers import AutoFeatureExtractor, AutoModel
|
||||
|
||||
repo = "m-a-p/MERT-v2-FullSong"
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
processor = AutoFeatureExtractor.from_pretrained(repo, trust_remote_code=True)
|
||||
model = AutoModel.from_pretrained(repo, trust_remote_code=True).eval().to(device)
|
||||
|
||||
audio, sr = sf.read("music.wav", dtype="float32", always_2d=True)
|
||||
audio = audio[:30 * sr].mean(axis=1) # First 30 seconds, mixed to mono.
|
||||
waveform = AF.resample(torch.from_numpy(audio), sr, processor.sampling_rate)
|
||||
inputs = processor(
|
||||
waveform.numpy(), sampling_rate=processor.sampling_rate, return_tensors="pt",
|
||||
).to(device)
|
||||
|
||||
with torch.inference_mode():
|
||||
output = model(**inputs, output_hidden_states=True)
|
||||
|
||||
frames = output.last_hidden_state # [batch, frames, 1024], 25 Hz
|
||||
layers = output.hidden_states # 24 tensors, one per block
|
||||
mask = output.feature_attention_mask[..., None]
|
||||
embedding = (frames * mask).sum(1) / mask.sum(1).clamp_min(1) # [batch, 1024]
|
||||
```
|
||||
|
||||
`hidden_states[0]` is block 1; `hidden_states[23]` is block 24. Remove the audio slice to process a complete song.
|
||||
|
||||
<a id="marble"></a>
|
||||
|
||||
## 📊 MARBLE
|
||||
|
||||
Reported frozen-encoder results on [MARBLE](https://github.com/a43992899/MARBLE). All scores are multiplied by 100; higher is better. **Bold** marks the best displayed value in each column, including ties.
|
||||
|
||||
Baseline scores and parameter counts are reproduced from Tables III and V of [PupuJEPA](https://arxiv.org/html/2606.25713v2); the baselines were not rerun for this release. Evaluation protocols may differ across sources.
|
||||
|
||||
**General music understanding.** MTT = MagnaTagATune; key = GiantSteps refined key accuracy; genre and beat = GTZAN; valence and arousal = EmoMusic. ROC = ROC-AUC; AP = average precision.
|
||||
|
||||
| Model | Params | MTT ROC | MTT AP | Key acc. | Genre acc. | Beat F1 | Valence R² | Arousal R² |
|
||||
|---|---:|---:|---:|---:|---:|---:|---:|---:|
|
||||
| MERT-Large | 330M | 90.6 | 37.9 | 64.1 | 77.6 | 86.8 | 56.7 | 76.1 |
|
||||
| Dasheng-1.2B | 1.2B | 91.5 | 40.4 | 58.0 | 81.4 | 87.7 | 57.4 | 75.0 |
|
||||
| MuQ | 310M | 90.5 | 38.5 | 63.2 | 83.8 | 90.1 | 58.3 | 76.4 |
|
||||
| MusicFM | 330M | 90.9 | 38.3 | 63.0 | 84.1 | 90.2 | 57.2 | 74.4 |
|
||||
| AudioMAE++ | 307M | 91.2 | 39.5 | 61.7 | 80.3 | 90.0 | 59.0 | 75.7 |
|
||||
| MATPAC++ | 307M | 90.6 | 38.2 | 63.7 | 81.4 | 90.1 | 57.8 | 74.7 |
|
||||
| A-JEPA | 307M | 91.0 | 39.2 | 65.0 | 83.8 | 90.0 | 57.4 | 74.8 |
|
||||
| PupuJEPA-Large | 307M | 91.7 | 40.8 | 66.1 | 86.9 | **91.0** | 62.5 | 76.8 |
|
||||
| PupuJEPA-Huge | 632M | 91.3 | 39.7 | 64.8 | 85.9 | 90.5 | 62.0 | 78.5 |
|
||||
| **MERT-v2-30s** | 632M | **91.91** | **41.29** | 66.97 | **91.72** | 90.59 | 63.23 | **80.01** |
|
||||
| **MERT-v2-FullSong** | 632M | 91.74 | 41.20 | **67.05** | 90.69 | 90.57 | **63.52** | 78.14 |
|
||||
|
||||
**MTG-Jamendo tagging.** Mood denotes mood/theme tags.
|
||||
|
||||
| Model | Params | Instrument ROC | Instrument AP | Mood ROC | Mood AP | Genre ROC | Genre AP | Top-50 ROC | Top-50 AP |
|
||||
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
||||
| MERT-Large | 330M | 75.5 | 18.8 | 75.3 | 13.5 | 86.1 | 18.0 | 82.6 | 29.1 |
|
||||
| Dasheng-1.2B | 1.2B | 75.0 | 19.0 | 76.1 | 15.5 | 85.5 | 18.8 | 82.4 | 29.6 |
|
||||
| MuQ | 310M | 74.8 | 19.1 | 73.7 | 13.2 | 85.4 | 19.1 | 83.0 | 30.2 |
|
||||
| MusicFM | 330M | 74.6 | 18.5 | 74.9 | 14.1 | 85.3 | 19.4 | 81.9 | 29.7 |
|
||||
| AudioMAE++ | 307M | 77.1 | 19.9 | 75.6 | 14.0 | 86.3 | 18.9 | 83.1 | 31.1 |
|
||||
| MATPAC++ | 307M | 77.2 | 19.7 | 75.1 | 14.1 | 85.7 | 19.6 | 82.5 | 30.2 |
|
||||
| A-JEPA | 307M | 76.6 | 19.3 | 74.6 | 14.3 | 85.5 | 19.2 | 82.5 | 29.6 |
|
||||
| PupuJEPA-Large | 307M | 78.4 | 21.2 | 76.2 | 15.3 | 86.1 | 20.1 | 82.8 | 30.5 |
|
||||
| PupuJEPA-Huge | 632M | 77.6 | 20.5 | 75.9 | 14.7 | 85.9 | 20.1 | 83.1 | 30.7 |
|
||||
| **MERT-v2-30s** | 632M | **80.27** | 22.89 | **79.44** | **16.68** | **88.01** | **21.22** | **84.18** | **32.17** |
|
||||
| **MERT-v2-FullSong** | 632M | **80.27** | **23.51** | 78.74 | 15.74 | 87.98 | 20.66 | 84.13 | 31.62 |
|
||||
|
||||
**Supplemental:** Chords1217 frame accuracy is **78.48** for MERT-v2-30s and **77.79** for MERT-v2-FullSong.
|
||||
|
||||
<!-- mert2-hf-validation:start -->
|
||||
**HF reproduction verified:** Both models reproduce all ten MARBLE tasks with the fixed evaluation settings. [Scores and best settings](marble_results.json).
|
||||
<!-- mert2-hf-validation:end -->
|
||||
|
||||
<a id="layer-guide"></a>
|
||||
|
||||
## 🎛️ Layer guide
|
||||
|
||||
Recommended probe settings with the encoder frozen. L1 = `hidden_states[0]`; All-layer MLP uses all 24 layers.
|
||||
|
||||
| Task / dataset | MERT-v2-30s layer | Probe LR | MERT-v2-FullSong layer | Probe LR |
|
||||
|---|---|---:|---|---:|
|
||||
| Genre · GTZAN | L23 | 5e-3 | L24 | 5e-4 |
|
||||
| Beat · GTZAN | L21 | 1e-3 | L23 | 1e-3 |
|
||||
| Key · GiantSteps | L4 | 1e-3 | L23 | 1e-3 |
|
||||
| Emotion · EmoMusic | All-layer MLP | 5e-5 | L24 | 5e-4 |
|
||||
| Chords · Chords1217 | All-layer MLP | 1e-4 | All-layer MLP | 5e-4 |
|
||||
| Tagging · MagnaTagATune | L22 | 1e-3 | L23 | 1e-3 |
|
||||
| Instrument · MTG-Jamendo | L14 | 1e-3 | L12 | 1e-3 |
|
||||
| Mood/theme · MTG-Jamendo | L16 | 1e-3 | L13 | 1e-3 |
|
||||
| Genre · MTG-Jamendo | L19 | 1e-3 | L16 | 1e-3 |
|
||||
| Top-50 · MTG-Jamendo | L13 | 1e-3 | L22 | 1e-3 |
|
||||
|
||||
<a id="citation"></a>
|
||||
|
||||
## 📚 Citation
|
||||
|
||||
**Technical report coming soon.** For now, please cite [MERT (ICLR 2024)](https://proceedings.iclr.cc/paper_files/paper/2024/hash/33dffa2e3d2ab74a783d1a8c292f66d9-Abstract-Conference.html) and [MARBLE (NeurIPS 2023)](https://proceedings.neurips.cc/paper_files/paper/2023/hash/7cbeec46f979618beafb4f46d8f39f36-Abstract-Datasets_and_Benchmarks.html) when using MERT-v2 or its benchmark results in your research.
|
||||
|
||||
```bibtex
|
||||
@inproceedings{li2024mert,
|
||||
title = {MERT: Acoustic Music Understanding Model with Large-Scale Self-supervised Training},
|
||||
author = {Li, Yizhi and Yuan, Ruibin and Zhang, Ge and Ma, Yinghao and Chen, Xingran and Yin, Hanzhi and Xiao, Chenghao and Lin, Chenghua and Ragni, Anton and Benetos, Emmanouil and Gyenge, Norbert and Dannenberg, Roger and Liu, Ruibo and Chen, Wenhu and Xia, Gus and Shi, Yemin and Huang, Wenhao and Wang, Zili and Guo, Yike and Fu, Jie},
|
||||
booktitle = {International Conference on Learning Representations},
|
||||
year = {2024},
|
||||
url = {https://proceedings.iclr.cc/paper_files/paper/2024/hash/33dffa2e3d2ab74a783d1a8c292f66d9-Abstract-Conference.html}
|
||||
}
|
||||
|
||||
@inproceedings{yuan2023marble,
|
||||
title = {MARBLE: Music Audio Representation Benchmark for Universal Evaluation},
|
||||
author = {Yuan, Ruibin and Ma, Yinghao and Li, Yizhi and Zhang, Ge and Chen, Xingran and Yin, Hanzhi and Zhuo, Le and Liu, Yiqi and Huang, Jiawen and Tian, Zeyue and Deng, Binyue and Wang, Ningzhi and Lin, Chenghua and Benetos, Emmanouil and Ragni, Anton and Gyenge, Norbert and Dannenberg, Roger and Chen, Wenhu and Xia, Gus and Xue, Wei and Liu, Si and Wang, Shi and Liu, Ruibo and Guo, Yike and Fu, Jie},
|
||||
booktitle = {Advances in Neural Information Processing Systems},
|
||||
volume = {36},
|
||||
pages = {39626--39647},
|
||||
year = {2023},
|
||||
doi = {10.52202/075280-1722},
|
||||
url = {https://proceedings.neurips.cc/paper_files/paper/2023/hash/7cbeec46f979618beafb4f46d8f39f36-Abstract-Datasets_and_Benchmarks.html}
|
||||
}
|
||||
```
|
||||
|
||||
Weights: [CC BY-NC 4.0](LICENSE). [Dependency notices](THIRD_PARTY_NOTICES.md).
|
||||
|
||||
**YuE2 family:** [Song generation](https://huggingface.co/m-a-p/YuE2-3B) · [Audio decoder](https://huggingface.co/m-a-p/YuE2-Vae) · [Benchmark decoder](https://huggingface.co/m-a-p/YuE2-Vae-legacy).
|
||||
@@ -0,0 +1,11 @@
|
||||
# Third-party notices
|
||||
|
||||
The MERT2 inference implementation uses separately installed PyTorch,
|
||||
torchaudio, Hugging Face Transformers, huggingface_hub, and safetensors.
|
||||
These dependencies retain their respective upstream licenses and notices;
|
||||
this model repository does not redistribute their source distributions.
|
||||
|
||||
The example additionally uses SoundFile, which retains its upstream license.
|
||||
|
||||
The MERT2 checkpoint weights are licensed separately under CC BY-NC 4.0;
|
||||
see [LICENSE](LICENSE). This weight license does not relicense dependencies.
|
||||
@@ -0,0 +1,41 @@
|
||||
{
|
||||
"architectures": [
|
||||
"MERT2Model"
|
||||
],
|
||||
"auto_map": {
|
||||
"AutoConfig": "configuration_mert2.MERT2Config",
|
||||
"AutoModel": "modeling_mert2.MERT2Model"
|
||||
},
|
||||
"context_seconds": 360.0,
|
||||
"conv_depthwise_kernel_size": 31,
|
||||
"frame_rate": 25.0,
|
||||
"hidden_size": 1024,
|
||||
"hop_length": 240,
|
||||
"initializer_range": 0.02,
|
||||
"inputs_to_logits_ratio": 960,
|
||||
"intermediate_size": 4096,
|
||||
"layer_norm_eps": 1e-05,
|
||||
"minimum_input_samples": 1025,
|
||||
"model_type": "mert2",
|
||||
"n_fft": 2048,
|
||||
"num_attention_heads": 16,
|
||||
"num_hidden_layers": 24,
|
||||
"num_mel_bins": 128,
|
||||
"rotary_embedding_base": 10000,
|
||||
"sampling_rate": 24000,
|
||||
"subsampling_channels": [
|
||||
128,
|
||||
512,
|
||||
1024
|
||||
],
|
||||
"subsampling_depths": [
|
||||
3,
|
||||
4,
|
||||
5
|
||||
],
|
||||
"subsampling_layer_norm_eps": 1e-06,
|
||||
"torch_dtype": "float32",
|
||||
"transformers_version": "4.53.2",
|
||||
"variant": "fs",
|
||||
"win_length": 2048
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
"""Configuration for MERT2 music representation models."""
|
||||
|
||||
from transformers import PretrainedConfig
|
||||
|
||||
|
||||
class MERT2Config(PretrainedConfig):
|
||||
"""Architecture shared by the 30-second and full-song MERT2 encoders."""
|
||||
|
||||
model_type = "mert2"
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
hidden_size=1024,
|
||||
intermediate_size=4096,
|
||||
num_hidden_layers=24,
|
||||
num_attention_heads=16,
|
||||
num_mel_bins=128,
|
||||
sampling_rate=24000,
|
||||
n_fft=2048,
|
||||
win_length=2048,
|
||||
hop_length=240,
|
||||
subsampling_channels=None,
|
||||
subsampling_depths=None,
|
||||
conv_depthwise_kernel_size=31,
|
||||
rotary_embedding_base=10000,
|
||||
layer_norm_eps=1e-5,
|
||||
subsampling_layer_norm_eps=1e-6,
|
||||
initializer_range=0.02,
|
||||
variant="30s",
|
||||
context_seconds=None,
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__(**kwargs)
|
||||
self.hidden_size = int(hidden_size)
|
||||
self.intermediate_size = int(intermediate_size)
|
||||
self.num_hidden_layers = int(num_hidden_layers)
|
||||
self.num_attention_heads = int(num_attention_heads)
|
||||
self.num_mel_bins = int(num_mel_bins)
|
||||
self.sampling_rate = int(sampling_rate)
|
||||
self.n_fft = int(n_fft)
|
||||
self.win_length = int(win_length)
|
||||
self.hop_length = int(hop_length)
|
||||
self.subsampling_channels = list(subsampling_channels or [num_mel_bins, 512, hidden_size])
|
||||
self.subsampling_depths = list(subsampling_depths or [3, 4, 5])
|
||||
self.conv_depthwise_kernel_size = int(conv_depthwise_kernel_size)
|
||||
self.rotary_embedding_base = int(rotary_embedding_base)
|
||||
self.layer_norm_eps = float(layer_norm_eps)
|
||||
self.subsampling_layer_norm_eps = float(subsampling_layer_norm_eps)
|
||||
self.initializer_range = float(initializer_range)
|
||||
self.variant = str(variant)
|
||||
self.context_seconds = float(context_seconds if context_seconds is not None else (360 if variant == "fs" else 30))
|
||||
self.inputs_to_logits_ratio = self.hop_length * 4
|
||||
self.frame_rate = self.sampling_rate / self.inputs_to_logits_ratio
|
||||
self.minimum_input_samples = self.n_fft // 2 + 1
|
||||
|
||||
if min(self.hidden_size, self.intermediate_size, self.num_hidden_layers, self.num_attention_heads) <= 0:
|
||||
raise ValueError("Encoder dimensions and layer counts must be positive.")
|
||||
if self.hidden_size % self.num_attention_heads or (self.hidden_size // self.num_attention_heads) % 2:
|
||||
raise ValueError("hidden_size must divide into an even head dimension.")
|
||||
if len(self.subsampling_channels) != 3 or len(self.subsampling_depths) != 3:
|
||||
raise ValueError("The subsampler must contain three channel widths and three depths.")
|
||||
if self.subsampling_channels[0] != self.num_mel_bins or self.subsampling_channels[-1] != self.hidden_size:
|
||||
raise ValueError("Subsampling widths must start at num_mel_bins and end at hidden_size.")
|
||||
if min(*self.subsampling_channels, *self.subsampling_depths, self.num_mel_bins, self.sampling_rate, self.hop_length) <= 0:
|
||||
raise ValueError("Frontend dimensions and sampling parameters must be positive.")
|
||||
if self.n_fft < self.win_length or self.win_length <= 0 or self.n_fft % 2:
|
||||
raise ValueError("n_fft must be even and at least win_length > 0.")
|
||||
if self.conv_depthwise_kernel_size <= 0 or self.conv_depthwise_kernel_size % 2 != 1:
|
||||
raise ValueError("conv_depthwise_kernel_size must be positive and odd.")
|
||||
if self.rotary_embedding_base <= 0 or min(self.layer_norm_eps, self.subsampling_layer_norm_eps) <= 0:
|
||||
raise ValueError("Rotary base and normalization epsilons must be positive.")
|
||||
if self.variant not in {"30s", "fs"}:
|
||||
raise ValueError("variant must be '30s' or 'fs'.")
|
||||
|
||||
|
||||
MERT2Config.register_for_auto_class()
|
||||
@@ -0,0 +1,327 @@
|
||||
{
|
||||
"schema_version": 2,
|
||||
"benchmark": "MARBLE",
|
||||
"metric_scale": 1,
|
||||
"display_scale": 100,
|
||||
"models": {
|
||||
"MERT-v2-30s": {
|
||||
"repo_id": "a43992899/MERT-v2-30s",
|
||||
"tasks": [
|
||||
{
|
||||
"dataset": "GTZAN",
|
||||
"task": "Genre classification",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 23,
|
||||
"hidden_state_index": 22
|
||||
},
|
||||
"learning_rate": 0.005,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"file_acc": 0.9172413945198059
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "GTZAN",
|
||||
"task": "Beat tracking",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 21,
|
||||
"hidden_state_index": 20
|
||||
},
|
||||
"learning_rate": 0.001,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"beat_f1": 0.9058816432952881
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "GiantSteps",
|
||||
"task": "Key detection",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 4,
|
||||
"hidden_state_index": 3
|
||||
},
|
||||
"learning_rate": 0.001,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"weighted_score": 0.6697019867549668
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "EmoMusic",
|
||||
"task": "Emotion recognition",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "learned_all_layer_mlp",
|
||||
"num_layers": 24
|
||||
},
|
||||
"learning_rate": 5e-05,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"valence_r2": 0.6322920322418213,
|
||||
"arousal_r2": 0.8000704646110535
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "Chords1217",
|
||||
"task": "Chord recognition",
|
||||
"supplemental": true,
|
||||
"representation": {
|
||||
"mode": "learned_all_layer_mlp",
|
||||
"num_layers": 24
|
||||
},
|
||||
"learning_rate": 0.0001,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"acc": 0.7847631573677063
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "MagnaTagATune",
|
||||
"task": "Tagging",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 22,
|
||||
"hidden_state_index": 21
|
||||
},
|
||||
"learning_rate": 0.001,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"auroc": 0.9190762042999268,
|
||||
"average_precision": 0.4129374325275421
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "MTG-Jamendo",
|
||||
"task": "Genre tagging",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 19,
|
||||
"hidden_state_index": 18
|
||||
},
|
||||
"learning_rate": 0.001,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"auroc": 0.8801141381263733,
|
||||
"average_precision": 0.21221663057804108
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "MTG-Jamendo",
|
||||
"task": "Instrument tagging",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 14,
|
||||
"hidden_state_index": 13
|
||||
},
|
||||
"learning_rate": 0.001,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"auroc": 0.8027445673942566,
|
||||
"average_precision": 0.2288619726896286
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "MTG-Jamendo",
|
||||
"task": "Mood/theme tagging",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 16,
|
||||
"hidden_state_index": 15
|
||||
},
|
||||
"learning_rate": 0.001,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"auroc": 0.7944093942642212,
|
||||
"average_precision": 0.16676680743694305
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "MTG-Jamendo",
|
||||
"task": "Top-50 tagging",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 13,
|
||||
"hidden_state_index": 12
|
||||
},
|
||||
"learning_rate": 0.001,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"auroc": 0.8417603969573975,
|
||||
"average_precision": 0.3217017948627472
|
||||
}
|
||||
}
|
||||
]
|
||||
},
|
||||
"MERT-v2-FullSong": {
|
||||
"repo_id": "a43992899/MERT-v2-FullSong",
|
||||
"tasks": [
|
||||
{
|
||||
"dataset": "GTZAN",
|
||||
"task": "Genre classification",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 24,
|
||||
"hidden_state_index": 23
|
||||
},
|
||||
"learning_rate": 0.0005,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"file_acc": 0.9068965315818787
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "GTZAN",
|
||||
"task": "Beat tracking",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 23,
|
||||
"hidden_state_index": 22
|
||||
},
|
||||
"learning_rate": 0.001,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"beat_f1": 0.9057297706604004
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "GiantSteps",
|
||||
"task": "Key detection",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 23,
|
||||
"hidden_state_index": 22
|
||||
},
|
||||
"learning_rate": 0.001,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"weighted_score": 0.6705298013245033
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "EmoMusic",
|
||||
"task": "Emotion recognition",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 24,
|
||||
"hidden_state_index": 23
|
||||
},
|
||||
"learning_rate": 0.0005,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"valence_r2": 0.6351668834686279,
|
||||
"arousal_r2": 0.7814431190490723
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "Chords1217",
|
||||
"task": "Chord recognition",
|
||||
"supplemental": true,
|
||||
"representation": {
|
||||
"mode": "learned_all_layer_mlp",
|
||||
"num_layers": 24
|
||||
},
|
||||
"learning_rate": 0.0005,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"acc": 0.7779459357261658
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "MagnaTagATune",
|
||||
"task": "Tagging",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 23,
|
||||
"hidden_state_index": 22
|
||||
},
|
||||
"learning_rate": 0.001,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"auroc": 0.9173608422279358,
|
||||
"average_precision": 0.4120141565799713
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "MTG-Jamendo",
|
||||
"task": "Genre tagging",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 16,
|
||||
"hidden_state_index": 15
|
||||
},
|
||||
"learning_rate": 0.001,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"auroc": 0.8797574043273926,
|
||||
"average_precision": 0.20656391978263855
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "MTG-Jamendo",
|
||||
"task": "Instrument tagging",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 12,
|
||||
"hidden_state_index": 11
|
||||
},
|
||||
"learning_rate": 0.001,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"auroc": 0.8026924133300781,
|
||||
"average_precision": 0.2351299524307251
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "MTG-Jamendo",
|
||||
"task": "Mood/theme tagging",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 13,
|
||||
"hidden_state_index": 12
|
||||
},
|
||||
"learning_rate": 0.001,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"auroc": 0.7874246835708618,
|
||||
"average_precision": 0.15743640065193176
|
||||
}
|
||||
},
|
||||
{
|
||||
"dataset": "MTG-Jamendo",
|
||||
"task": "Top-50 tagging",
|
||||
"supplemental": false,
|
||||
"representation": {
|
||||
"mode": "fixed_layer",
|
||||
"layer": 22,
|
||||
"hidden_state_index": 21
|
||||
},
|
||||
"learning_rate": 0.001,
|
||||
"probe_seed": 1234,
|
||||
"metrics": {
|
||||
"auroc": 0.8413298726081848,
|
||||
"average_precision": 0.3162302076816559
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,361 @@
|
||||
"""Standalone MERT2 waveform-to-representation inference."""
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Optional, Tuple
|
||||
|
||||
import torch
|
||||
from torch import nn
|
||||
from torch.nn import functional as F
|
||||
from torchaudio.transforms import AmplitudeToDB, MelScale, Spectrogram
|
||||
from transformers import PreTrainedModel
|
||||
from transformers.utils import ModelOutput
|
||||
|
||||
from .configuration_mert2 import MERT2Config
|
||||
|
||||
|
||||
@dataclass
|
||||
class MERT2ModelOutput(ModelOutput):
|
||||
"""Frame representations and their validity mask.
|
||||
|
||||
``hidden_states`` contains one tensor per Conformer block, without an input
|
||||
embedding. ``feature_attention_mask`` is True at valid output frames.
|
||||
"""
|
||||
|
||||
last_hidden_state: Optional[torch.FloatTensor] = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor, ...]] = None
|
||||
feature_attention_mask: Optional[torch.BoolTensor] = None
|
||||
|
||||
|
||||
class MERT2MelFrontend(nn.Module):
|
||||
"""Power log-mel features with fixed, checkpoint-specific normalization."""
|
||||
|
||||
def __init__(self, config):
|
||||
super().__init__()
|
||||
# Filter construction requires real tensors even during meta loading.
|
||||
with torch.device("cpu"):
|
||||
self.register_buffer("mel_mean", torch.zeros(config.num_mel_bins, dtype=torch.float32))
|
||||
self.register_buffer("mel_std", torch.ones(config.num_mel_bins, dtype=torch.float32))
|
||||
self.spectrogram = Spectrogram(
|
||||
n_fft=config.n_fft,
|
||||
win_length=config.win_length,
|
||||
hop_length=config.hop_length,
|
||||
power=2.0,
|
||||
).float()
|
||||
self.mel_scale = MelScale(
|
||||
n_mels=config.num_mel_bins,
|
||||
sample_rate=config.sampling_rate,
|
||||
n_stft=config.n_fft // 2 + 1,
|
||||
).float()
|
||||
self.amplitude_to_db = AmplitudeToDB(stype="power", top_db=None)
|
||||
|
||||
def _apply(self, fn, recurse=True):
|
||||
# Preserve the original float32 values, including on model.half().
|
||||
saved = [(module, name, value) for module in self.modules() for name, value in module._buffers.items() if value is not None]
|
||||
super()._apply(fn, recurse=recurse)
|
||||
for module, name, original in saved:
|
||||
current = module._buffers[name]
|
||||
if original.is_floating_point():
|
||||
module._buffers[name] = (
|
||||
current.float() if original.is_meta else original.to(device=current.device, dtype=torch.float32)
|
||||
)
|
||||
return self
|
||||
|
||||
@torch.no_grad()
|
||||
def forward(self, waveform):
|
||||
with torch.autocast(device_type=waveform.device.type, enabled=False):
|
||||
spectrum = self.spectrogram(waveform.float())
|
||||
mel = self.amplitude_to_db(self.mel_scale(spectrum))
|
||||
mel = mel[..., :-1].transpose(-1, -2)
|
||||
return (mel - self.mel_mean) / self.mel_std.clamp_min(1e-5)
|
||||
|
||||
|
||||
class Transpose(nn.Module):
|
||||
def forward(self, hidden_states):
|
||||
return hidden_states.transpose(1, 2)
|
||||
|
||||
|
||||
class GlobalResponseNorm(nn.Module):
|
||||
def __init__(self, dim):
|
||||
super().__init__()
|
||||
self.weight = nn.Parameter(torch.zeros(1, 1, dim))
|
||||
self.bias = nn.Parameter(torch.zeros(1, 1, dim))
|
||||
|
||||
def forward(self, hidden_states):
|
||||
magnitude = torch.norm(hidden_states, p=2, dim=1, keepdim=True)
|
||||
normalized = magnitude / (magnitude.mean(dim=-1, keepdim=True) + 1e-6)
|
||||
return self.weight * (hidden_states * normalized) + self.bias + hidden_states
|
||||
|
||||
|
||||
class ConvNextLayer(nn.Module):
|
||||
def __init__(self, dim, eps):
|
||||
super().__init__()
|
||||
self.depthwise_block = nn.Sequential(
|
||||
Transpose(), nn.Conv1d(dim, dim, 7, padding=3, groups=dim), Transpose()
|
||||
)
|
||||
self.pointwise_block = nn.Sequential(
|
||||
nn.LayerNorm(dim, eps=eps),
|
||||
nn.Linear(dim, 4 * dim),
|
||||
nn.GELU(),
|
||||
GlobalResponseNorm(4 * dim),
|
||||
nn.Linear(4 * dim, dim),
|
||||
)
|
||||
|
||||
def forward(self, hidden_states):
|
||||
return hidden_states + self.pointwise_block(self.depthwise_block(hidden_states))
|
||||
|
||||
|
||||
class ConvNextBlock(nn.Module):
|
||||
def __init__(self, in_channels, out_channels, stride, depth, eps):
|
||||
super().__init__()
|
||||
self.resampling_layer = (
|
||||
nn.Sequential(
|
||||
nn.LayerNorm(in_channels, eps=eps),
|
||||
Transpose(),
|
||||
nn.Conv1d(in_channels, out_channels, 2, stride=stride),
|
||||
Transpose(),
|
||||
)
|
||||
if in_channels != out_channels or stride > 1 else nn.Identity()
|
||||
)
|
||||
self.convnext_layers = nn.Sequential(*[ConvNextLayer(out_channels, eps) for _ in range(depth)])
|
||||
|
||||
def forward(self, hidden_states):
|
||||
return self.convnext_layers(self.resampling_layer(hidden_states))
|
||||
|
||||
|
||||
class RotaryEmbedding(nn.Module):
|
||||
def __init__(self, config):
|
||||
super().__init__()
|
||||
self.head_dim = config.hidden_size // config.num_attention_heads
|
||||
self.base = config.rotary_embedding_base
|
||||
with torch.device("cpu"):
|
||||
inverse_frequency = 1.0 / (
|
||||
self.base ** (torch.arange(0, self.head_dim, 2, dtype=torch.float32) / self.head_dim)
|
||||
)
|
||||
self.register_buffer("inv_freq", inverse_frequency, persistent=False)
|
||||
self._sequence_length = 0
|
||||
self._cache_device = None
|
||||
self._cos = None
|
||||
self._sin = None
|
||||
|
||||
def _apply(self, fn, recurse=True):
|
||||
original = self.inv_freq
|
||||
super()._apply(fn, recurse=recurse)
|
||||
self.inv_freq = original.to(device=self.inv_freq.device, dtype=torch.float32)
|
||||
return self
|
||||
|
||||
def forward(self, hidden_states):
|
||||
length = hidden_states.shape[1]
|
||||
if self._cos is None or self._cache_device != hidden_states.device or length > self._sequence_length:
|
||||
positions = torch.arange(length, device=hidden_states.device, dtype=self.inv_freq.dtype)
|
||||
frequencies = torch.einsum("i,j->ij", positions, self.inv_freq)
|
||||
angles = torch.cat((frequencies, frequencies), dim=-1)
|
||||
self._cos = angles.cos()[:, None, None, :]
|
||||
self._sin = angles.sin()[:, None, None, :]
|
||||
self._sequence_length = length
|
||||
self._cache_device = hidden_states.device
|
||||
return (
|
||||
self._cos[:length].to(hidden_states.dtype).permute(1, 0, 2, 3),
|
||||
self._sin[:length].to(hidden_states.dtype).permute(1, 0, 2, 3),
|
||||
)
|
||||
|
||||
|
||||
def rotate_half(value):
|
||||
first, second = value.chunk(2, dim=-1)
|
||||
return torch.cat((-second, first), dim=-1)
|
||||
|
||||
|
||||
class SelfAttention(nn.Module):
|
||||
def __init__(self, config):
|
||||
super().__init__()
|
||||
self.config = config
|
||||
self.num_heads = config.num_attention_heads
|
||||
self.head_dim = config.hidden_size // self.num_heads
|
||||
self.query_proj = nn.Linear(config.hidden_size, config.hidden_size)
|
||||
self.key_proj = nn.Linear(config.hidden_size, config.hidden_size)
|
||||
self.value_proj = nn.Linear(config.hidden_size, config.hidden_size)
|
||||
self.out_proj = nn.Linear(config.hidden_size, config.hidden_size)
|
||||
|
||||
def forward(self, hidden_states, position_embeddings):
|
||||
batch, time, width = hidden_states.shape
|
||||
shape = (batch, time, self.num_heads, self.head_dim)
|
||||
query = self.query_proj(hidden_states).reshape(shape)
|
||||
key = self.key_proj(hidden_states).reshape(shape)
|
||||
value = self.value_proj(hidden_states).reshape(shape)
|
||||
cos, sin = position_embeddings
|
||||
query = query * cos + rotate_half(query) * sin
|
||||
key = key * cos + rotate_half(key) * sin
|
||||
if self.config._attn_implementation == "flash_attention_2":
|
||||
if query.device.type != "cuda" or query.dtype not in (torch.float16, torch.bfloat16):
|
||||
raise ValueError("flash_attention_2 requires CUDA and float16/bfloat16 activations; use autocast or load a reduced-precision model.")
|
||||
try:
|
||||
from flash_attn import flash_attn_func
|
||||
except ImportError as error:
|
||||
raise ImportError("Install flash-attn to use flash_attention_2, or select attn_implementation='sdpa'.") from error
|
||||
attended = flash_attn_func(query.contiguous(), key.contiguous(), value.contiguous(), dropout_p=0.0, causal=False)
|
||||
else:
|
||||
attended = F.scaled_dot_product_attention(
|
||||
query.transpose(1, 2), key.transpose(1, 2), value.transpose(1, 2), dropout_p=0.0, is_causal=False
|
||||
).transpose(1, 2)
|
||||
return self.out_proj(attended.reshape(batch, time, width))
|
||||
|
||||
|
||||
class FeedForward(nn.Module):
|
||||
def __init__(self, config):
|
||||
super().__init__()
|
||||
self.w_1 = nn.Linear(config.hidden_size, config.intermediate_size)
|
||||
self.w_2 = nn.Linear(config.intermediate_size, config.hidden_size)
|
||||
|
||||
def forward(self, hidden_states):
|
||||
return self.w_2(F.gelu(self.w_1(hidden_states)))
|
||||
|
||||
|
||||
class ConvolutionModule(nn.Module):
|
||||
def __init__(self, config):
|
||||
super().__init__()
|
||||
width = config.hidden_size
|
||||
kernel = config.conv_depthwise_kernel_size
|
||||
self.layer_norm = nn.LayerNorm(width, eps=config.layer_norm_eps)
|
||||
self.conv_block = nn.Sequential(
|
||||
Transpose(),
|
||||
nn.Conv1d(width, 2 * width, 1, bias=False),
|
||||
nn.GLU(dim=1),
|
||||
nn.Conv1d(width, width, kernel, padding=(kernel - 1) // 2, groups=width, bias=False),
|
||||
nn.Sequential(Transpose(), nn.LayerNorm(width, eps=config.layer_norm_eps), Transpose()),
|
||||
nn.GELU(),
|
||||
nn.Conv1d(width, width, 1, bias=False),
|
||||
Transpose(),
|
||||
)
|
||||
|
||||
def forward(self, hidden_states):
|
||||
return self.conv_block(self.layer_norm(hidden_states))
|
||||
|
||||
|
||||
class ConformerBlock(nn.Module):
|
||||
def __init__(self, config):
|
||||
super().__init__()
|
||||
width, eps = config.hidden_size, config.layer_norm_eps
|
||||
self.ffn1_layer_norm = nn.LayerNorm(width, eps=eps)
|
||||
self.ffn1 = FeedForward(config)
|
||||
self.attn_layer_norm = nn.LayerNorm(width, eps=eps)
|
||||
self.attn = SelfAttention(config)
|
||||
self.conv_module = ConvolutionModule(config)
|
||||
self.ffn2_layer_norm = nn.LayerNorm(width, eps=eps)
|
||||
self.ffn2 = FeedForward(config)
|
||||
self.final_layer_norm = nn.LayerNorm(width, eps=eps)
|
||||
|
||||
def forward(self, hidden_states, position_embeddings):
|
||||
hidden_states = hidden_states + 0.5 * self.ffn1(self.ffn1_layer_norm(hidden_states))
|
||||
hidden_states = self.attn(self.attn_layer_norm(hidden_states), position_embeddings) + hidden_states
|
||||
hidden_states = self.conv_module(hidden_states) + hidden_states
|
||||
hidden_states = hidden_states + 0.5 * self.ffn2(self.ffn2_layer_norm(hidden_states))
|
||||
return self.final_layer_norm(hidden_states)
|
||||
|
||||
|
||||
class MERT2Model(PreTrainedModel):
|
||||
"""Encode mono waveforms into 25 Hz MERT2 frame representations.
|
||||
|
||||
Inputs are floating-point mono waveforms sampled at ``config.sampling_rate``.
|
||||
Do not standardize their amplitude. An optional waveform attention mask must
|
||||
contain a prefix of ones followed by zeros. Without a mask, every input
|
||||
sample, including any caller-provided silence or padding, is processed.
|
||||
"""
|
||||
|
||||
config_class = MERT2Config
|
||||
base_model_prefix = ""
|
||||
main_input_name = "input_values"
|
||||
_supports_sdpa = True
|
||||
_supports_flash_attn_2 = True
|
||||
_no_split_modules = ["ConvNextBlock", "ConformerBlock"]
|
||||
|
||||
def __init__(self, config):
|
||||
super().__init__(config)
|
||||
if config._attn_implementation not in {"sdpa", "flash_attention_2"}:
|
||||
raise ValueError("MERT2 supports attn_implementation='sdpa' or 'flash_attention_2'.")
|
||||
self.feature_extractor = MERT2MelFrontend(config)
|
||||
channels = [config.num_mel_bins] + config.subsampling_channels
|
||||
self.subsampling_module = nn.Sequential(*[
|
||||
ConvNextBlock(channels[i], channels[i + 1], (1, 2, 2)[i], config.subsampling_depths[i], config.subsampling_layer_norm_eps)
|
||||
for i in range(3)
|
||||
])
|
||||
self.layers = nn.ModuleList([ConformerBlock(config) for _ in range(config.num_hidden_layers)])
|
||||
self.embed_positions = RotaryEmbedding(config)
|
||||
self.post_init()
|
||||
|
||||
def _init_weights(self, module):
|
||||
if isinstance(module, (nn.Linear, nn.Conv1d)):
|
||||
nn.init.normal_(module.weight, mean=0.0, std=self.config.initializer_range)
|
||||
if module.bias is not None:
|
||||
nn.init.zeros_(module.bias)
|
||||
elif isinstance(module, nn.LayerNorm):
|
||||
nn.init.ones_(module.weight)
|
||||
nn.init.zeros_(module.bias)
|
||||
|
||||
def _get_feat_extract_output_lengths(self, input_lengths):
|
||||
return input_lengths // self.config.inputs_to_logits_ratio
|
||||
|
||||
def _encode(self, input_values, output_hidden_states):
|
||||
mel = self.feature_extractor(input_values)
|
||||
input_dtype = self.subsampling_module[0].convnext_layers[0].depthwise_block[1].weight.dtype
|
||||
hidden = self.subsampling_module(mel.to(dtype=input_dtype))
|
||||
positions = self.embed_positions(hidden)
|
||||
states = [] if output_hidden_states else None
|
||||
for layer in self.layers:
|
||||
hidden = layer(hidden, positions)
|
||||
if states is not None:
|
||||
states.append(hidden)
|
||||
return hidden, tuple(states) if states is not None else None
|
||||
|
||||
def forward(self, input_values, attention_mask=None, output_hidden_states=None, return_dict=None):
|
||||
"""Return final frames, optionally all block states, and a frame mask.
|
||||
|
||||
Different valid waveform lengths are encoded separately so convolution
|
||||
and global normalization never incorporate another sample's padding.
|
||||
Outputs are zero-padded to the longest valid feature sequence.
|
||||
"""
|
||||
output_hidden_states = self.config.output_hidden_states if output_hidden_states is None else output_hidden_states
|
||||
return_dict = self.config.use_return_dict if return_dict is None else return_dict
|
||||
if input_values.ndim != 2 or not input_values.is_floating_point() or input_values.shape[0] == 0:
|
||||
raise ValueError("input_values must be a nonempty floating-point tensor of shape [batch, samples].")
|
||||
if input_values.shape[1] < self.config.minimum_input_samples:
|
||||
raise ValueError(f"Each waveform must contain at least {self.config.minimum_input_samples} samples.")
|
||||
if not torch.isfinite(input_values).all():
|
||||
raise ValueError("input_values must contain only finite samples.")
|
||||
batch, samples = input_values.shape
|
||||
if attention_mask is None:
|
||||
hidden, states = self._encode(input_values, output_hidden_states)
|
||||
feature_mask = torch.ones(hidden.shape[:2], dtype=torch.bool, device=hidden.device)
|
||||
else:
|
||||
if attention_mask.shape != input_values.shape:
|
||||
raise ValueError("attention_mask must have the same shape as input_values.")
|
||||
attention_mask = attention_mask.to(device=input_values.device)
|
||||
if not ((attention_mask == 0) | (attention_mask == 1)).all():
|
||||
raise ValueError("attention_mask must contain only zeros and ones.")
|
||||
mask = attention_mask.bool()
|
||||
lengths = mask.sum(dim=1)
|
||||
expected = torch.arange(samples, device=input_values.device)[None, :] < lengths[:, None]
|
||||
if not torch.equal(mask, expected):
|
||||
raise ValueError("attention_mask must be right padded: valid samples followed by padding.")
|
||||
if (lengths < self.config.minimum_input_samples).any():
|
||||
raise ValueError(f"Each waveform must contain at least {self.config.minimum_input_samples} valid samples.")
|
||||
feature_lengths = self._get_feat_extract_output_lengths(lengths)
|
||||
maximum = int(feature_lengths.max().item())
|
||||
feature_mask = torch.arange(maximum, device=input_values.device)[None, :] < feature_lengths[:, None]
|
||||
hidden, state_values = None, None
|
||||
for length in torch.unique(lengths, sorted=True).tolist():
|
||||
indices = torch.where(lengths == length)[0]
|
||||
group_hidden, group_states = self._encode(input_values.index_select(0, indices)[:, :length], output_hidden_states)
|
||||
padding = (0, 0, 0, maximum - group_hidden.shape[1])
|
||||
if hidden is None:
|
||||
hidden = group_hidden.new_zeros(batch, maximum, self.config.hidden_size)
|
||||
if output_hidden_states:
|
||||
state_values = [torch.zeros_like(hidden) for _ in self.layers]
|
||||
hidden = hidden.index_copy(0, indices, F.pad(group_hidden, padding))
|
||||
if state_values is not None:
|
||||
for i, value in enumerate(group_states):
|
||||
state_values[i] = state_values[i].index_copy(0, indices, F.pad(value, padding))
|
||||
states = tuple(state_values) if state_values is not None else None
|
||||
output = MERT2ModelOutput(last_hidden_state=hidden, hidden_states=states, feature_attention_mask=feature_mask)
|
||||
return output if return_dict else output.to_tuple()
|
||||
|
||||
|
||||
MERT2Model.register_for_auto_class("AutoModel")
|
||||
@@ -0,0 +1,9 @@
|
||||
{
|
||||
"feature_extractor_type": "Wav2Vec2FeatureExtractor",
|
||||
"feature_size": 1,
|
||||
"sampling_rate": 24000,
|
||||
"padding_value": 0.0,
|
||||
"padding_side": "right",
|
||||
"return_attention_mask": true,
|
||||
"do_normalize": false
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
{
|
||||
"format": "safetensors",
|
||||
"dtype": "float32",
|
||||
"parameters": 632295808,
|
||||
"state_tensors": 876,
|
||||
"filename": "model.safetensors",
|
||||
"sha256": "e6dd2ab187d6dd62b6521cd7d8f932e237acf0c5757745a7232082e28391350d",
|
||||
"bytes": 2529812848
|
||||
}
|
||||
@@ -2,9 +2,7 @@
|
||||
[YuE](https://github.com/multimodal-art-projection/YuE) is a groundbreaking series of open-source foundation models designed for music generation, specifically for transforming lyrics into full songs (lyrics2song). you can use it in comfyUI
|
||||
|
||||
# Update
|
||||
* install quantization moudle (mmgp and exllamav2) if you need ;
|
||||
* mmgp and exllamav2 的量化库在菜单选中的时候才会需要,避免有些人安装不上;
|
||||
|
||||
* update to yue2 / 简单更新到yue2版本;
|
||||
|
||||
# 1. Installation
|
||||
|
||||
@@ -18,106 +16,50 @@ git clone https://github.com/smthemex/ComfyUI_YuE.git
|
||||
```
|
||||
pip install -r requirements.txt
|
||||
```
|
||||
* triton is not necessary,I don't test it,you can try
|
||||
* triton库不是必须的,不过节点里的加速方法用了或许更快,不保真
|
||||
* use -no-deps if descript-audiotools need high torch 如果descript-audiotools安装需要高版本的torch ,使用pip install -no-deps descript-audiotools
|
||||
|
||||
* if use mmgp
|
||||
```
|
||||
pip install mmgp
|
||||
```
|
||||
* if use exllamav2
|
||||
```
|
||||
pip install exllamav2
|
||||
```
|
||||
- or
|
||||
```
|
||||
git clone https://github.com/turboderp/exllamav2
|
||||
cd exllamav2
|
||||
pip install -r requirements.txt
|
||||
pip install .
|
||||
```
|
||||
|
||||
# 3.models
|
||||
* 3.1 download from [here](https://huggingface.co/m-a-p/xcodec_mini_infer/tree/main/final_ckpt) and [here](https://huggingface.co/m-a-p/YuE-upsampler/tree/main) path like below,模型路径结构如下
|
||||
* 3.1 base infer [3B dit](https://huggingface.co/m-a-p/YuE2-3B) and [vae](https://huggingface.co/m-a-p/YuE2-Vae) ,only model.safetensors and vae.safetensors is needed /只需要模型文件;
|
||||
* 3.1 cover mode [SheetSage2](https://huggingface.co/m-a-p/SheetSage2) and [MERT2](https://huggingface.co/m-a-p/MERT-v2-FullSong) ,only model.safetensors and vae.safetensors is needed /只需要模型文件;
|
||||
|
||||
* base infer:
|
||||
```
|
||||
-- ComfyUI/models/yue
|
||||
├── ckpt_00360000.pth
|
||||
├── decoder_131000.pth
|
||||
├── decoder_151000.pth
|
||||
-- ComfyUI/models/diffusion_models
|
||||
├── yue2_model.safetensors # rename from model.safetensor or not
|
||||
-- ComfyUI/models/vae
|
||||
├── yue2_vae.safetensors # rename from model.safetensor or not
|
||||
```
|
||||
* 3.2 download from [here](https://huggingface.co/m-a-p/xcodec_mini_infer/tree/main/semantic_ckpts/hf_1_325000), path like below,模型路径结构如下
|
||||
* if use cover:
|
||||
```
|
||||
-- ComfyUI/custom_nodes/ComfyUI_YuE/inference/xcodec_mini_infer/semantic_ckpts/hf_1_325000/
|
||||
├── pytorch_model.bin
|
||||
-- ComfyUI/models/clip
|
||||
├── mert2_model.safetensors # rename from model.safetensor or not
|
||||
├── sheetsage2.safetensors # rename from model.safetensor or not
|
||||
```
|
||||
|
||||
* 3.3 if your GPU is 4090 or 5090 or best,just use repo below,Of course, you can also just fill out the repo,如果你是4090以上,可以用fp16,只填repo会自动下载,以下是离线版;
|
||||
* 3.3.1 english [YuE-s1-7B-anneal-en-cot](https://huggingface.co/m-a-p/YuE-s1-7B-anneal-en-cot) or[YuE-s1-7B-anneal-en-icl](https://huggingface.co/m-a-p/YuE-s1-7B-anneal-en-icl) or chinese [YuE-s1-7B-anneal-zh-cot](https://huggingface.co/m-a-p/YuE-s1-7B-anneal-zh-cot),or other ,path like below,模型路径结构如下
|
||||
```
|
||||
-- anypath/YuE-s1-7B-anneal-en-icl # 11.5G
|
||||
├── config.json
|
||||
├── generation_config.json
|
||||
├── model.safetensors.index.json
|
||||
├── tokenizer.model
|
||||
├── model-00001-of-00003.safetensors
|
||||
├── model-00002-of-00003.safetensors
|
||||
├── model-00003-of-00003.safetensors
|
||||
```
|
||||
* 3.3.2 just one [YuE-s2-1B-general](https://huggingface.co/m-a-p/YuE-s2-1B-general/tree/main) path like below,模型路径结构如下
|
||||
```
|
||||
-- anypath/YuE-s2-1B-general # 3.65G
|
||||
├── config.json
|
||||
├── generation_config.json
|
||||
├── model.safetensors
|
||||
├── tokenizer.model
|
||||
```
|
||||
* 3.4 if your VRAM=<16G,you can use 3.3 origin repo and [int8 or int4](https://github.com/alisson-anjos/YuE-Interface),[exllamav2](https://github.com/sgsdxzy/YuE-exllamav2),[deepbeepmeep](https://github.com/deepbeepmeep/YuEGP) Three quantitative acceleration methods and models,Thanks to these open-source projects ,below is list.如果显存在16G以下,可以用社区的三种量化加速方法和模型,列表如下:
|
||||
- [exllamav2](https://huggingface.co/Doctor-Shotgun/YuE-s1-7B-anneal-en-cot-exl2) use fp16 Q8,Q6... [@sgsdxzy](https://github.com/sgsdxzy)
|
||||
- [int8](https://huggingface.co/Alissonerdx/YuE-s1-7B-anneal-en-cot-int8) bitsandbytes [@alisson-anjos](https://github.com/alisson-anjos)
|
||||
- [exllamav2](https://huggingface.co/collections/Alissonerdx/yue-models-exllamav2-67a539be76b5225ebda95323) use fp16 Q8,Q6...[@alisson-anjos](https://github.com/alisson-anjos)
|
||||
- [deepbeepmeep](https://github.com/deepbeepmeep/YuEGP) use fp16,int8... [@deepbeepmeep](https://github.com/deepbeepmeep)
|
||||
3.4.1 path like below,模型路径结构如下
|
||||
```
|
||||
-- anypath/YuE-s1-7B-anneal-en-cot-exl2-8.0bpw
|
||||
├── config.json
|
||||
....
|
||||
-- anypath/YuE-s2-1B-general-exl2-8.0bpw
|
||||
├── config.json
|
||||
....
|
||||
```
|
||||
# 4.Use Tips
|
||||
* if VRAM>=24G,use origin repo,'quantization_model'choice 'fp16',keep 'use_mmgp' false,prompt_end_time is sampler length,try 30s first.(best quality);
|
||||
* if VRAM<=16G,use origin repo,'quantization_model'choice 'fp16',keep 'use_mmgp' ture,mmgp_profile choice 2,prompt_end_time is sampler length,try 30s first (best quality,slowly);
|
||||
* if VRAM<=16G,use int8 repo,'quantization_model'choice 'int8',keep 'use_mmgp' false,prompt_end_time is sampler length,try 30s first.(nromal quality,very slowly 6716s,don't try it)
|
||||
* if VRAM<=16G,use exllamav2 Q8 repo,'quantization_model'choice 'exllamav2',keep 'use_mmgp' false,exllamav2_cache_mode choice Q8,prompt_end_time is sampler length,try 30s first.(nromal quality,very fast)
|
||||
* if VRAM>=16G,use origin repo,'quantization_model'choice 'exllamav2',keep 'use_mmgp' false,exllamav2_cache_mode choice fp16,prompt_end_time is sampler length,try 30s first.(best quality, fast and maybe OOM if VRAM<24)
|
||||
* 显存大于24G,用原生的repo,quantization_model选fp16,关闭use_mmgp,prompt_end_time就是渲染时长先设置为30秒测试(普通玩家的最佳效果)
|
||||
* 显存小于等于16G,用原生的repo,quantization_model选fp16,开启use_mmgp,prompt_end_time就是渲染时长先设置为30秒测试(效果好,但是慢,需要大内存)
|
||||
* 显存小于等于16G,用int8的repo,'quantization_model'选int8,关闭use_mmgp,prompt_end_time就是渲染时长先设置为30秒测试(效果还行,速度奇慢6716s,不要尝试)
|
||||
* 显存小于等于16G,用exllamav2的Q8 repo,'quantization_model'选exllamav2,关闭use_mmgp,exllamav2_cache_mode选择Q8,prompt_end_time就是渲染时长先设置为30秒测试(效果一般,速度非常快)
|
||||
* 显存大于等于16G,用原生的repo,'quantization_model'选exllamav2,关闭use_mmgp,exllamav2_cache_mode选择fp16,prompt_end_time就是渲染时长先设置为30秒测试(效果和速度未测试)
|
||||
* if "use_dual_tracks_prompt" set ture ,will Instrumental from file 'pop.00001.Instrumental.mp3' or if set "use_audio_prompt" ture and "use_dual_tracks_prompt" False will Instrumental from file "pop.00001.mp3",you can replace it ,but keep name in same.
|
||||
* 开启use_dual_tracks_prompt 会借鉴'pop.00001.Instrumental.mp3',开启"use_audio_prompt"并关闭"use_dual_tracks_prompt" 会借鉴 "pop.00001.mp3",你可以用其他歌曲替换掉这两个,但是逐一保持命名不要动,因为在代码里写死了。我迟点再改。
|
||||
* use_mmgp model have 5 choice(mmgp_profile) ,3-4 will quantization model auto,mmgp模式在mmgp_profile有5个选项,大于等于3会自动量化,数字越大需要内存越大,需要显存越小,但是越慢.
|
||||
# 4.Example
|
||||

|
||||
|
||||
# 5.Prompt Engineering Guide 提示词和歌词撰写官方指导
|
||||
* look [Here](https://github.com/multimodal-art-projection/YuE?tab=readme-ov-file#prompt-engineering-guide) to find how to edit your Genre Tagging Prompt and Lyrics Prompt,链接直达官方的歌词和提示词指导。
|
||||
* use [top_200_tags.json](https://github.com/smthemex/ComfyUI_YuE/blob/main/top_200_tags.json) to edit your prompts,查看top_200_tags.json来撰写你的歌词提示
|
||||
|
||||
# 6.Example
|
||||
* use fp16 and use_mmgp
|
||||

|
||||
* use int8 bitsandbytes
|
||||

|
||||
|
||||
# 7.Citation
|
||||
# 5.Citation
|
||||
```
|
||||
@misc{yuan2025yue,
|
||||
title={YuE: Open Music Foundation Models for Full-Song Generation},
|
||||
author={Ruibin Yuan and Hanfeng Lin and Shawn Guo and Ge Zhang and Jiahao Pan and Yongyi Zang and Haohe Liu and Xingjian Du and Xeron Du and Zhen Ye and Tianyu Zheng and Yinghao Ma and Minghao Liu and Lijun Yu and Zeyue Tian and Ziya Zhou and Liumeng Xue and Xingwei Qu and Yizhi Li and Tianhao Shen and Ziyang Ma and Shangda Wu and Jun Zhan and Chunhui Wang and Yatian Wang and Xiaohuan Zhou and Xiaowei Chi and Xinyue Zhang and Zhenzhu Yang and Yiming Liang and Xiangzhou Wang and Shansong Liu and Lingrui Mei and Peng Li and Yong Chen and Chenghua Lin and Xie Chen and Gus Xia and Zhaoxiang Zhang and Chao Zhang and Wenhu Chen and Xinyu Zhou and Xipeng Qiu and Roger Dannenberg and Jiaheng Liu and Jian Yang and Stephen Huang and Wei Xue and Xu Tan and Yike Guo},
|
||||
howpublished={\url{https://github.com/multimodal-art-projection/YuE}},
|
||||
year={2025},
|
||||
note={GitHub repository}
|
||||
@article{li2023mert,
|
||||
title = {{MERT}: Acoustic Music Understanding Model with Large-Scale Self-supervised Training},
|
||||
author = {Li, Yizhi and Yuan, Ruibin and Zhang, Ge and Ma, Yinghao and Chen, Xingran and Yin, Hanzhi and Xiao, Chenghao and Lin, Chenghua and Ragni, Anton and Benetos, Emmanouil and Gyenge, Norbert and Dannenberg, Roger and Liu, Ruibo and Chen, Wenhu and Xia, Gus and Shi, Yemin and Huang, Wenhao and Wang, Zili and Guo, Yike and Fu, Jie},
|
||||
journal = {arXiv preprint arXiv:2306.00107},
|
||||
year = {2023},
|
||||
eprint = {2306.00107},
|
||||
archivePrefix = {arXiv},
|
||||
url = {https://arxiv.org/abs/2306.00107}
|
||||
}
|
||||
|
||||
@article{yuan2025yue,
|
||||
title = {{YuE}: Scaling Open Foundation Models for Long-Form Music Generation},
|
||||
author = {Yuan, Ruibin and Lin, Hanfeng and Guo, Shuyue and Zhang, Ge and Pan, Jiahao and Zang, Yongyi and Liu, Haohe and Liang, Yiming and Ma, Wenye and Du, Xingjian and Du, Xinrun and Ye, Zhen and Zheng, Tianyu and Jiang, Zhengxuan and Ma, Yinghao and Liu, Minghao and Tian, Zeyue and Zhou, Ziya and Xue, Liumeng and Qu, Xingwei and Li, Yizhi and Wu, Shangda and Shen, Tianhao and Ma, Ziyang and Zhan, Jun and Wang, Chunhui and Wang, Yatian and Chi, Xiaowei and Zhang, Xinyue and Yang, Zhenzhu and Wang, Xiangzhou and Liu, Shansong and Mei, Lingrui and Li, Peng and Wang, Junjie and Yu, Jianwei and Pang, Guojian and Li, Xu and Wang, Zihao and Zhou, Xiaohuan and Yu, Lijun and Benetos, Emmanouil and Chen, Yong and Lin, Chenghua and Chen, Xie and Xia, Gus and Zhang, Zhaoxiang and Zhang, Chao and Chen, Wenhu and Zhou, Xinyu and Qiu, Xipeng and Dannenberg, Roger and Liu, Jiaheng and Yang, Jian and Huang, Wenhao and Xue, Wei and Tan, Xu and Guo, Yike},
|
||||
journal = {arXiv preprint arXiv:2503.08638},
|
||||
year = {2025},
|
||||
eprint = {2503.08638},
|
||||
archivePrefix = {arXiv},
|
||||
url = {https://arxiv.org/abs/2503.08638}
|
||||
}
|
||||
|
||||
```
|
||||
|
||||
|
||||
@@ -0,0 +1,195 @@
|
||||
---
|
||||
license: cc-by-nc-4.0
|
||||
library_name: transformers
|
||||
base_model: m-a-p/MERT-v2-FullSong
|
||||
base_model_relation: adapter
|
||||
tags:
|
||||
- audio
|
||||
- music
|
||||
- music-transcription
|
||||
- midi
|
||||
- abc-notation
|
||||
- custom_code
|
||||
---
|
||||
<h1 align="center">🤗 SheetSage2</h1>
|
||||
<p align="center"><strong>Music audio to editable scores</strong></p>
|
||||
<p align="center">Melody · Chords · Beats · Key · Structure</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="https://map-yue2.github.io/">🎵 YuE2 project</a>
|
||||
·
|
||||
<a href="#quick-start">🚀 Quick start</a>
|
||||
·
|
||||
<a href="#benchmarks">📊 Benchmarks</a>
|
||||
·
|
||||
<a href="#citation">📚 Citation</a>
|
||||
</p>
|
||||
<p align="center">
|
||||
<a href="https://huggingface.co/m-a-p/YuE2-3B"><img alt="🤗 YuE2-3B" src="https://img.shields.io/badge/YuE2--3B-374151?logo=huggingface&logoColor=FFD21E" height="20" /></a>
|
||||
|
||||
<a href="https://huggingface.co/m-a-p/YuE2-Vae"><img alt="🤗 YuE2-Vae" src="https://img.shields.io/badge/YuE2--Vae-374151?logo=huggingface&logoColor=FFD21E" height="20" /></a>
|
||||
|
||||
<a href="https://huggingface.co/m-a-p/YuE2-Vae-legacy"><img alt="🤗 YuE2-Vae-legacy" src="https://img.shields.io/badge/YuE2--Vae--legacy-374151?logo=huggingface&logoColor=FFD21E" height="20" /></a>
|
||||
|
||||
<a href="https://huggingface.co/m-a-p/MERT-v2-30s"><img alt="🤗 MERT-v2-30s" src="https://img.shields.io/badge/MERT--v2--30s-374151?logo=huggingface&logoColor=FFD21E" height="20" /></a>
|
||||
|
||||
<a href="https://huggingface.co/m-a-p/MERT-v2-FullSong"><img alt="🤗 MERT-v2-FullSong" src="https://img.shields.io/badge/MERT--v2--FullSong-374151?logo=huggingface&logoColor=FFD21E" height="20" /></a>
|
||||
|
||||
<a href="https://huggingface.co/datasets/m-a-p/WildSongBench"><img alt="🤗 WildSongBench" src="https://img.shields.io/badge/WildSongBench-374151?logo=huggingface&logoColor=FFD21E" height="20" /></a>
|
||||
|
||||
<a href="https://huggingface.co/m-a-p/SheetSage2"><img alt="SheetSage2" src="https://img.shields.io/badge/SheetSage2-374151?logo=huggingface&logoColor=FFD21E" height="20" /></a>
|
||||
</p>
|
||||
|
||||
**SheetSage2 turns music recordings into lead sheets and timed musical annotations.** Transcribe a complete song, edit its ABC or MIDI, render a piano preview, or extract embeddings and token predictions for your own tools.
|
||||
|
||||
Built on [MERT-v2-FullSong](https://huggingface.co/m-a-p/MERT-v2-FullSong), with adapters that merge automatically when you load the model.
|
||||
|
||||

|
||||
|
||||
<a id="quick-start"></a>
|
||||
|
||||
## 🚀 Quick start
|
||||
|
||||
Use Python 3.10 or 3.11 and FFmpeg 6.1 with its shared libraries. Sign in with access to this repository and its MERT-v2 parent:
|
||||
|
||||
```bash
|
||||
python -m pip install huggingface-hub==0.36.0
|
||||
huggingface-cli download m-a-p/SheetSage2 --local-dir SheetSage2
|
||||
cd SheetSage2
|
||||
python -m pip install torch==2.8.0 torchaudio==2.8.0 --index-url https://download.pytorch.org/whl/cu126
|
||||
python -m pip install -r requirements.txt
|
||||
```
|
||||
|
||||
```python
|
||||
import torch
|
||||
from transformers import AutoModel
|
||||
|
||||
model = AutoModel.from_pretrained(
|
||||
"m-a-p/SheetSage2", trust_remote_code=True,
|
||||
).eval().to("cuda" if torch.cuda.is_available() else "cpu")
|
||||
|
||||
result = model.transcribe("song.mp3", output_dir="output")
|
||||
```
|
||||
|
||||
Files are decoded, mixed to mono and resampled automatically. Long songs use overlapping windows.
|
||||
|
||||
| Output | File |
|
||||
|---|---|
|
||||
| Editable score | `score.abc` |
|
||||
| Melody and chord accompaniment | `transcription.mid` |
|
||||
| Separate melodies and chords | `melody_vocal.mid`, `melody_instrumental.mid`, `chords.mid` |
|
||||
| Timed annotations | `events.json`, `*.lab` |
|
||||
|
||||
Omit `output_dir` to keep results in memory. Pass a Tensor or NumPy waveform with its sample rate:
|
||||
|
||||
```python
|
||||
# waveform: [samples] or [channels, samples]
|
||||
result = model.transcribe(waveform, sampling_rate=24000)
|
||||
abc = result["abc"] # str
|
||||
midi = result["midi"] # bytes
|
||||
events = result["events"] # list of timed events
|
||||
```
|
||||
|
||||
Paths, encoded audio bytes and binary streams are also accepted. `result["midis"]` contains the separate MIDI parts; `result["labs"]` contains annotation text. Request `export_logits=True`, `export_scores=True` or `export_embeddings=True` for CPU tensors in `result["tensors"]`, grouped by window; add `output_hidden_states=True` for all 24 MERT layers. With `output_dir`, these tensors are saved as safetensors instead. Low-level `forward()` returns logits and optional hidden states; `generate()` returns symbolic tokens.
|
||||
|
||||
Save a self-contained model for offline use:
|
||||
|
||||
```python
|
||||
model.save_pretrained("sheetsage2-local")
|
||||
model = AutoModel.from_pretrained(
|
||||
"sheetsage2-local", trust_remote_code=True, local_files_only=True,
|
||||
)
|
||||
```
|
||||
|
||||
### Melody-only ABC for covers
|
||||
|
||||
Set `melody_only=True` to retain both the `Vocal` and `Ins` melodies while omitting chord symbols from the ABC and chord accompaniment from playback/combined MIDI. Raw predicted annotations remain available; the default full transcription is unchanged.
|
||||
|
||||
```python
|
||||
result = model.transcribe("song.mp3", output_dir="cover-score", melody_only=True)
|
||||
abc = result["abc"] # Also saved as cover-score/score.abc.
|
||||
```
|
||||
|
||||
```bash
|
||||
python infer.py song.mp3 --output cover-score --melody-only
|
||||
```
|
||||
|
||||
Review the transcription, then pass `result["abc"]` or the saved `cover-score/score.abc` to [YuE2](https://huggingface.co/m-a-p/YuE2-3B) as `abc`, with `cot="melody"` and your target `style` and `lyrics`. If the requested ABC cannot be produced, Python raises an error (partial results are available as `error.result`) and the CLI exits with a nonzero status.
|
||||
|
||||
### Command line and rendering
|
||||
|
||||
```bash
|
||||
python infer.py song.mp3 --output output
|
||||
|
||||
# Optional piano audio and printable sheet music:
|
||||
python setup_render.py
|
||||
python infer.py song.mp3 --output output --render-audio --render-score
|
||||
|
||||
# Render existing results without loading the model:
|
||||
python render.py --input output --output rendered --audio --score pdf,svg,png
|
||||
```
|
||||
|
||||
On minimal Linux servers, use `python setup_render.py --with-deps`. Audio follows the original MIDI timing. Scores preserve the vocal and instrumental staves. For separate piano previews, use `--render-parts vocal,instrumental,chords` with `infer.py`, or `--parts vocal,instrumental,chords` with `render.py`.
|
||||
|
||||
Python also accepts `render_audio=True, render_score="pdf,svg,png"`. In memory mode, `result["rendered"]` contains `audio` (part → WAV bytes) and `score` (format → pages as PDF/PNG bytes or SVG text).
|
||||
|
||||
<a id="benchmarks"></a>
|
||||
|
||||
## 📊 Benchmarks
|
||||
|
||||
Scores (%) on H800 with BF16 inference; higher is better. Use `preset="paper"` or `--preset paper` with the dataset prompts in [scores and settings](benchmark_results.json).
|
||||
|
||||
| Task | Dataset | Metric | SheetSage1 | madmom | Specialist | SheetSage2 |
|
||||
|---|---|---|---:|---:|---:|---:|
|
||||
| Beat | GTZAN | F1 | 85.79 | 85.79 | **88.75** [Beat This!][beat] | 85.65 |
|
||||
| Beat | osu2017 | F1 | 91.55 | 91.55 | 88.18 [Beat This!][beat] | **92.29** |
|
||||
| Downbeat | GTZAN | F1 | 64.33 | 64.33 | 78.28 [Beat This!][beat] | **79.51** |
|
||||
| Downbeat | osu2017 | F1 | 83.22 | 83.22 | 84.99 [Beat This!][beat] | **91.97** |
|
||||
| Key | GiantSteps | Weighted | 43.89 | 74.62 | 72.09 [S-KEY][key] | **77.73** |
|
||||
| Key | GTZAN | Weighted | 54.56 | 72.05 | 74.43 [S-KEY][key] | **75.77** |
|
||||
| Chord | osu2017 | Maj/min | 79.82 | 77.42 | 84.59 [Jiang et al.][chord] | **90.08** |
|
||||
| Chord | Chords1217 | Maj/min | 72.98 | 83.52* | **84.09**† [ChordFormer][chordformer] | 83.81 |
|
||||
| Structure | HarmonixSet | Accuracy | — | — | 80.03 [SongFormer][structure] | **80.51** |
|
||||
| Structure | HarmonixSet | F1 @ 0.5 s | — | — | **70.63** [SongFormer][structure] | 67.96 |
|
||||
| Structure | HarmonixSet | F1 @ 3 s | — | — | 79.50 [SongFormer][structure] | **82.86** |
|
||||
| Melody | RWC-Pop | Vocal F1 | 62.71 | — | 62.71 [SheetSage1][sheetsage] | **82.51** |
|
||||
| Melody | RWC-Pop | Full F1 | 64.02 | — | 64.02 [SheetSage1][sheetsage] | **75.29** |
|
||||
|
||||
Melody F1 uses pitch classes. SheetSage1 beat/downbeat results use madmom. *The madmom chord model includes Chords1217 in training. †ChordFormer uses five-fold cross-validation; SheetSage2 uses one checkpoint across all tracks.
|
||||
|
||||
[beat]: https://doi.org/10.5281/zenodo.14877491
|
||||
[key]: https://doi.org/10.1109/ICASSP49660.2025.10890222
|
||||
[chord]: https://archives.ismir.net/ismir2019/paper/000078.pdf
|
||||
[chordformer]: https://arxiv.org/abs/2502.11840
|
||||
[structure]: https://arxiv.org/abs/2510.02797
|
||||
[sheetsage]: https://arxiv.org/abs/2212.01884
|
||||
|
||||
<a id="citation"></a>
|
||||
|
||||
## 📚 Citation
|
||||
|
||||
**Technical report coming soon.** For now, please cite [YuE](https://arxiv.org/abs/2503.08638) and [MERT](https://proceedings.iclr.cc/paper_files/paper/2024/hash/33dffa2e3d2ab74a783d1a8c292f66d9-Abstract-Conference.html) when using SheetSage2 in your research.
|
||||
|
||||
```bibtex
|
||||
@article{yuan2025yue,
|
||||
title = {{YuE}: Scaling Open Foundation Models for Long-Form Music Generation},
|
||||
author = {Yuan, Ruibin and Lin, Hanfeng and Guo, Shuyue and Zhang, Ge and Pan, Jiahao and Zang, Yongyi and Liu, Haohe and Liang, Yiming and Ma, Wenye and Du, Xingjian and Du, Xinrun and Ye, Zhen and Zheng, Tianyu and Jiang, Zhengxuan and Ma, Yinghao and Liu, Minghao and Tian, Zeyue and Zhou, Ziya and Xue, Liumeng and Qu, Xingwei and Li, Yizhi and Wu, Shangda and Shen, Tianhao and Ma, Ziyang and Zhan, Jun and Wang, Chunhui and Wang, Yatian and Chi, Xiaowei and Zhang, Xinyue and Yang, Zhenzhu and Wang, Xiangzhou and Liu, Shansong and Mei, Lingrui and Li, Peng and Wang, Junjie and Yu, Jianwei and Pang, Guojian and Li, Xu and Wang, Zihao and Zhou, Xiaohuan and Yu, Lijun and Benetos, Emmanouil and Chen, Yong and Lin, Chenghua and Chen, Xie and Xia, Gus and Zhang, Zhaoxiang and Zhang, Chao and Chen, Wenhu and Zhou, Xinyu and Qiu, Xipeng and Dannenberg, Roger and Liu, Jiaheng and Yang, Jian and Huang, Wenhao and Xue, Wei and Tan, Xu and Guo, Yike},
|
||||
journal = {arXiv preprint arXiv:2503.08638},
|
||||
year = {2025},
|
||||
eprint = {2503.08638},
|
||||
archivePrefix = {arXiv},
|
||||
url = {https://arxiv.org/abs/2503.08638}
|
||||
}
|
||||
|
||||
@inproceedings{li2024mert,
|
||||
title = {MERT: Acoustic Music Understanding Model with Large-Scale Self-supervised Training},
|
||||
author = {Li, Yizhi and Yuan, Ruibin and Zhang, Ge and Ma, Yinghao and Chen, Xingran and Yin, Hanzhi and Xiao, Chenghao and Lin, Chenghua and Ragni, Anton and Benetos, Emmanouil and Gyenge, Norbert and Dannenberg, Roger and Liu, Ruibo and Chen, Wenhu and Xia, Gus and Shi, Yemin and Huang, Wenhao and Wang, Zili and Guo, Yike and Fu, Jie},
|
||||
booktitle = {International Conference on Learning Representations},
|
||||
year = {2024},
|
||||
url = {https://proceedings.iclr.cc/paper_files/paper/2024/hash/33dffa2e3d2ab74a783d1a8c292f66d9-Abstract-Conference.html}
|
||||
}
|
||||
```
|
||||
|
||||
Weights: [CC BY-NC 4.0](LICENSE). [Third-party notices](THIRD_PARTY_NOTICES.md).
|
||||
|
||||
**YuE2 family:** [Song generation](https://huggingface.co/m-a-p/YuE2-3B) · [Music representations](https://huggingface.co/m-a-p/MERT-v2-FullSong) · [Audio decoder](https://huggingface.co/m-a-p/YuE2-Vae).
|
||||
@@ -0,0 +1,8 @@
|
||||
# Third-party notices
|
||||
|
||||
- MERT-v2-FullSong provides the public music encoder. Its weight terms remain CC BY-NC 4.0.
|
||||
- The BART decoder uses Hugging Face Transformers (Apache 2.0); PyTorch uses its BSD-style license.
|
||||
- abcjs 6.6.3 is distributed under MIT; its license accompanies the rendering assets.
|
||||
- Playwright is distributed under Apache 2.0. Chromium is installed separately with its accompanying notices.
|
||||
- FluidR3 piano samples by Frank Wen use CC BY 3.0 US. Attribution and source information accompany the samples.
|
||||
- Other installed dependencies retain their respective licenses.
|
||||
@@ -0,0 +1 @@
|
||||
"""SheetSage2 standalone inference."""
|
||||
@@ -0,0 +1,92 @@
|
||||
"""File decoding and waveform preparation for music transcription."""
|
||||
from pathlib import Path
|
||||
import io
|
||||
import math
|
||||
import numbers
|
||||
import shutil
|
||||
import subprocess
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
SAMPLE_RATE = 24000
|
||||
|
||||
|
||||
def _encoded_bytes(audio):
|
||||
if isinstance(audio, (bytes, bytearray, memoryview)):
|
||||
return bytes(audio)
|
||||
if callable(getattr(audio, "read", None)):
|
||||
position = None
|
||||
try:
|
||||
position = audio.tell()
|
||||
audio.seek(0)
|
||||
except (AttributeError, OSError, ValueError, io.UnsupportedOperation):
|
||||
position = None
|
||||
try:
|
||||
value = audio.read()
|
||||
finally:
|
||||
if position is not None:
|
||||
audio.seek(position)
|
||||
if not isinstance(value, (bytes, bytearray)):
|
||||
raise ValueError("Audio streams must return encoded audio bytes")
|
||||
return bytes(value)
|
||||
return None
|
||||
|
||||
|
||||
def load_audio(audio, *, sampling_rate=None, max_seconds=None, preset="default"):
|
||||
if max_seconds is not None and (not math.isfinite(max_seconds) or max_seconds <= 0):
|
||||
raise ValueError("max_seconds must be finite and positive")
|
||||
encoded = _encoded_bytes(audio)
|
||||
if encoded is not None and not encoded:
|
||||
raise ValueError("Encoded audio must not be empty")
|
||||
if isinstance(audio, (str, Path)) or encoded is not None:
|
||||
if preset == "paper":
|
||||
import torchaudio
|
||||
source = io.BytesIO(encoded) if encoded is not None else str(audio)
|
||||
info = torchaudio.info(source, backend="ffmpeg")
|
||||
if info.num_frames <= 0:
|
||||
raise ValueError("Cannot determine source audio length")
|
||||
if encoded is not None:
|
||||
source.seek(0)
|
||||
waveform, rate = torchaudio.load(source, frame_offset=0, num_frames=int(info.num_frames),
|
||||
backend="ffmpeg", channels_first=True)
|
||||
waveform = waveform.mean(dim=0)
|
||||
if rate != SAMPLE_RATE:
|
||||
waveform = torchaudio.functional.resample(waveform, rate, SAMPLE_RATE)
|
||||
else:
|
||||
if not shutil.which("ffmpeg"):
|
||||
raise RuntimeError("FFmpeg is required to read audio files; install it and add it to PATH")
|
||||
source = "pipe:0" if encoded is not None else str(Path(audio).resolve())
|
||||
command = ["ffmpeg", "-v", "error", "-nostdin", "-i", source, "-vn"]
|
||||
if max_seconds is not None:
|
||||
command += ["-t", str(float(max_seconds))]
|
||||
command += ["-ac", "1", "-ar", str(SAMPLE_RATE), "-f", "f32le", "pipe:1"]
|
||||
result = subprocess.run(command, input=encoded, capture_output=True, timeout=600, check=False)
|
||||
if result.returncode:
|
||||
raise ValueError("Cannot decode audio: " + result.stderr.decode(errors="replace")[-1200:])
|
||||
waveform = torch.from_numpy(np.frombuffer(result.stdout, dtype="<f4").copy())
|
||||
else:
|
||||
if not isinstance(sampling_rate, numbers.Real) or not math.isfinite(sampling_rate) or sampling_rate <= 0 or sampling_rate != int(sampling_rate):
|
||||
raise ValueError("Provide sampling_rate for an array or tensor waveform")
|
||||
waveform = torch.as_tensor(audio, dtype=torch.float32, device="cpu")
|
||||
if waveform.ndim == 2:
|
||||
if waveform.shape[0] > 32:
|
||||
raise ValueError("Multichannel audio must have shape [channels, samples]")
|
||||
waveform = waveform.mean(dim=0)
|
||||
if waveform.ndim != 1:
|
||||
raise ValueError("Audio must have shape [samples] or [channels, samples]")
|
||||
if sampling_rate != SAMPLE_RATE:
|
||||
import torchaudio
|
||||
waveform = torchaudio.functional.resample(waveform, int(sampling_rate), SAMPLE_RATE)
|
||||
waveform = waveform.float().contiguous()
|
||||
if max_seconds is not None:
|
||||
waveform = waveform[:round(max_seconds * SAMPLE_RATE)]
|
||||
if waveform.numel() < 1025 or not torch.isfinite(waveform).all():
|
||||
raise ValueError("Audio must contain at least 1025 finite samples at 24 kHz")
|
||||
return waveform
|
||||
|
||||
|
||||
def slice_audio(audio, start, seconds):
|
||||
offset, count = round(start * SAMPLE_RATE), round(seconds * SAMPLE_RATE)
|
||||
result = audio[offset:offset + count]
|
||||
return torch.nn.functional.pad(result, (0, count - len(result)))
|
||||
@@ -0,0 +1,351 @@
|
||||
{
|
||||
"scores": [
|
||||
{
|
||||
"task": "beat",
|
||||
"dataset": "GTZAN",
|
||||
"metric": "F1",
|
||||
"score": 85.65,
|
||||
"settings": {
|
||||
"preset": "paper",
|
||||
"sample_rate": 24000,
|
||||
"window_seconds": 300,
|
||||
"overlap_seconds": 100,
|
||||
"lookahead_seconds": 0,
|
||||
"max_length": 5120,
|
||||
"decoding": "greedy",
|
||||
"attention": "sdpa",
|
||||
"weight_dtype": "float32",
|
||||
"autocast_dtype": "bfloat16",
|
||||
"batch_size": 1,
|
||||
"prompts": [
|
||||
"timestamp",
|
||||
"downbeat_meter",
|
||||
"key"
|
||||
],
|
||||
"device": "NVIDIA H800",
|
||||
"cuda_version": "12.6"
|
||||
}
|
||||
},
|
||||
{
|
||||
"task": "beat",
|
||||
"dataset": "osu2017",
|
||||
"metric": "F1",
|
||||
"score": 92.29,
|
||||
"settings": {
|
||||
"preset": "paper",
|
||||
"sample_rate": 24000,
|
||||
"window_seconds": 300,
|
||||
"overlap_seconds": 100,
|
||||
"lookahead_seconds": 0,
|
||||
"max_length": 5120,
|
||||
"decoding": "greedy",
|
||||
"attention": "sdpa",
|
||||
"weight_dtype": "float32",
|
||||
"autocast_dtype": "bfloat16",
|
||||
"batch_size": 1,
|
||||
"prompts": [
|
||||
"timestamp",
|
||||
"downbeat_meter",
|
||||
"key",
|
||||
"chord_full"
|
||||
],
|
||||
"device": "NVIDIA H800",
|
||||
"cuda_version": "12.6"
|
||||
}
|
||||
},
|
||||
{
|
||||
"task": "downbeat",
|
||||
"dataset": "GTZAN",
|
||||
"metric": "F1",
|
||||
"score": 79.51,
|
||||
"settings": {
|
||||
"preset": "paper",
|
||||
"sample_rate": 24000,
|
||||
"window_seconds": 300,
|
||||
"overlap_seconds": 100,
|
||||
"lookahead_seconds": 0,
|
||||
"max_length": 5120,
|
||||
"decoding": "greedy",
|
||||
"attention": "sdpa",
|
||||
"weight_dtype": "float32",
|
||||
"autocast_dtype": "bfloat16",
|
||||
"batch_size": 1,
|
||||
"prompts": [
|
||||
"timestamp",
|
||||
"downbeat_meter",
|
||||
"key"
|
||||
],
|
||||
"device": "NVIDIA H800",
|
||||
"cuda_version": "12.6"
|
||||
}
|
||||
},
|
||||
{
|
||||
"task": "downbeat",
|
||||
"dataset": "osu2017",
|
||||
"metric": "F1",
|
||||
"score": 91.97,
|
||||
"settings": {
|
||||
"preset": "paper",
|
||||
"sample_rate": 24000,
|
||||
"window_seconds": 300,
|
||||
"overlap_seconds": 100,
|
||||
"lookahead_seconds": 0,
|
||||
"max_length": 5120,
|
||||
"decoding": "greedy",
|
||||
"attention": "sdpa",
|
||||
"weight_dtype": "float32",
|
||||
"autocast_dtype": "bfloat16",
|
||||
"batch_size": 1,
|
||||
"prompts": [
|
||||
"timestamp",
|
||||
"downbeat_meter",
|
||||
"key",
|
||||
"chord_full"
|
||||
],
|
||||
"device": "NVIDIA H800",
|
||||
"cuda_version": "12.6"
|
||||
}
|
||||
},
|
||||
{
|
||||
"task": "key",
|
||||
"dataset": "GiantSteps",
|
||||
"metric": "weighted_accuracy",
|
||||
"score": 77.73,
|
||||
"settings": {
|
||||
"preset": "paper",
|
||||
"sample_rate": 24000,
|
||||
"window_seconds": 300,
|
||||
"overlap_seconds": 100,
|
||||
"lookahead_seconds": 0,
|
||||
"max_length": 5120,
|
||||
"decoding": "greedy",
|
||||
"attention": "sdpa",
|
||||
"weight_dtype": "float32",
|
||||
"autocast_dtype": "bfloat16",
|
||||
"batch_size": 1,
|
||||
"prompts": [
|
||||
"timestamp",
|
||||
"downbeat_meter",
|
||||
"key"
|
||||
],
|
||||
"device": "NVIDIA H800",
|
||||
"cuda_version": "12.6"
|
||||
}
|
||||
},
|
||||
{
|
||||
"task": "key",
|
||||
"dataset": "GTZAN",
|
||||
"metric": "weighted_accuracy",
|
||||
"score": 75.77,
|
||||
"settings": {
|
||||
"preset": "paper",
|
||||
"sample_rate": 24000,
|
||||
"window_seconds": 300,
|
||||
"overlap_seconds": 100,
|
||||
"lookahead_seconds": 0,
|
||||
"max_length": 5120,
|
||||
"decoding": "greedy",
|
||||
"attention": "sdpa",
|
||||
"weight_dtype": "float32",
|
||||
"autocast_dtype": "bfloat16",
|
||||
"batch_size": 1,
|
||||
"prompts": [
|
||||
"timestamp",
|
||||
"downbeat_meter",
|
||||
"key"
|
||||
],
|
||||
"device": "NVIDIA H800",
|
||||
"cuda_version": "12.6"
|
||||
}
|
||||
},
|
||||
{
|
||||
"task": "chord",
|
||||
"dataset": "osu2017",
|
||||
"metric": "majmin",
|
||||
"score": 90.08,
|
||||
"settings": {
|
||||
"preset": "paper",
|
||||
"sample_rate": 24000,
|
||||
"window_seconds": 300,
|
||||
"overlap_seconds": 100,
|
||||
"lookahead_seconds": 0,
|
||||
"max_length": 5120,
|
||||
"decoding": "greedy",
|
||||
"attention": "sdpa",
|
||||
"weight_dtype": "float32",
|
||||
"autocast_dtype": "bfloat16",
|
||||
"batch_size": 1,
|
||||
"prompts": [
|
||||
"timestamp",
|
||||
"downbeat_meter",
|
||||
"key",
|
||||
"chord_full"
|
||||
],
|
||||
"device": "NVIDIA H800",
|
||||
"cuda_version": "12.6"
|
||||
}
|
||||
},
|
||||
{
|
||||
"task": "chord",
|
||||
"dataset": "Chords1217",
|
||||
"metric": "majmin",
|
||||
"score": 83.81,
|
||||
"settings": {
|
||||
"preset": "paper",
|
||||
"sample_rate": 24000,
|
||||
"window_seconds": 300,
|
||||
"overlap_seconds": 100,
|
||||
"lookahead_seconds": 0,
|
||||
"max_length": 5120,
|
||||
"decoding": "greedy",
|
||||
"attention": "sdpa",
|
||||
"weight_dtype": "float32",
|
||||
"autocast_dtype": "bfloat16",
|
||||
"batch_size": 1,
|
||||
"prompts": [
|
||||
"timestamp",
|
||||
"downbeat_meter",
|
||||
"chord_full"
|
||||
],
|
||||
"device": "NVIDIA H800",
|
||||
"cuda_version": "12.6"
|
||||
}
|
||||
},
|
||||
{
|
||||
"task": "structure",
|
||||
"dataset": "HarmonixSet",
|
||||
"metric": "accuracy",
|
||||
"score": 80.51,
|
||||
"settings": {
|
||||
"preset": "paper",
|
||||
"sample_rate": 24000,
|
||||
"window_seconds": 300,
|
||||
"overlap_seconds": 100,
|
||||
"lookahead_seconds": 0,
|
||||
"max_length": 5120,
|
||||
"decoding": "greedy",
|
||||
"attention": "sdpa",
|
||||
"weight_dtype": "float32",
|
||||
"autocast_dtype": "bfloat16",
|
||||
"batch_size": 1,
|
||||
"prompts": [
|
||||
"timestamp",
|
||||
"downbeat_meter",
|
||||
"structure"
|
||||
],
|
||||
"device": "NVIDIA H800",
|
||||
"cuda_version": "12.6"
|
||||
}
|
||||
},
|
||||
{
|
||||
"task": "structure",
|
||||
"dataset": "HarmonixSet",
|
||||
"metric": "F1@0.5s",
|
||||
"score": 67.96,
|
||||
"settings": {
|
||||
"preset": "paper",
|
||||
"sample_rate": 24000,
|
||||
"window_seconds": 300,
|
||||
"overlap_seconds": 100,
|
||||
"lookahead_seconds": 0,
|
||||
"max_length": 5120,
|
||||
"decoding": "greedy",
|
||||
"attention": "sdpa",
|
||||
"weight_dtype": "float32",
|
||||
"autocast_dtype": "bfloat16",
|
||||
"batch_size": 1,
|
||||
"prompts": [
|
||||
"timestamp",
|
||||
"downbeat_meter",
|
||||
"structure"
|
||||
],
|
||||
"device": "NVIDIA H800",
|
||||
"cuda_version": "12.6"
|
||||
}
|
||||
},
|
||||
{
|
||||
"task": "structure",
|
||||
"dataset": "HarmonixSet",
|
||||
"metric": "F1@3s",
|
||||
"score": 82.86,
|
||||
"settings": {
|
||||
"preset": "paper",
|
||||
"sample_rate": 24000,
|
||||
"window_seconds": 300,
|
||||
"overlap_seconds": 100,
|
||||
"lookahead_seconds": 0,
|
||||
"max_length": 5120,
|
||||
"decoding": "greedy",
|
||||
"attention": "sdpa",
|
||||
"weight_dtype": "float32",
|
||||
"autocast_dtype": "bfloat16",
|
||||
"batch_size": 1,
|
||||
"prompts": [
|
||||
"timestamp",
|
||||
"downbeat_meter",
|
||||
"structure"
|
||||
],
|
||||
"device": "NVIDIA H800",
|
||||
"cuda_version": "12.6"
|
||||
}
|
||||
},
|
||||
{
|
||||
"task": "melody",
|
||||
"dataset": "RWC-Pop",
|
||||
"metric": "vocal_pitch_class_F1",
|
||||
"score": 82.51,
|
||||
"settings": {
|
||||
"preset": "paper",
|
||||
"sample_rate": 24000,
|
||||
"window_seconds": 300,
|
||||
"overlap_seconds": 100,
|
||||
"lookahead_seconds": 0,
|
||||
"max_length": 5120,
|
||||
"decoding": "greedy",
|
||||
"attention": "sdpa",
|
||||
"weight_dtype": "float32",
|
||||
"autocast_dtype": "bfloat16",
|
||||
"batch_size": 1,
|
||||
"prompts": [
|
||||
"timestamp",
|
||||
"downbeat_meter",
|
||||
"structure",
|
||||
"key",
|
||||
"chord_full",
|
||||
"melody_full"
|
||||
],
|
||||
"device": "NVIDIA H800",
|
||||
"cuda_version": "12.6"
|
||||
}
|
||||
},
|
||||
{
|
||||
"task": "melody",
|
||||
"dataset": "RWC-Pop",
|
||||
"metric": "full_pitch_class_F1",
|
||||
"score": 75.29,
|
||||
"settings": {
|
||||
"preset": "paper",
|
||||
"sample_rate": 24000,
|
||||
"window_seconds": 300,
|
||||
"overlap_seconds": 100,
|
||||
"lookahead_seconds": 0,
|
||||
"max_length": 5120,
|
||||
"decoding": "greedy",
|
||||
"attention": "sdpa",
|
||||
"weight_dtype": "float32",
|
||||
"autocast_dtype": "bfloat16",
|
||||
"batch_size": 1,
|
||||
"prompts": [
|
||||
"timestamp",
|
||||
"downbeat_meter",
|
||||
"structure",
|
||||
"key",
|
||||
"chord_full",
|
||||
"melody_full"
|
||||
],
|
||||
"device": "NVIDIA H800",
|
||||
"cuda_version": "12.6"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,75 @@
|
||||
{
|
||||
"architectures": [
|
||||
"SheetSage2Model"
|
||||
],
|
||||
"auto_map": {
|
||||
"AutoConfig": "configuration_sheetsage2.SheetSage2Config",
|
||||
"AutoModel": "modeling_sheetsage2.SheetSage2Model",
|
||||
"AutoModelForSeq2SeqLM": "modeling_sheetsage2.SheetSage2Model",
|
||||
"AutoProcessor": "processing_sheetsage2.SheetSage2Processor"
|
||||
},
|
||||
"backbone_config": {
|
||||
"architectures": [
|
||||
"MERT2Model"
|
||||
],
|
||||
"context_seconds": 360.0,
|
||||
"conv_depthwise_kernel_size": 31,
|
||||
"frame_rate": 25.0,
|
||||
"hidden_size": 1024,
|
||||
"hop_length": 240,
|
||||
"initializer_range": 0.02,
|
||||
"inputs_to_logits_ratio": 960,
|
||||
"intermediate_size": 4096,
|
||||
"layer_norm_eps": 1e-05,
|
||||
"minimum_input_samples": 1025,
|
||||
"model_type": "mert2",
|
||||
"n_fft": 2048,
|
||||
"num_attention_heads": 16,
|
||||
"num_hidden_layers": 24,
|
||||
"num_mel_bins": 128,
|
||||
"rotary_embedding_base": 10000,
|
||||
"sampling_rate": 24000,
|
||||
"subsampling_channels": [
|
||||
128,
|
||||
512,
|
||||
1024
|
||||
],
|
||||
"subsampling_depths": [
|
||||
3,
|
||||
4,
|
||||
5
|
||||
],
|
||||
"subsampling_layer_norm_eps": 1e-06,
|
||||
"torch_dtype": "float32",
|
||||
"transformers_version": "4.53.2",
|
||||
"variant": "fs",
|
||||
"win_length": 2048
|
||||
},
|
||||
"base_model_name_or_path": "m-a-p/MERT-v2-FullSong",
|
||||
"base_model_revision": "d8ba1c745e733b3908ce6ad16ebeb17ac7600a42",
|
||||
"base_model_sha256": "e6dd2ab187d6dd62b6521cd7d8f932e237acf0c5757745a7232082e28391350d",
|
||||
"bos_token_id": 1,
|
||||
"decoder_dropout": 0.1,
|
||||
"decoder_layers": 6,
|
||||
"decoder_start_token_id": 1,
|
||||
"encoder_attn_implementation": "sdpa",
|
||||
"eos_token_id": 2,
|
||||
"hidden_size": 512,
|
||||
"input_audio_length": 300.0,
|
||||
"intermediate_size": 2048,
|
||||
"is_encoder_decoder": true,
|
||||
"lora_alpha": 128.0,
|
||||
"lora_rank": 64,
|
||||
"max_output_seq_len": 5120,
|
||||
"model_type": "sheetsage2",
|
||||
"num_attention_heads": 8,
|
||||
"pad_token_id": 0,
|
||||
"sampling_rate": 24000,
|
||||
"time_hz": 100,
|
||||
"tokenizer_fingerprint": "5ba3325af0344c7f",
|
||||
"tokenizer_schema_version": "v1",
|
||||
"transformers_version": "4.45.2",
|
||||
"use_cache": true,
|
||||
"vocab_size": 31678,
|
||||
"weights_format": "adapter"
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
"""Configuration for MERT2 music representation models."""
|
||||
|
||||
from transformers import PretrainedConfig
|
||||
|
||||
|
||||
class MERT2Config(PretrainedConfig):
|
||||
"""Architecture shared by the 30-second and full-song MERT2 encoders."""
|
||||
|
||||
model_type = "mert2"
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
hidden_size=1024,
|
||||
intermediate_size=4096,
|
||||
num_hidden_layers=24,
|
||||
num_attention_heads=16,
|
||||
num_mel_bins=128,
|
||||
sampling_rate=24000,
|
||||
n_fft=2048,
|
||||
win_length=2048,
|
||||
hop_length=240,
|
||||
subsampling_channels=None,
|
||||
subsampling_depths=None,
|
||||
conv_depthwise_kernel_size=31,
|
||||
rotary_embedding_base=10000,
|
||||
layer_norm_eps=1e-5,
|
||||
subsampling_layer_norm_eps=1e-6,
|
||||
initializer_range=0.02,
|
||||
variant="30s",
|
||||
context_seconds=None,
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__(**kwargs)
|
||||
self.hidden_size = int(hidden_size)
|
||||
self.intermediate_size = int(intermediate_size)
|
||||
self.num_hidden_layers = int(num_hidden_layers)
|
||||
self.num_attention_heads = int(num_attention_heads)
|
||||
self.num_mel_bins = int(num_mel_bins)
|
||||
self.sampling_rate = int(sampling_rate)
|
||||
self.n_fft = int(n_fft)
|
||||
self.win_length = int(win_length)
|
||||
self.hop_length = int(hop_length)
|
||||
self.subsampling_channels = list(subsampling_channels or [num_mel_bins, 512, hidden_size])
|
||||
self.subsampling_depths = list(subsampling_depths or [3, 4, 5])
|
||||
self.conv_depthwise_kernel_size = int(conv_depthwise_kernel_size)
|
||||
self.rotary_embedding_base = int(rotary_embedding_base)
|
||||
self.layer_norm_eps = float(layer_norm_eps)
|
||||
self.subsampling_layer_norm_eps = float(subsampling_layer_norm_eps)
|
||||
self.initializer_range = float(initializer_range)
|
||||
self.variant = str(variant)
|
||||
self.context_seconds = float(context_seconds if context_seconds is not None else (360 if variant == "fs" else 30))
|
||||
self.inputs_to_logits_ratio = self.hop_length * 4
|
||||
self.frame_rate = self.sampling_rate / self.inputs_to_logits_ratio
|
||||
self.minimum_input_samples = self.n_fft // 2 + 1
|
||||
|
||||
if min(self.hidden_size, self.intermediate_size, self.num_hidden_layers, self.num_attention_heads) <= 0:
|
||||
raise ValueError("Encoder dimensions and layer counts must be positive.")
|
||||
if self.hidden_size % self.num_attention_heads or (self.hidden_size // self.num_attention_heads) % 2:
|
||||
raise ValueError("hidden_size must divide into an even head dimension.")
|
||||
if len(self.subsampling_channels) != 3 or len(self.subsampling_depths) != 3:
|
||||
raise ValueError("The subsampler must contain three channel widths and three depths.")
|
||||
if self.subsampling_channels[0] != self.num_mel_bins or self.subsampling_channels[-1] != self.hidden_size:
|
||||
raise ValueError("Subsampling widths must start at num_mel_bins and end at hidden_size.")
|
||||
if min(*self.subsampling_channels, *self.subsampling_depths, self.num_mel_bins, self.sampling_rate, self.hop_length) <= 0:
|
||||
raise ValueError("Frontend dimensions and sampling parameters must be positive.")
|
||||
if self.n_fft < self.win_length or self.win_length <= 0 or self.n_fft % 2:
|
||||
raise ValueError("n_fft must be even and at least win_length > 0.")
|
||||
if self.conv_depthwise_kernel_size <= 0 or self.conv_depthwise_kernel_size % 2 != 1:
|
||||
raise ValueError("conv_depthwise_kernel_size must be positive and odd.")
|
||||
if self.rotary_embedding_base <= 0 or min(self.layer_norm_eps, self.subsampling_layer_norm_eps) <= 0:
|
||||
raise ValueError("Rotary base and normalization epsilons must be positive.")
|
||||
if self.variant not in {"30s", "fs"}:
|
||||
raise ValueError("variant must be '30s' or 'fs'.")
|
||||
|
||||
|
||||
MERT2Config.register_for_auto_class()
|
||||
@@ -0,0 +1,88 @@
|
||||
"""Configuration for SheetSage2 audio-to-symbolic transcription."""
|
||||
|
||||
from transformers import PretrainedConfig
|
||||
|
||||
from .configuration_mert2 import MERT2Config
|
||||
|
||||
|
||||
class SheetSage2Config(PretrainedConfig):
|
||||
model_type = "sheetsage2"
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
vocab_size=31678,
|
||||
hidden_size=512,
|
||||
decoder_layers=6,
|
||||
num_attention_heads=8,
|
||||
intermediate_size=2048,
|
||||
decoder_dropout=0.1,
|
||||
input_audio_length=300.0,
|
||||
max_output_seq_len=5120,
|
||||
time_hz=100,
|
||||
sampling_rate=24000,
|
||||
lora_rank=64,
|
||||
lora_alpha=128,
|
||||
weights_format="adapter",
|
||||
backbone_config=None,
|
||||
encoder_attn_implementation="sdpa",
|
||||
base_model_name_or_path="m-a-p/MERT-v2-FullSong",
|
||||
base_model_revision="d8ba1c745e733b3908ce6ad16ebeb17ac7600a42",
|
||||
base_model_sha256="e6dd2ab187d6dd62b6521cd7d8f932e237acf0c5757745a7232082e28391350d",
|
||||
tokenizer_schema_version="v1",
|
||||
tokenizer_fingerprint="5ba3325af0344c7f",
|
||||
**kwargs,
|
||||
):
|
||||
for name, default in (
|
||||
("is_encoder_decoder", True), ("tie_word_embeddings", True),
|
||||
("pad_token_id", 0), ("bos_token_id", 1), ("eos_token_id", 2),
|
||||
("decoder_start_token_id", 1),
|
||||
("use_cache", True),
|
||||
):
|
||||
kwargs.setdefault(name, default)
|
||||
super().__init__(**kwargs)
|
||||
self.vocab_size = int(vocab_size)
|
||||
self.hidden_size = int(hidden_size)
|
||||
self.decoder_layers = int(decoder_layers)
|
||||
self.num_attention_heads = int(num_attention_heads)
|
||||
self.intermediate_size = int(intermediate_size)
|
||||
self.decoder_dropout = float(decoder_dropout)
|
||||
self.input_audio_length = float(input_audio_length)
|
||||
self.max_output_seq_len = int(max_output_seq_len)
|
||||
self.time_hz = int(time_hz)
|
||||
self.sampling_rate = int(sampling_rate)
|
||||
self.lora_rank = int(lora_rank)
|
||||
self.lora_alpha = float(lora_alpha)
|
||||
self.weights_format = str(weights_format)
|
||||
self.backbone_config = dict(backbone_config or MERT2Config(variant="fs").to_dict())
|
||||
self.backbone_config.pop("_name_or_path", None)
|
||||
self.backbone_config.pop("auto_map", None)
|
||||
self.encoder_attn_implementation = str(encoder_attn_implementation)
|
||||
self.base_model_name_or_path = str(base_model_name_or_path)
|
||||
self.base_model_revision = str(base_model_revision)
|
||||
self.base_model_sha256 = str(base_model_sha256)
|
||||
self.tokenizer_schema_version = str(tokenizer_schema_version)
|
||||
self.tokenizer_fingerprint = str(tokenizer_fingerprint)
|
||||
self.architectures = ["SheetSage2Model"]
|
||||
self.auto_map = {
|
||||
"AutoConfig": "configuration_sheetsage2.SheetSage2Config",
|
||||
"AutoModel": "modeling_sheetsage2.SheetSage2Model",
|
||||
"AutoModelForSeq2SeqLM": "modeling_sheetsage2.SheetSage2Model",
|
||||
"AutoProcessor": "processing_sheetsage2.SheetSage2Processor",
|
||||
}
|
||||
if self.weights_format not in {"adapter", "merged"}:
|
||||
raise ValueError("weights_format must be 'adapter' or 'merged'.")
|
||||
if self.encoder_attn_implementation not in {"sdpa", "flash_attention_2"}:
|
||||
raise ValueError("Select encoder attention 'sdpa' or 'flash_attention_2'.")
|
||||
if min(self.vocab_size, self.hidden_size, self.decoder_layers, self.num_attention_heads,
|
||||
self.intermediate_size, self.input_audio_length, self.max_output_seq_len,
|
||||
self.time_hz, self.sampling_rate, self.lora_rank, self.lora_alpha) <= 0:
|
||||
raise ValueError("Model dimensions and timing parameters must be positive.")
|
||||
if self.hidden_size % self.num_attention_heads:
|
||||
raise ValueError("hidden_size must be divisible by num_attention_heads.")
|
||||
if not 0 <= self.decoder_dropout < 1:
|
||||
raise ValueError("decoder_dropout must lie in [0, 1).")
|
||||
if self.sampling_rate != self.backbone_config["sampling_rate"]:
|
||||
raise ValueError("Processor and encoder sampling rates must match.")
|
||||
|
||||
|
||||
SheetSage2Config.register_for_auto_class()
|
||||
@@ -0,0 +1,14 @@
|
||||
import numpy as np
|
||||
|
||||
|
||||
DURATION_TEMPLATES = np.array(
|
||||
[
|
||||
1, 2, 3, 4, 6, 8, 12, 16, 24, 32, 48, 64,
|
||||
96, 128, 192, 256, 384, 512, 768, 1024,
|
||||
1536, 2048, 3072, 4096,
|
||||
],
|
||||
dtype=np.int32,
|
||||
)
|
||||
duration_boundaries = (DURATION_TEMPLATES[:-1] + DURATION_TEMPLATES[1:]) / 2
|
||||
|
||||
|
||||
@@ -0,0 +1,200 @@
|
||||
"""Lossless decoded events/LAB/MIDI, then independently validated ABC."""
|
||||
from pathlib import Path
|
||||
import json
|
||||
from fractions import Fraction
|
||||
|
||||
import numpy as np
|
||||
import pretty_midi
|
||||
|
||||
from .notation_sheetsage2 import generate_abc_from_data
|
||||
from .io_sheetsage2 import atomic_write_text
|
||||
from .generation_sheetsage2 import field_text
|
||||
from .midi_sheetsage2 import build_playback, midi_bytes
|
||||
|
||||
|
||||
def rows_text(rows):
|
||||
return "".join("\t".join(str(x) for x in row) + "\n" for row in rows)
|
||||
|
||||
|
||||
def write_rows(path, rows):
|
||||
atomic_write_text(path, rows_text(rows))
|
||||
|
||||
|
||||
def interval_rows(events, field, duration):
|
||||
rows = [[e["time"], 0, e["values"][field]] for e in events if field in e["values"]]
|
||||
for i, row in enumerate(rows):
|
||||
row[1] = rows[i + 1][0] if i + 1 < len(rows) else duration
|
||||
return [r for r in rows if r[1] > r[0]]
|
||||
|
||||
|
||||
def rhythm_rows(events):
|
||||
rows, meter = [], None
|
||||
for event in events:
|
||||
rhythm = event["values"].get("rhythm", {})
|
||||
meter = rhythm.get("meter", meter)
|
||||
eighth = rhythm.get("eighth_position")
|
||||
if eighth is not None and meter is not None:
|
||||
# The model stores eighth-note position, not denominator-beat index.
|
||||
position = Fraction(int(eighth) * int(meter[1]), 8)
|
||||
if position.denominator != 1:
|
||||
raise ValueError(f"Eighth position {eighth} is off the {meter[0]}/{meter[1]} beat grid")
|
||||
if not 0 <= position < meter[0]:
|
||||
raise ValueError(f"Eighth position {eighth} is outside meter {meter}")
|
||||
rows.append([float(event["time"]), int(position) + 1, int(meter[0]), int(meter[1])])
|
||||
return rows
|
||||
|
||||
|
||||
def _midi(notes, paper=False):
|
||||
midi = pretty_midi.PrettyMIDI(resolution=220 if paper else 960)
|
||||
names = ((0, "vocal_melody"), (1, "instrumental_melody")) if paper else ((0, "Vocal"), (1, "Ins"))
|
||||
for track, name in names:
|
||||
instrument = pretty_midi.Instrument(program=0, name=name)
|
||||
for start, end, pitch, source in notes:
|
||||
if track == source and end > start:
|
||||
instrument.notes.append(pretty_midi.Note(100, pitch, start, end))
|
||||
midi.instruments.append(instrument)
|
||||
return midi
|
||||
|
||||
|
||||
def notation_notes(notes):
|
||||
"""A monophonic notation view, retaining the exact raw prediction separately."""
|
||||
result, diagnostics = [], []
|
||||
for track in (0, 1):
|
||||
ordered = sorted((list(n) for n in notes if n[3] == track), key=lambda n: (n[0], n[2], n[1]))
|
||||
for i, note in enumerate(ordered):
|
||||
if i + 1 < len(ordered) and note[1] > ordered[i + 1][0] + 1e-6:
|
||||
note[1] = ordered[i + 1][0]
|
||||
diagnostics.append(f"notation only: clipped track {track} note at {note[0]:.3f} to next onset")
|
||||
if note[1] > note[0] + 1e-6:
|
||||
result.append(note)
|
||||
return sorted(result), diagnostics
|
||||
|
||||
|
||||
def export_result(decoded, tokenizer, output_dir=None, duration=None, paper=False, *, melody_only=False):
|
||||
"""Build ABC, MIDI and LAB in memory, optionally writing the same bytes.
|
||||
|
||||
Statistics retain their file-interface names. ``payload`` contains the
|
||||
in-memory ABC text, MIDI bytes, decoded event list, LAB texts and playback.
|
||||
melody_only removes chords from notation and playback without changing
|
||||
decoded events or raw LAB annotations.
|
||||
"""
|
||||
if not isinstance(melody_only, bool):
|
||||
raise ValueError("melody_only must be True or False")
|
||||
if duration is None:
|
||||
raise TypeError("duration is required")
|
||||
texts, midis, notation_midis, labs = {}, {}, {}, {}
|
||||
|
||||
def keep_rows(name, rows):
|
||||
text = rows_text(rows)
|
||||
texts[name] = text
|
||||
if name.endswith(".lab"):
|
||||
labs[name[:-4]] = text
|
||||
|
||||
events = decoded["events"]
|
||||
notes = []
|
||||
for event in events:
|
||||
start = float(event["time"])
|
||||
for note in event["values"].get("melody", ()):
|
||||
end = max(start + 0.04, float(note["end_time"])) if paper else min(duration, float(note["end_time"]))
|
||||
if end > start:
|
||||
notes.append([start, end, int(note["pitch"]), int(note["track"])])
|
||||
if not paper:
|
||||
notes.sort()
|
||||
texts["events.json"] = json.dumps(decoded, indent=2)
|
||||
keep_rows("events.tsv", [["time", "global_subbeat", "fields"]] +
|
||||
[[e["time"], e["global_subbeat"], field_text(e, tokenizer)] for e in events])
|
||||
midis["melody"] = midi_bytes(_midi(notes, paper=paper))
|
||||
lab_notes = [[f"{a:.6f}", f"{b:.6f}", p, t] for a, b, p, t in notes] if paper else notes
|
||||
keep_rows("melody_full.lab", lab_notes)
|
||||
for track, name in ((0, "vocal"), (1, "instrumental")):
|
||||
selected = [n for n in notes if n[3] == track]
|
||||
keep_rows(f"melody_{name}.lab", [n[:3] for n in lab_notes if n[3] == track])
|
||||
midis[f"melody_{name}"] = midi_bytes(_midi(selected, paper=paper))
|
||||
intervals = {}
|
||||
for field in ("chord", "key", "structure"):
|
||||
rows = interval_rows(events, field, duration)
|
||||
if paper:
|
||||
rows = [[float(e["time"]), None, e["values"][field]] for e in events if field in e["values"]]
|
||||
for i, row in enumerate(rows):
|
||||
row[1] = max(row[0], rows[i + 1][0]) if i + 1 < len(rows) else float(duration)
|
||||
intervals[field] = rows
|
||||
keep_rows(f"{field}.lab", rows)
|
||||
if paper:
|
||||
rows = []
|
||||
for event in events:
|
||||
rhythm, stamp = event["values"].get("rhythm"), event["values"].get("timestamp")
|
||||
if rhythm is None and stamp is None:
|
||||
continue
|
||||
meter = rhythm.get("meter") if isinstance(rhythm, dict) else None
|
||||
eighth = rhythm.get("eighth_position") if isinstance(rhythm, dict) else None
|
||||
rows.append([f"{event['time']:.6f}", "" if stamp is None else f"{float(stamp):.6f}",
|
||||
"" if eighth is None else int(eighth), "" if meter is None else f"{meter[0]}/{meter[1]}"])
|
||||
keep_rows("beat_meter.lab", rows)
|
||||
raw_rhythm = [[e["time"], json.dumps(e["values"].get("rhythm", {}))]
|
||||
for e in events if "rhythm" in e["values"] or "timestamp" in e["values"]]
|
||||
keep_rows("rhythm_events.lab", raw_rhythm)
|
||||
diagnostics, abc_error, score, text = [], None, None, None
|
||||
try:
|
||||
beats = rhythm_rows(events)
|
||||
keep_rows("beat.lab", beats)
|
||||
keep_rows("downbeat.lab", [[r[0]] for r in beats if r[1] == 1])
|
||||
if len(beats) < 2:
|
||||
raise ValueError("At least two decoded beats are required for ABC")
|
||||
# canonical notation's last beat is a boundary, so append it explicitly. Continue the
|
||||
# final tempo only as far as the audio/last note and let it pad the bar.
|
||||
abc_beats = [list(b) for b in beats]
|
||||
period = float(np.median(np.diff([b[0] for b in beats[-9:]])))
|
||||
if period <= 0:
|
||||
raise ValueError("Decoded beats must increase in time")
|
||||
end = max(duration, max((n[1] for n in notes), default=0))
|
||||
while abc_beats[-1][0] < end - 1e-6:
|
||||
prev = abc_beats[-1]
|
||||
abc_beats.append([prev[0] + period, prev[1] % prev[2] + 1, prev[2], prev[3]])
|
||||
keep_rows("notation/song_beats.txt", abc_beats)
|
||||
notation_intervals = {}
|
||||
for field, suffix in (("chord", "chords"), ("key", "keys"), ("structure", "structures")):
|
||||
rows = interval_rows(events, field, duration)
|
||||
# Clip to the actual beat domain: events beyond it have no ABC bin.
|
||||
rows = [[max(abc_beats[0][0], a), min(abc_beats[-1][0], b), v]
|
||||
for a, b, v in rows if b > abc_beats[0][0] and a < abc_beats[-1][0]]
|
||||
notation_intervals[field] = rows
|
||||
keep_rows(f"notation/song_{suffix}.txt", rows)
|
||||
if not interval_rows(events, "key", duration):
|
||||
raise ValueError("No key was decoded; cannot construct a keyed ABC score")
|
||||
clean, adjustments = notation_notes(notes)
|
||||
diagnostics.extend(adjustments)
|
||||
notation_midis["notation/song_melody.mid"] = midi_bytes(_midi(clean))
|
||||
text, score = generate_abc_from_data(
|
||||
notation_midis["notation/song_melody.mid"], abc_beats,
|
||||
notation_intervals["chord"], notation_intervals["key"], notation_intervals["structure"],
|
||||
melody_only=melody_only,
|
||||
)
|
||||
diagnostics.extend(score.diagnostics)
|
||||
texts["score.abc"] = text
|
||||
abc_measures = len(score.measures)
|
||||
except (ValueError, FileNotFoundError) as exc:
|
||||
abc_error = str(exc)
|
||||
abc_measures = 0
|
||||
playback, playback_midis = build_playback(
|
||||
midis["melody"], [] if melody_only else intervals["chord"], score, duration)
|
||||
midis.update(playback_midis)
|
||||
texts["playback.json"] = json.dumps(playback, indent=2)
|
||||
diagnostics.extend(playback["warnings"])
|
||||
if output_dir is not None:
|
||||
output_dir = Path(output_dir)
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
for name, content in texts.items():
|
||||
atomic_write_text(output_dir / name, content)
|
||||
for name, content in {**{f"{name}.mid": content for name, content in midis.items()}, **notation_midis}.items():
|
||||
path = output_dir / name
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_bytes(content)
|
||||
if abc_error is not None:
|
||||
(output_dir / "score.abc").unlink(missing_ok=True)
|
||||
else:
|
||||
(output_dir / "score.browser.abc").unlink(missing_ok=True)
|
||||
payload = dict(abc=text, midi=midis["transcription"], midis=midis,
|
||||
events=events, labs=labs, playback=playback)
|
||||
return dict(melody_notes=len(notes), vocal_notes=sum(n[3] == 0 for n in notes),
|
||||
instrumental_notes=sum(n[3] == 1 for n in notes), events=len(events),
|
||||
abc_measures=abc_measures, abc_error=abc_error, diagnostics=diagnostics, payload=payload)
|
||||
@@ -0,0 +1,634 @@
|
||||
"""Prompt grammar, cached generation and overlap context from the evaluated model."""
|
||||
import copy
|
||||
from contextlib import nullcontext
|
||||
from pathlib import Path
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
FULL_TASK_PROMPTS = ("timestamp", "downbeat_meter", "structure", "key", "chord_full", "melody_full")
|
||||
FIELD_TO_INDEX = {"timestamp": 0, "rhythm": 1, "structure": 2, "key": 3, "chord": 4, "melody": 5}
|
||||
|
||||
|
||||
class PromptGrammarState:
|
||||
def __init__(self, tokenizer):
|
||||
self.tokenizer = tokenizer
|
||||
self.generated_events = 0
|
||||
self.in_shift = True
|
||||
self.shift_run = 0
|
||||
self.payload_count = 0
|
||||
self.last_field_index = -1
|
||||
self.incomplete = None
|
||||
|
||||
def _allow_field_starts(self, allowed):
|
||||
tokenizer = self.tokenizer
|
||||
if self.last_field_index < FIELD_TO_INDEX["timestamp"]:
|
||||
allowed[tokenizer.time_token_start : tokenizer.time_token_end] = True
|
||||
if self.last_field_index < FIELD_TO_INDEX["rhythm"]:
|
||||
allowed[tokenizer.meter_token_start : tokenizer.meter_token_end] = True
|
||||
allowed[
|
||||
tokenizer.eighth_position_token_start : tokenizer.eighth_position_token_end
|
||||
] = True
|
||||
if self.last_field_index < FIELD_TO_INDEX["structure"]:
|
||||
allowed[
|
||||
tokenizer.structure_token_start : tokenizer.structure_token_end
|
||||
] = True
|
||||
if self.last_field_index < FIELD_TO_INDEX["key"]:
|
||||
allowed[tokenizer.key_token_start : tokenizer.key_token_end] = True
|
||||
if self.last_field_index < FIELD_TO_INDEX["chord"]:
|
||||
allowed[
|
||||
tokenizer.full_chord_token_start : tokenizer.full_chord_token_end
|
||||
] = True
|
||||
if self.last_field_index <= FIELD_TO_INDEX["melody"]:
|
||||
allowed[tokenizer.pitch_token_start : tokenizer.pitch_token_end] = True
|
||||
|
||||
def allowed(self, device):
|
||||
tokenizer = self.tokenizer
|
||||
allowed = torch.zeros(tokenizer.n_tokens, dtype=torch.bool, device=device)
|
||||
can_end = self.payload_count > 0
|
||||
|
||||
if can_end:
|
||||
allowed[tokenizer.eos_token] = True
|
||||
if self.payload_count > 0 or self.in_shift:
|
||||
if self.shift_run < 4:
|
||||
allowed[
|
||||
tokenizer.subbeat_shift_token_start : tokenizer.subbeat_shift_token_end
|
||||
] = True
|
||||
|
||||
if self.incomplete == "rhythm_after_meter":
|
||||
allowed[
|
||||
tokenizer.eighth_position_token_start : tokenizer.eighth_position_token_end
|
||||
] = True
|
||||
return allowed
|
||||
|
||||
if self.incomplete == "melody_after_pitch":
|
||||
allowed[tokenizer.duration_token_start : tokenizer.duration_token_end] = True
|
||||
allowed[tokenizer.pitch_token_start : tokenizer.pitch_token_end] = True
|
||||
return allowed
|
||||
|
||||
self._allow_field_starts(allowed)
|
||||
return allowed
|
||||
|
||||
def update(self, token):
|
||||
tokenizer = self.tokenizer
|
||||
token = int(token)
|
||||
token_type = tokenizer.token_type(token)
|
||||
if token == tokenizer.eos_token:
|
||||
return True
|
||||
if token_type == "subbeat_shift":
|
||||
if not self.in_shift and self.payload_count > 0:
|
||||
self.generated_events += 1
|
||||
self.payload_count = 0
|
||||
self.last_field_index = -1
|
||||
self.incomplete = None
|
||||
self.in_shift = True
|
||||
self.shift_run += 1
|
||||
return False
|
||||
|
||||
self.in_shift = False
|
||||
self.shift_run = 0
|
||||
self.payload_count += 1
|
||||
if token_type == "time":
|
||||
self.last_field_index = FIELD_TO_INDEX["timestamp"]
|
||||
self.incomplete = None
|
||||
elif token_type == "meter":
|
||||
self.last_field_index = FIELD_TO_INDEX["rhythm"]
|
||||
self.incomplete = "rhythm_after_meter"
|
||||
elif token_type == "eighth_position":
|
||||
self.last_field_index = FIELD_TO_INDEX["rhythm"]
|
||||
self.incomplete = None
|
||||
elif token_type == "structure":
|
||||
self.last_field_index = FIELD_TO_INDEX["structure"]
|
||||
self.incomplete = None
|
||||
elif token_type == "key":
|
||||
self.last_field_index = FIELD_TO_INDEX["key"]
|
||||
self.incomplete = None
|
||||
elif token_type == "chord_full":
|
||||
self.last_field_index = FIELD_TO_INDEX["chord"]
|
||||
self.incomplete = None
|
||||
elif token_type == "pitch":
|
||||
self.last_field_index = FIELD_TO_INDEX["melody"]
|
||||
self.incomplete = "melody_after_pitch"
|
||||
elif token_type == "duration":
|
||||
self.last_field_index = FIELD_TO_INDEX["melody"]
|
||||
self.incomplete = None
|
||||
else:
|
||||
raise RuntimeError(f"Unexpected prompt token type {token_type!r}")
|
||||
return False
|
||||
|
||||
|
||||
def inference_autocast(device, dtype=torch.bfloat16):
|
||||
if device.type != "cuda" or dtype is None:
|
||||
return nullcontext()
|
||||
return torch.autocast(device_type="cuda", dtype=dtype)
|
||||
|
||||
|
||||
def select_cache_batch(cache, indices, batch_size):
|
||||
if cache is None:
|
||||
return None
|
||||
batch_select = getattr(cache, "batch_select_indices", None)
|
||||
if callable(batch_select):
|
||||
batch_select(indices)
|
||||
return cache
|
||||
if torch.is_tensor(cache):
|
||||
if cache.ndim > 0 and cache.shape[0] == batch_size:
|
||||
return cache.index_select(0, indices)
|
||||
return cache
|
||||
if isinstance(cache, tuple):
|
||||
return tuple(select_cache_batch(item, indices, batch_size) for item in cache)
|
||||
if isinstance(cache, list):
|
||||
return [select_cache_batch(item, indices, batch_size) for item in cache]
|
||||
if isinstance(cache, dict):
|
||||
return {
|
||||
key: select_cache_batch(value, indices, batch_size)
|
||||
for key, value in cache.items()
|
||||
}
|
||||
return cache
|
||||
|
||||
|
||||
@torch.inference_mode()
|
||||
def constrained_prompt_generate_batch(
|
||||
model,
|
||||
audio,
|
||||
prompts,
|
||||
max_sequence_length,
|
||||
prefix_tokens=None,
|
||||
autocast_dtype=torch.bfloat16,
|
||||
stop_time_seconds=None,
|
||||
progress_callback=None,
|
||||
memory=None,
|
||||
step_callback=None,
|
||||
):
|
||||
tokenizer = model.tokenizer
|
||||
prefix = list(prefix_tokens) if prefix_tokens is not None else tokenizer.prompt_prefix(prompts)
|
||||
if not prefix or prefix[0] != tokenizer.sos_token:
|
||||
raise ValueError("generation prefix must begin with <|sos|>")
|
||||
if prefix[-1] == tokenizer.eos_token:
|
||||
prefix = prefix[:-1]
|
||||
if audio.ndim != 2 or audio.shape[0] < 1:
|
||||
raise ValueError(f"audio must have shape [batch, samples], got {tuple(audio.shape)}")
|
||||
states = [PromptGrammarState(tokenizer) for _ in range(audio.shape[0])]
|
||||
stop_times = [
|
||||
None if stop_time_seconds is None else float(stop_time_seconds)
|
||||
for _ in states
|
||||
]
|
||||
out_index = prefix.index(tokenizer.out_token)
|
||||
for state in states:
|
||||
for token in prefix[out_index + 1 :]:
|
||||
state.update(token)
|
||||
output_tokens = [list(prefix) for _ in states]
|
||||
active_sample_ids = list(range(audio.shape[0]))
|
||||
decoder_input = torch.tensor(
|
||||
[prefix] * audio.shape[0],
|
||||
dtype=torch.long,
|
||||
device=audio.device,
|
||||
)
|
||||
past_key_values = None
|
||||
if memory is None:
|
||||
with inference_autocast(audio.device, autocast_dtype):
|
||||
memory = model.encode(audio)
|
||||
current_length = len(prefix)
|
||||
while active_sample_ids and current_length < int(max_sequence_length):
|
||||
with inference_autocast(audio.device, autocast_dtype):
|
||||
logits, past_key_values = model.decode(
|
||||
memory,
|
||||
decoder_input,
|
||||
use_cache=True,
|
||||
past_key_values=past_key_values,
|
||||
)
|
||||
next_logits = logits[:, -1].float()
|
||||
allowed = torch.stack(
|
||||
[state.allowed(next_logits.device) for state in states],
|
||||
dim=0,
|
||||
)
|
||||
masked_logits = next_logits.masked_fill(~allowed, float("-inf"))
|
||||
next_tokens = masked_logits.argmax(dim=-1)
|
||||
if step_callback is not None:
|
||||
step_callback(current_length, active_sample_ids, next_logits, masked_logits)
|
||||
next_token_values = next_tokens.tolist()
|
||||
keep_positions = []
|
||||
for position, (sample_id, state, stop_time, token) in enumerate(
|
||||
zip(active_sample_ids, states, stop_times, next_token_values)
|
||||
):
|
||||
output_tokens[sample_id].append(int(token))
|
||||
finished = state.update(token)
|
||||
if (
|
||||
not finished
|
||||
and stop_time is not None
|
||||
and tokenizer.time_token_start <= int(token) < tokenizer.time_token_end
|
||||
and tokenizer.token_to_time_id(token) / tokenizer.time_hz >= stop_time
|
||||
):
|
||||
output_tokens[sample_id].append(tokenizer.eos_token)
|
||||
finished = True
|
||||
if not finished:
|
||||
keep_positions.append(position)
|
||||
current_length += 1
|
||||
if progress_callback is not None and current_length % 64 == 0:
|
||||
progress_callback(current_length)
|
||||
if not keep_positions:
|
||||
break
|
||||
|
||||
old_batch_size = len(active_sample_ids)
|
||||
decoder_input = next_tokens[:, None]
|
||||
if len(keep_positions) == old_batch_size:
|
||||
continue
|
||||
|
||||
keep = torch.tensor(keep_positions, dtype=torch.long, device=audio.device)
|
||||
active_sample_ids = [active_sample_ids[position] for position in keep_positions]
|
||||
states = [states[position] for position in keep_positions]
|
||||
stop_times = [stop_times[position] for position in keep_positions]
|
||||
memory = memory.index_select(0, keep)
|
||||
past_key_values = select_cache_batch(past_key_values, keep, old_batch_size)
|
||||
decoder_input = decoder_input.index_select(0, keep)
|
||||
|
||||
for tokens in output_tokens:
|
||||
if tokens[-1] != tokenizer.eos_token:
|
||||
tokens.append(tokenizer.eos_token)
|
||||
return [torch.tensor(tokens, dtype=torch.long) for tokens in output_tokens]
|
||||
|
||||
|
||||
@torch.inference_mode()
|
||||
def constrained_prompt_generate(
|
||||
model,
|
||||
audio,
|
||||
prompts,
|
||||
max_sequence_length,
|
||||
prefix_tokens=None,
|
||||
autocast_dtype=torch.bfloat16,
|
||||
stop_time_seconds=None,
|
||||
progress_callback=None,
|
||||
memory=None,
|
||||
step_callback=None,
|
||||
):
|
||||
if audio.shape[0] != 1:
|
||||
raise ValueError(
|
||||
"constrained_prompt_generate expects batch size 1; "
|
||||
"use constrained_prompt_generate_batch for batched inference"
|
||||
)
|
||||
return constrained_prompt_generate_batch(
|
||||
model,
|
||||
audio,
|
||||
prompts,
|
||||
max_sequence_length,
|
||||
prefix_tokens=prefix_tokens,
|
||||
autocast_dtype=autocast_dtype,
|
||||
stop_time_seconds=stop_time_seconds,
|
||||
progress_callback=progress_callback,
|
||||
memory=memory,
|
||||
step_callback=step_callback,
|
||||
)[0]
|
||||
|
||||
|
||||
def decode_generated_tokens(tokenizer, tokens, song_id, window_index=None):
|
||||
try:
|
||||
return tokenizer.decode_sequence(tokens, strict=True), None
|
||||
except ValueError as exc:
|
||||
recoverable_errors = (
|
||||
"empty event at subbeat",
|
||||
"belongs to inactive output field",
|
||||
)
|
||||
if not any(message in str(exc) for message in recoverable_errors):
|
||||
raise
|
||||
decoded = tokenizer.decode_sequence(tokens, strict=False)
|
||||
warning = {
|
||||
"song_id": song_id,
|
||||
"warning": "strict decode failed; recovered in non-strict decode",
|
||||
"error": str(exc),
|
||||
}
|
||||
if window_index is not None:
|
||||
warning["window_index"] = int(window_index)
|
||||
return decoded, warning
|
||||
|
||||
|
||||
def event_time_map(decoded, target_seconds):
|
||||
anchors = []
|
||||
for event in decoded["events"]:
|
||||
value = event["values"].get("timestamp")
|
||||
if value is not None:
|
||||
anchors.append((int(event["subbeat"]), float(value)))
|
||||
if not anchors:
|
||||
return lambda step: min(float(target_seconds), max(0.0, float(step) * 0.125))
|
||||
anchors = sorted(dict(anchors).items())
|
||||
steps = np.asarray([item[0] for item in anchors], dtype=np.float64)
|
||||
times = np.asarray([item[1] for item in anchors], dtype=np.float64)
|
||||
if len(anchors) >= 2:
|
||||
step_seconds = float(np.median(np.diff(times) / np.maximum(np.diff(steps), 1)))
|
||||
if not np.isfinite(step_seconds) or step_seconds <= 0:
|
||||
step_seconds = 0.125
|
||||
else:
|
||||
step_seconds = 0.125
|
||||
|
||||
def lookup(step):
|
||||
step = float(step)
|
||||
if step <= steps[0]:
|
||||
return float(np.clip(times[0] + (step - steps[0]) * step_seconds, 0, target_seconds))
|
||||
if step >= steps[-1]:
|
||||
return float(np.clip(times[-1] + (step - steps[-1]) * step_seconds, 0, target_seconds))
|
||||
return float(np.interp(step, steps, times))
|
||||
|
||||
return lookup
|
||||
|
||||
|
||||
def field_text(event, tokenizer):
|
||||
parts = []
|
||||
for field in tokenizer.event_field_order:
|
||||
value = event["values"].get(field)
|
||||
if value is None:
|
||||
continue
|
||||
if field == "melody":
|
||||
note_parts = []
|
||||
for note in value:
|
||||
duration = note["duration_bin"]
|
||||
note_parts.append(
|
||||
f"pitch={note['pitch']}:track={note['track']}:dur_bin={duration}:dur_steps={note['duration_steps']}"
|
||||
)
|
||||
parts.append("melody=[" + ",".join(note_parts) + "]")
|
||||
elif isinstance(value, dict):
|
||||
parts.append(field + "=" + ",".join(f"{k}:{v}" for k, v in value.items()))
|
||||
else:
|
||||
parts.append(f"{field}={value}")
|
||||
return "; ".join(parts)
|
||||
|
||||
|
||||
def event_start_time(event, time_lookup):
|
||||
if "time" in event:
|
||||
return float(event["time"])
|
||||
return float(time_lookup(event["subbeat"]))
|
||||
|
||||
|
||||
def write_events_tsv(decoded, tokenizer, time_lookup, path):
|
||||
with Path(path).open("w", encoding="utf-8") as f:
|
||||
f.write("event_index\tsubbeat\ttime\tfields\ttokens\n")
|
||||
for index, event in enumerate(decoded["events"]):
|
||||
token_text = " ".join(
|
||||
tokenizer.describe(token)
|
||||
for field in tokenizer.event_field_order
|
||||
for token in event["tokens_by_field"].get(field, ())
|
||||
)
|
||||
f.write(
|
||||
f"{index}\t{event['subbeat']}\t{event_start_time(event, time_lookup):.6f}\t"
|
||||
f"{field_text(event, tokenizer)}\t{token_text}\n"
|
||||
)
|
||||
|
||||
|
||||
def clone_decoded_event(event):
|
||||
return {
|
||||
"subbeat": int(event["subbeat"]),
|
||||
"tokens_by_field": {
|
||||
field: [int(token) for token in tokens]
|
||||
for field, tokens in event["tokens_by_field"].items()
|
||||
},
|
||||
"values": copy.deepcopy(event["values"]),
|
||||
}
|
||||
|
||||
|
||||
def refresh_event_values(event, tokenizer, prompts):
|
||||
event["values"] = {
|
||||
field: tokenizer._decode_field(field, tokens, prompts)
|
||||
for field, tokens in event["tokens_by_field"].items()
|
||||
if tokens
|
||||
}
|
||||
|
||||
|
||||
def set_event_local_timestamp(event, tokenizer, local_time):
|
||||
time_id = int(round(float(local_time) * tokenizer.time_hz))
|
||||
time_id = max(0, min(time_id, tokenizer.n_time_tokens - 1))
|
||||
event["tokens_by_field"]["timestamp"] = [tokenizer.time_id_to_token(time_id)]
|
||||
|
||||
|
||||
def active_context_before(events, tokenizer, time_abs):
|
||||
state = {}
|
||||
for event in events:
|
||||
event_time = event.get("time")
|
||||
if event_time is None or float(event_time) > float(time_abs) + 1e-6:
|
||||
continue
|
||||
for field in ("structure", "key", "chord"):
|
||||
tokens = event["tokens_by_field"].get(field)
|
||||
if tokens:
|
||||
state[field] = [int(token) for token in tokens]
|
||||
rhythm_tokens = event["tokens_by_field"].get("rhythm", ())
|
||||
meter_tokens = [
|
||||
int(token)
|
||||
for token in rhythm_tokens
|
||||
if tokenizer.token_type(token) == "meter"
|
||||
]
|
||||
if meter_tokens:
|
||||
state["meter"] = meter_tokens[:1]
|
||||
return state
|
||||
|
||||
|
||||
def apply_prefix_context(event, context, tokenizer, prompts):
|
||||
for field in ("structure", "key", "chord"):
|
||||
if field not in event["tokens_by_field"] and field in context:
|
||||
event["tokens_by_field"][field] = list(context[field])
|
||||
|
||||
rhythm_tokens = list(event["tokens_by_field"].get("rhythm", ()))
|
||||
has_meter = any(tokenizer.token_type(token) == "meter" for token in rhythm_tokens)
|
||||
has_eighth = any(
|
||||
tokenizer.token_type(token) == "eighth_position" for token in rhythm_tokens
|
||||
)
|
||||
if has_eighth and not has_meter and "meter" in context:
|
||||
event["tokens_by_field"]["rhythm"] = list(context["meter"]) + rhythm_tokens
|
||||
|
||||
refresh_event_values(event, tokenizer, prompts)
|
||||
|
||||
|
||||
def build_overlap_prefix_tokens(
|
||||
stitched_events,
|
||||
tokenizer,
|
||||
prompts,
|
||||
window_start,
|
||||
prefix_end,
|
||||
):
|
||||
eps = 1e-4
|
||||
source_events = [
|
||||
event
|
||||
for event in stitched_events
|
||||
if float(window_start) - eps <= float(event.get("time", -1.0)) < float(prefix_end) - eps
|
||||
]
|
||||
source_events.sort(
|
||||
key=lambda event: (
|
||||
int(
|
||||
event.get(
|
||||
"global_subbeat",
|
||||
event.get("source_subbeat", event["subbeat"]),
|
||||
)
|
||||
),
|
||||
float(event.get("time", 0.0)),
|
||||
)
|
||||
)
|
||||
first_beat_index = next(
|
||||
(
|
||||
index
|
||||
for index, event in enumerate(source_events)
|
||||
if "timestamp" in event["values"] or "rhythm" in event["values"]
|
||||
),
|
||||
None,
|
||||
)
|
||||
if first_beat_index is None:
|
||||
return None, None, None
|
||||
|
||||
source_events = source_events[first_beat_index:]
|
||||
base_subbeat = int(
|
||||
source_events[0].get(
|
||||
"global_subbeat",
|
||||
source_events[0].get("source_subbeat", source_events[0]["subbeat"]),
|
||||
)
|
||||
)
|
||||
context = active_context_before(
|
||||
stitched_events,
|
||||
tokenizer,
|
||||
source_events[0]["time"],
|
||||
)
|
||||
prefix_events = []
|
||||
for source in source_events:
|
||||
event = clone_decoded_event(source)
|
||||
source_subbeat = int(
|
||||
source.get(
|
||||
"global_subbeat",
|
||||
source.get("source_subbeat", source["subbeat"]),
|
||||
)
|
||||
)
|
||||
event["subbeat"] = max(0, source_subbeat - base_subbeat)
|
||||
if "timestamp" in event["tokens_by_field"]:
|
||||
set_event_local_timestamp(
|
||||
event,
|
||||
tokenizer,
|
||||
float(source["time"]) - float(window_start),
|
||||
)
|
||||
refresh_event_values(event, tokenizer, prompts)
|
||||
prefix_events.append(event)
|
||||
|
||||
apply_prefix_context(prefix_events[0], context, tokenizer, prompts)
|
||||
prefix_decoded = {
|
||||
"schema_version": tokenizer.schema_version,
|
||||
"prompts": prompts,
|
||||
"events": prefix_events,
|
||||
"has_eos": False,
|
||||
}
|
||||
return (
|
||||
prefix_decoded,
|
||||
tokenizer.encode_decoded_sequence(prefix_decoded),
|
||||
base_subbeat,
|
||||
)
|
||||
|
||||
|
||||
def stitched_window_events(
|
||||
decoded,
|
||||
time_lookup,
|
||||
window_start,
|
||||
accept_start,
|
||||
accept_end,
|
||||
song_duration,
|
||||
window_index,
|
||||
global_subbeat_base=0,
|
||||
):
|
||||
accepted = []
|
||||
eps = 1e-4
|
||||
for event in decoded["events"]:
|
||||
local_time = float(time_lookup(event["subbeat"]))
|
||||
abs_time = float(window_start) + local_time
|
||||
if abs_time < float(accept_start) - eps:
|
||||
continue
|
||||
if abs_time >= float(accept_end) - eps or abs_time >= float(song_duration) - eps:
|
||||
continue
|
||||
|
||||
output = clone_decoded_event(event)
|
||||
output["time"] = float(np.clip(abs_time, 0.0, song_duration))
|
||||
output["window_index"] = int(window_index)
|
||||
output["window_start"] = float(window_start)
|
||||
output["source_subbeat"] = int(event["subbeat"])
|
||||
output["global_subbeat"] = int(global_subbeat_base) + int(event["subbeat"])
|
||||
if "timestamp" in output["values"]:
|
||||
output["values"]["timestamp"] = output["time"]
|
||||
|
||||
notes = output["values"].get("melody")
|
||||
if notes is not None:
|
||||
fixed_notes = []
|
||||
for note in notes:
|
||||
fixed_note = copy.deepcopy(note)
|
||||
duration_steps = int(fixed_note["duration_steps"])
|
||||
local_end = float(time_lookup(int(event["subbeat"]) + duration_steps))
|
||||
end_time = float(window_start) + local_end
|
||||
end_time = min(
|
||||
float(song_duration),
|
||||
max(output["time"] + 0.04, end_time),
|
||||
)
|
||||
fixed_note["end_time"] = end_time
|
||||
fixed_notes.append(fixed_note)
|
||||
output["values"]["melody"] = fixed_notes
|
||||
accepted.append(output)
|
||||
return accepted
|
||||
|
||||
|
||||
def write_window_tokens(path, windows, tokenizer):
|
||||
with Path(path).open("w", encoding="utf-8") as f:
|
||||
for window in windows:
|
||||
f.write(
|
||||
f"# window_index={window['window_index']} "
|
||||
f"start={window['start']:.6f} end={window['end']:.6f} "
|
||||
f"prefix_end={window['prefix_end']:.6f} "
|
||||
f"accept=[{window['accept_start']:.6f},{window['accept_end']:.6f}) "
|
||||
f"generation_stop={window['generation_stop']} "
|
||||
f"prefix_tokens={window['prefix_tokens']} "
|
||||
f"tokens={int(window['tokens'].numel())}\n"
|
||||
)
|
||||
for index, token in enumerate(window["tokens"].tolist()):
|
||||
f.write(f"{index}\t{token}\t{tokenizer.describe(token)}\n")
|
||||
f.write("\n")
|
||||
|
||||
|
||||
@torch.inference_mode()
|
||||
def generate(model, input_values, prompts=None, *, attention_mask=None, max_length=None,
|
||||
max_new_tokens=None, prefix_tokens=None,
|
||||
autocast_dtype=torch.bfloat16, stop_time_seconds=None,
|
||||
return_dict_in_generate=False, output_logits=False, output_scores=False,
|
||||
progress_callback=None):
|
||||
"""Generate symbolic token sequences with the model's event grammar.
|
||||
|
||||
Inputs are 24 kHz waveforms [batch, samples]. Scores/logits are per generated
|
||||
step, after the prompt prefix, with completed batch items filled by NaNs.
|
||||
"""
|
||||
from transformers.utils import ModelOutput
|
||||
if input_values.ndim == 1:
|
||||
input_values = input_values.unsqueeze(0)
|
||||
if attention_mask is not None and attention_mask.ndim == 1:
|
||||
attention_mask = attention_mask.unsqueeze(0)
|
||||
input_values = input_values.to(next(model.parameters()).device)
|
||||
prompts = model.tokenizer.normalize_prompts(prompts or FULL_TASK_PROMPTS)
|
||||
prefix = list(prefix_tokens) if prefix_tokens is not None else model.tokenizer.prompt_prefix(prompts)
|
||||
prefix_length = len(prefix) - (1 if prefix and prefix[-1] == model.tokenizer.eos_token else 0)
|
||||
if max_new_tokens is not None:
|
||||
if max_length is not None:
|
||||
raise ValueError("Specify max_new_tokens or max_length, not both.")
|
||||
if not isinstance(max_new_tokens, int) or isinstance(max_new_tokens, bool) or max_new_tokens < 1:
|
||||
raise ValueError("max_new_tokens must be a positive integer.")
|
||||
max_length = prefix_length + max_new_tokens
|
||||
limit = model.max_output_seq_len if max_length is None else max_length
|
||||
if not isinstance(limit, int) or isinstance(limit, bool) or not prefix_length < limit <= model.max_output_seq_len:
|
||||
raise ValueError("The token limit must exceed the prefix length and fit the decoder context.")
|
||||
memory = None
|
||||
if attention_mask is not None:
|
||||
with inference_autocast(input_values.device, autocast_dtype):
|
||||
memory = model.get_audio_features(input_values, attention_mask=attention_mask).encoder_last_hidden_state
|
||||
raw, scores, positions = [], [], []
|
||||
batch_size = input_values.shape[0]
|
||||
def capture(position, ids, logits, masked):
|
||||
positions.append(position)
|
||||
for enabled, values, output in ((output_logits, logits, raw), (output_scores, masked, scores)):
|
||||
if enabled:
|
||||
row = torch.full((batch_size, values.shape[-1]), float('nan'), dtype=torch.float32)
|
||||
row[ids] = values.detach().cpu()
|
||||
output.append(row)
|
||||
tokens = constrained_prompt_generate_batch(
|
||||
model, input_values, prompts, limit,
|
||||
prefix_tokens=prefix_tokens, autocast_dtype=autocast_dtype,
|
||||
stop_time_seconds=stop_time_seconds, progress_callback=progress_callback,
|
||||
step_callback=capture if output_logits or output_scores else None,
|
||||
memory=memory,
|
||||
)
|
||||
sequences = torch.nn.utils.rnn.pad_sequence(tokens, batch_first=True, padding_value=model.tokenizer.pad_token)
|
||||
if return_dict_in_generate or output_logits or output_scores:
|
||||
return ModelOutput(sequences=sequences, logits=tuple(raw) if output_logits else None,
|
||||
scores=tuple(scores) if output_scores else None,
|
||||
token_positions=torch.tensor(positions, dtype=torch.long))
|
||||
return sequences
|
||||
@@ -0,0 +1,78 @@
|
||||
"""Transcribe a music recording with SheetSage2."""
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
SCRIPT_DIR = Path(__file__).absolute().parent
|
||||
sys.path.insert(0, str(SCRIPT_DIR))
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="SheetSage2: music audio to ABC, MIDI, annotations and features")
|
||||
parser.add_argument("audio", type=Path)
|
||||
parser.add_argument("--output", type=Path, default=Path("output"))
|
||||
local = SCRIPT_DIR
|
||||
parser.add_argument("--model", default=str(local) if (local / "config.json").exists() else "m-a-p/SheetSage2")
|
||||
parser.add_argument("--revision", help="Model and code commit on Hugging Face")
|
||||
parser.add_argument("--device", default="auto")
|
||||
parser.add_argument("--dtype", choices=("bf16", "fp32"), default="bf16")
|
||||
parser.add_argument("--preset", choices=("default", "paper"), default="default")
|
||||
parser.add_argument("--max-seconds", type=float)
|
||||
parser.add_argument("--overlap", type=float)
|
||||
parser.add_argument("--lookahead", type=float)
|
||||
parser.add_argument("--prompts", nargs="+")
|
||||
parser.add_argument("--melody-only", action="store_true",
|
||||
help="Keep vocal and instrumental melodies; omit chords from ABC and playback")
|
||||
parser.add_argument("--export-logits", action="store_true")
|
||||
parser.add_argument("--export-scores", action="store_true")
|
||||
parser.add_argument("--export-embeddings", action="store_true")
|
||||
parser.add_argument("--all-layers", action="store_true", help="Also export all 24 MERT block features")
|
||||
parser.add_argument("--render-audio", action="store_true")
|
||||
parser.add_argument("--render-score", nargs="?", const="pdf", default=False, help="pdf,svg,png (default: pdf)")
|
||||
parser.add_argument("--render-parts", default="mix", help="mix,melody,vocal,instrumental,chords,all")
|
||||
parser.add_argument("--local-files-only", action="store_true")
|
||||
args = parser.parse_args()
|
||||
if args.render_audio or args.render_score:
|
||||
from rendering_sheetsage2 import validate_render_options
|
||||
try:
|
||||
validate_render_options(audio=args.render_audio, score=args.render_score, parts=args.render_parts)
|
||||
except ValueError as exc:
|
||||
parser.error(str(exc))
|
||||
import torch
|
||||
from transformers import AutoModel
|
||||
torch.set_num_threads(min(4, torch.get_num_threads()))
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu" if args.device == "auto" else args.device
|
||||
if args.device != "auto":
|
||||
device = args.device
|
||||
model = AutoModel.from_pretrained(
|
||||
args.model, revision=args.revision, code_revision=args.revision,
|
||||
local_files_only=args.local_files_only, trust_remote_code=True,
|
||||
).eval().to(device)
|
||||
def progress(value):
|
||||
if value["stage"] == "encoding":
|
||||
print(f"Window {value['window']}/{value['windows']}", flush=True)
|
||||
options = dict(dtype=args.dtype, preset=args.preset, max_seconds=args.max_seconds,
|
||||
overlap_seconds=args.overlap, lookahead_seconds=args.lookahead,
|
||||
export_logits=args.export_logits, export_scores=args.export_scores,
|
||||
export_embeddings=args.export_embeddings, output_hidden_states=args.all_layers,
|
||||
render_audio=args.render_audio, render_score=args.render_score,
|
||||
render_parts=tuple(args.render_parts.split(",")), progress=progress)
|
||||
if args.prompts:
|
||||
options["prompts"] = args.prompts
|
||||
if args.melody_only:
|
||||
options["melody_only"] = True
|
||||
try:
|
||||
result = model.transcribe(args.audio, output_dir=args.output, **options)
|
||||
except (ValueError, RuntimeError, FileNotFoundError) as exc:
|
||||
parser.exit(1, f"SheetSage2: {exc}\n")
|
||||
if args.melody_only and (result.get("abc_error") or not result.get("abc")):
|
||||
reason = result.get("abc_error") or "no ABC score was produced"
|
||||
parser.exit(1, f"SheetSage2: melody-only ABC unavailable: {reason}. "
|
||||
f"Transcription outputs were saved to {args.output.resolve()}.\n")
|
||||
print(f"Saved transcription to {args.output.resolve()}")
|
||||
if result.get("abc_error"):
|
||||
print(f"ABC unavailable: {result['abc_error']}. MIDI and annotations were saved.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,67 @@
|
||||
import os
|
||||
import stat
|
||||
import tempfile
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def _target_path(path) -> Path:
|
||||
target = Path(path)
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
return target
|
||||
|
||||
|
||||
def _flush_path(path: Path) -> None:
|
||||
with path.open("rb+") as handle:
|
||||
os.fsync(handle.fileno())
|
||||
|
||||
|
||||
def _replacement_mode(target: Path) -> int:
|
||||
try:
|
||||
return stat.S_IMODE(target.stat().st_mode)
|
||||
except FileNotFoundError:
|
||||
mask = os.umask(0)
|
||||
os.umask(mask)
|
||||
return 0o666 & ~mask
|
||||
|
||||
|
||||
@contextmanager
|
||||
def atomic_output_path(path):
|
||||
target = _target_path(path)
|
||||
mode = _replacement_mode(target)
|
||||
fd, tmp_name = tempfile.mkstemp(
|
||||
# Keep the temporary basename independent of the final basename. Some
|
||||
# annotation IDs are already close to NAME_MAX; repeating the target
|
||||
# name here made an otherwise valid final path fail with ENAMETOOLONG.
|
||||
prefix=".tmp_",
|
||||
suffix=".tmp",
|
||||
dir=str(target.parent),
|
||||
)
|
||||
os.close(fd)
|
||||
tmp_path = Path(tmp_name)
|
||||
try:
|
||||
yield str(tmp_path)
|
||||
os.chmod(tmp_path, mode)
|
||||
_flush_path(tmp_path)
|
||||
os.replace(tmp_path, target)
|
||||
except Exception:
|
||||
try:
|
||||
tmp_path.unlink()
|
||||
except FileNotFoundError:
|
||||
pass
|
||||
raise
|
||||
|
||||
|
||||
def atomic_write_text(path, text: str, encoding: str = "utf-8", newline: str = "\n") -> str:
|
||||
with atomic_output_path(path) as tmp_path:
|
||||
with open(tmp_path, "w", encoding=encoding, newline=newline) as handle:
|
||||
handle.write(text)
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
return str(path)
|
||||
|
||||
|
||||
def atomic_write_pretty_midi(midi, path) -> str:
|
||||
with atomic_output_path(path) as tmp_path:
|
||||
midi.write(tmp_path)
|
||||
return str(path)
|
||||
@@ -0,0 +1,29 @@
|
||||
|
||||
|
||||
|
||||
STRUCTURE_LABELS = ['silence', 'intro', 'outro', 'verse', 'chorus', 'bridge', 'pre-chorus', 'post-chorus',
|
||||
'interlude', 'fade-out', 'loop', 'rap', 'preshot', 'irregular']
|
||||
|
||||
|
||||
STRUCTURE_LABELS = ['silence', 'intro', 'outro', 'verse', 'chorus', 'bridge', 'pre-chorus', 'post-chorus',
|
||||
'interlude', 'fade-out', 'loop', 'rap', 'preshot', 'irregular', 'instrumental',
|
||||
'intro and verse', 'pre-chorus and chorus', 'verse and pre-chorus', 'solo', 'theme', 'development', 'variation', 'pre-outro']
|
||||
|
||||
|
||||
MIREX_STRUCTURE_LABEL_REDUCTION = {
|
||||
'silence': 'silence',
|
||||
'intro': 'intro',
|
||||
'outro': 'outro',
|
||||
'verse': 'verse',
|
||||
'chorus': 'chorus',
|
||||
'bridge': 'bridge',
|
||||
'pre-chorus': 'verse',
|
||||
'post-chorus': 'verse',
|
||||
'interlude': 'inst',
|
||||
'inst': 'inst',
|
||||
'fade-out': 'outro',
|
||||
'loop': 'chorus',
|
||||
'rap': 'verse',
|
||||
'preshot': 'inst',
|
||||
'irregular': 'verse',
|
||||
}
|
||||
@@ -0,0 +1,92 @@
|
||||
"""Original-time MIDI playback and a separate score-to-audio measure map."""
|
||||
import json
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
|
||||
import mir_eval.chord
|
||||
import numpy as np
|
||||
import pretty_midi
|
||||
|
||||
from .io_sheetsage2 import atomic_write_text
|
||||
|
||||
|
||||
def chord_pitches(label):
|
||||
if label in {"N", "X", "?"}:
|
||||
return []
|
||||
root, bitmap, bass = mir_eval.chord.encode(label, reduce_extended_chords=True)
|
||||
if root < 0:
|
||||
return []
|
||||
# Keep inversion bass below the chord; no ABC chord-name reinterpretation.
|
||||
upper = [48 + root + int(interval) for interval in np.flatnonzero(bitmap > 0)]
|
||||
return sorted(set([36 + (root + bass) % 12] + upper))
|
||||
|
||||
|
||||
def measure_map(score, duration):
|
||||
rows, position = [], 0.0
|
||||
for measure in score.measures:
|
||||
length = measure.abc_numerator / measure.abc_denominator
|
||||
actual = measure.numerator / measure.denominator
|
||||
start = float(score.beats[measure.start_beat].time)
|
||||
end = float(score.beats[measure.end_beat].time)
|
||||
rows.append(dict(index=measure.index, start=start, end=min(end, duration),
|
||||
score_start=position, score_end=position + length,
|
||||
leading_rest=length - actual if measure.pad_before else 0.0,
|
||||
trailing_rest=length - actual if not measure.pad_before else 0.0))
|
||||
position += length
|
||||
return rows
|
||||
|
||||
|
||||
def midi_bytes(midi):
|
||||
"""Serialize a PrettyMIDI object without touching the filesystem."""
|
||||
stream = BytesIO()
|
||||
midi.write(stream)
|
||||
return stream.getvalue()
|
||||
|
||||
|
||||
def build_playback(melody_midi, chord_rows, score, duration):
|
||||
"""Return playback metadata and MIDI bytes, preserving MIDI tick rounding."""
|
||||
if isinstance(melody_midi, pretty_midi.PrettyMIDI):
|
||||
melody_midi = midi_bytes(melody_midi)
|
||||
midi = pretty_midi.PrettyMIDI(BytesIO(melody_midi))
|
||||
chord_track = pretty_midi.Instrument(0, name="Chords")
|
||||
warnings = []
|
||||
for start, end, label in chord_rows:
|
||||
start, end = max(0.0, float(start)), min(duration, float(end))
|
||||
try:
|
||||
pitches = chord_pitches(label)
|
||||
except mir_eval.chord.InvalidChordException as exc:
|
||||
warnings.append(f"Chord playback skipped {label}: {exc}")
|
||||
continue
|
||||
# Rearticulate long chord spans at downbeats, including repeated bars.
|
||||
downbeats = [b.time for b in score.beats if b.beat_id == 1] if score is not None else []
|
||||
cuts = [start] + [time for time in downbeats if start < time < end] + [end]
|
||||
for a, b in zip(cuts, cuts[1:]):
|
||||
if b <= a:
|
||||
continue
|
||||
for pitch in pitches:
|
||||
chord_track.notes.append(pretty_midi.Note(48, pitch, a, b))
|
||||
midi.instruments.append(chord_track)
|
||||
transcription = midi_bytes(midi)
|
||||
chords = pretty_midi.PrettyMIDI(resolution=midi.resolution)
|
||||
chords.instruments.append(chord_track)
|
||||
midis = {"transcription": transcription, "chords": midi_bytes(chords)}
|
||||
# Decode the serialized bytes, so playback follows the returned MIDI exactly.
|
||||
midi = pretty_midi.PrettyMIDI(BytesIO(transcription))
|
||||
tracks = [dict(name=instrument.name, program=int(instrument.program),
|
||||
notes=[dict(pitch=int(n.pitch), start=float(n.start), end=float(n.end), velocity=int(n.velocity))
|
||||
for n in instrument.notes]) for instrument in midi.instruments]
|
||||
data = dict(version=1, duration=duration, midi="transcription.mid", tracks=tracks,
|
||||
measures=measure_map(score, duration) if score is not None else [], warnings=warnings)
|
||||
return data, midis
|
||||
|
||||
|
||||
def export_playback(directory, score, duration):
|
||||
"""Write playback files for an existing directory of exported annotations."""
|
||||
directory = Path(directory)
|
||||
chord_path = directory / "chord.lab"
|
||||
rows = [line.split(maxsplit=2) for line in chord_path.read_text(encoding="utf-8").splitlines()] if chord_path.exists() else []
|
||||
data, midis = build_playback((directory / "melody.mid").read_bytes(), rows, score, duration)
|
||||
for name, content in midis.items():
|
||||
(directory / f"{name}.mid").write_bytes(content)
|
||||
atomic_write_text(directory / "playback.json", json.dumps(data, indent=2))
|
||||
return data
|
||||
@@ -0,0 +1,361 @@
|
||||
"""Standalone MERT2 waveform-to-representation inference."""
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Optional, Tuple
|
||||
|
||||
import torch
|
||||
from torch import nn
|
||||
from torch.nn import functional as F
|
||||
from torchaudio.transforms import AmplitudeToDB, MelScale, Spectrogram
|
||||
from transformers import PreTrainedModel
|
||||
from transformers.utils import ModelOutput
|
||||
|
||||
from .configuration_mert2 import MERT2Config
|
||||
|
||||
|
||||
@dataclass
|
||||
class MERT2ModelOutput(ModelOutput):
|
||||
"""Frame representations and their validity mask.
|
||||
|
||||
``hidden_states`` contains one tensor per Conformer block, without an input
|
||||
embedding. ``feature_attention_mask`` is True at valid output frames.
|
||||
"""
|
||||
|
||||
last_hidden_state: Optional[torch.FloatTensor] = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor, ...]] = None
|
||||
feature_attention_mask: Optional[torch.BoolTensor] = None
|
||||
|
||||
|
||||
class MERT2MelFrontend(nn.Module):
|
||||
"""Power log-mel features with fixed, checkpoint-specific normalization."""
|
||||
|
||||
def __init__(self, config):
|
||||
super().__init__()
|
||||
# Filter construction requires real tensors even during meta loading.
|
||||
with torch.device("cpu"):
|
||||
self.register_buffer("mel_mean", torch.zeros(config.num_mel_bins, dtype=torch.float32))
|
||||
self.register_buffer("mel_std", torch.ones(config.num_mel_bins, dtype=torch.float32))
|
||||
self.spectrogram = Spectrogram(
|
||||
n_fft=config.n_fft,
|
||||
win_length=config.win_length,
|
||||
hop_length=config.hop_length,
|
||||
power=2.0,
|
||||
).float()
|
||||
self.mel_scale = MelScale(
|
||||
n_mels=config.num_mel_bins,
|
||||
sample_rate=config.sampling_rate,
|
||||
n_stft=config.n_fft // 2 + 1,
|
||||
).float()
|
||||
self.amplitude_to_db = AmplitudeToDB(stype="power", top_db=None)
|
||||
|
||||
def _apply(self, fn, recurse=True):
|
||||
# Preserve the original float32 values, including on model.half().
|
||||
saved = [(module, name, value) for module in self.modules() for name, value in module._buffers.items() if value is not None]
|
||||
super()._apply(fn, recurse=recurse)
|
||||
for module, name, original in saved:
|
||||
current = module._buffers[name]
|
||||
if original.is_floating_point():
|
||||
module._buffers[name] = (
|
||||
current.float() if original.is_meta else original.to(device=current.device, dtype=torch.float32)
|
||||
)
|
||||
return self
|
||||
|
||||
@torch.no_grad()
|
||||
def forward(self, waveform):
|
||||
with torch.autocast(device_type=waveform.device.type, enabled=False):
|
||||
spectrum = self.spectrogram(waveform.float())
|
||||
mel = self.amplitude_to_db(self.mel_scale(spectrum))
|
||||
mel = mel[..., :-1].transpose(-1, -2)
|
||||
return (mel - self.mel_mean) / self.mel_std.clamp_min(1e-5)
|
||||
|
||||
|
||||
class Transpose(nn.Module):
|
||||
def forward(self, hidden_states):
|
||||
return hidden_states.transpose(1, 2)
|
||||
|
||||
|
||||
class GlobalResponseNorm(nn.Module):
|
||||
def __init__(self, dim):
|
||||
super().__init__()
|
||||
self.weight = nn.Parameter(torch.zeros(1, 1, dim))
|
||||
self.bias = nn.Parameter(torch.zeros(1, 1, dim))
|
||||
|
||||
def forward(self, hidden_states):
|
||||
magnitude = torch.norm(hidden_states, p=2, dim=1, keepdim=True)
|
||||
normalized = magnitude / (magnitude.mean(dim=-1, keepdim=True) + 1e-6)
|
||||
return self.weight * (hidden_states * normalized) + self.bias + hidden_states
|
||||
|
||||
|
||||
class ConvNextLayer(nn.Module):
|
||||
def __init__(self, dim, eps):
|
||||
super().__init__()
|
||||
self.depthwise_block = nn.Sequential(
|
||||
Transpose(), nn.Conv1d(dim, dim, 7, padding=3, groups=dim), Transpose()
|
||||
)
|
||||
self.pointwise_block = nn.Sequential(
|
||||
nn.LayerNorm(dim, eps=eps),
|
||||
nn.Linear(dim, 4 * dim),
|
||||
nn.GELU(),
|
||||
GlobalResponseNorm(4 * dim),
|
||||
nn.Linear(4 * dim, dim),
|
||||
)
|
||||
|
||||
def forward(self, hidden_states):
|
||||
return hidden_states + self.pointwise_block(self.depthwise_block(hidden_states))
|
||||
|
||||
|
||||
class ConvNextBlock(nn.Module):
|
||||
def __init__(self, in_channels, out_channels, stride, depth, eps):
|
||||
super().__init__()
|
||||
self.resampling_layer = (
|
||||
nn.Sequential(
|
||||
nn.LayerNorm(in_channels, eps=eps),
|
||||
Transpose(),
|
||||
nn.Conv1d(in_channels, out_channels, 2, stride=stride),
|
||||
Transpose(),
|
||||
)
|
||||
if in_channels != out_channels or stride > 1 else nn.Identity()
|
||||
)
|
||||
self.convnext_layers = nn.Sequential(*[ConvNextLayer(out_channels, eps) for _ in range(depth)])
|
||||
|
||||
def forward(self, hidden_states):
|
||||
return self.convnext_layers(self.resampling_layer(hidden_states))
|
||||
|
||||
|
||||
class RotaryEmbedding(nn.Module):
|
||||
def __init__(self, config):
|
||||
super().__init__()
|
||||
self.head_dim = config.hidden_size // config.num_attention_heads
|
||||
self.base = config.rotary_embedding_base
|
||||
with torch.device("cpu"):
|
||||
inverse_frequency = 1.0 / (
|
||||
self.base ** (torch.arange(0, self.head_dim, 2, dtype=torch.float32) / self.head_dim)
|
||||
)
|
||||
self.register_buffer("inv_freq", inverse_frequency, persistent=False)
|
||||
self._sequence_length = 0
|
||||
self._cache_device = None
|
||||
self._cos = None
|
||||
self._sin = None
|
||||
|
||||
def _apply(self, fn, recurse=True):
|
||||
original = self.inv_freq
|
||||
super()._apply(fn, recurse=recurse)
|
||||
self.inv_freq = original.to(device=self.inv_freq.device, dtype=torch.float32)
|
||||
return self
|
||||
|
||||
def forward(self, hidden_states):
|
||||
length = hidden_states.shape[1]
|
||||
if self._cos is None or self._cache_device != hidden_states.device or length > self._sequence_length:
|
||||
positions = torch.arange(length, device=hidden_states.device, dtype=self.inv_freq.dtype)
|
||||
frequencies = torch.einsum("i,j->ij", positions, self.inv_freq)
|
||||
angles = torch.cat((frequencies, frequencies), dim=-1)
|
||||
self._cos = angles.cos()[:, None, None, :]
|
||||
self._sin = angles.sin()[:, None, None, :]
|
||||
self._sequence_length = length
|
||||
self._cache_device = hidden_states.device
|
||||
return (
|
||||
self._cos[:length].to(hidden_states.dtype).permute(1, 0, 2, 3),
|
||||
self._sin[:length].to(hidden_states.dtype).permute(1, 0, 2, 3),
|
||||
)
|
||||
|
||||
|
||||
def rotate_half(value):
|
||||
first, second = value.chunk(2, dim=-1)
|
||||
return torch.cat((-second, first), dim=-1)
|
||||
|
||||
|
||||
class SelfAttention(nn.Module):
|
||||
def __init__(self, config):
|
||||
super().__init__()
|
||||
self.config = config
|
||||
self.num_heads = config.num_attention_heads
|
||||
self.head_dim = config.hidden_size // self.num_heads
|
||||
self.query_proj = nn.Linear(config.hidden_size, config.hidden_size)
|
||||
self.key_proj = nn.Linear(config.hidden_size, config.hidden_size)
|
||||
self.value_proj = nn.Linear(config.hidden_size, config.hidden_size)
|
||||
self.out_proj = nn.Linear(config.hidden_size, config.hidden_size)
|
||||
|
||||
def forward(self, hidden_states, position_embeddings):
|
||||
batch, time, width = hidden_states.shape
|
||||
shape = (batch, time, self.num_heads, self.head_dim)
|
||||
query = self.query_proj(hidden_states).reshape(shape)
|
||||
key = self.key_proj(hidden_states).reshape(shape)
|
||||
value = self.value_proj(hidden_states).reshape(shape)
|
||||
cos, sin = position_embeddings
|
||||
query = query * cos + rotate_half(query) * sin
|
||||
key = key * cos + rotate_half(key) * sin
|
||||
if self.config._attn_implementation == "flash_attention_2":
|
||||
if query.device.type != "cuda" or query.dtype not in (torch.float16, torch.bfloat16):
|
||||
raise ValueError("flash_attention_2 requires CUDA and float16/bfloat16 activations; use autocast or load a reduced-precision model.")
|
||||
try:
|
||||
from flash_attn import flash_attn_func
|
||||
except ImportError as error:
|
||||
raise ImportError("Install flash-attn to use flash_attention_2, or select attn_implementation='sdpa'.") from error
|
||||
attended = flash_attn_func(query.contiguous(), key.contiguous(), value.contiguous(), dropout_p=0.0, causal=False)
|
||||
else:
|
||||
attended = F.scaled_dot_product_attention(
|
||||
query.transpose(1, 2), key.transpose(1, 2), value.transpose(1, 2), dropout_p=0.0, is_causal=False
|
||||
).transpose(1, 2)
|
||||
return self.out_proj(attended.reshape(batch, time, width))
|
||||
|
||||
|
||||
class FeedForward(nn.Module):
|
||||
def __init__(self, config):
|
||||
super().__init__()
|
||||
self.w_1 = nn.Linear(config.hidden_size, config.intermediate_size)
|
||||
self.w_2 = nn.Linear(config.intermediate_size, config.hidden_size)
|
||||
|
||||
def forward(self, hidden_states):
|
||||
return self.w_2(F.gelu(self.w_1(hidden_states)))
|
||||
|
||||
|
||||
class ConvolutionModule(nn.Module):
|
||||
def __init__(self, config):
|
||||
super().__init__()
|
||||
width = config.hidden_size
|
||||
kernel = config.conv_depthwise_kernel_size
|
||||
self.layer_norm = nn.LayerNorm(width, eps=config.layer_norm_eps)
|
||||
self.conv_block = nn.Sequential(
|
||||
Transpose(),
|
||||
nn.Conv1d(width, 2 * width, 1, bias=False),
|
||||
nn.GLU(dim=1),
|
||||
nn.Conv1d(width, width, kernel, padding=(kernel - 1) // 2, groups=width, bias=False),
|
||||
nn.Sequential(Transpose(), nn.LayerNorm(width, eps=config.layer_norm_eps), Transpose()),
|
||||
nn.GELU(),
|
||||
nn.Conv1d(width, width, 1, bias=False),
|
||||
Transpose(),
|
||||
)
|
||||
|
||||
def forward(self, hidden_states):
|
||||
return self.conv_block(self.layer_norm(hidden_states))
|
||||
|
||||
|
||||
class ConformerBlock(nn.Module):
|
||||
def __init__(self, config):
|
||||
super().__init__()
|
||||
width, eps = config.hidden_size, config.layer_norm_eps
|
||||
self.ffn1_layer_norm = nn.LayerNorm(width, eps=eps)
|
||||
self.ffn1 = FeedForward(config)
|
||||
self.attn_layer_norm = nn.LayerNorm(width, eps=eps)
|
||||
self.attn = SelfAttention(config)
|
||||
self.conv_module = ConvolutionModule(config)
|
||||
self.ffn2_layer_norm = nn.LayerNorm(width, eps=eps)
|
||||
self.ffn2 = FeedForward(config)
|
||||
self.final_layer_norm = nn.LayerNorm(width, eps=eps)
|
||||
|
||||
def forward(self, hidden_states, position_embeddings):
|
||||
hidden_states = hidden_states + 0.5 * self.ffn1(self.ffn1_layer_norm(hidden_states))
|
||||
hidden_states = self.attn(self.attn_layer_norm(hidden_states), position_embeddings) + hidden_states
|
||||
hidden_states = self.conv_module(hidden_states) + hidden_states
|
||||
hidden_states = hidden_states + 0.5 * self.ffn2(self.ffn2_layer_norm(hidden_states))
|
||||
return self.final_layer_norm(hidden_states)
|
||||
|
||||
|
||||
class MERT2Model(PreTrainedModel):
|
||||
"""Encode mono waveforms into 25 Hz MERT2 frame representations.
|
||||
|
||||
Inputs are floating-point mono waveforms sampled at ``config.sampling_rate``.
|
||||
Do not standardize their amplitude. An optional waveform attention mask must
|
||||
contain a prefix of ones followed by zeros. Without a mask, every input
|
||||
sample, including any caller-provided silence or padding, is processed.
|
||||
"""
|
||||
|
||||
config_class = MERT2Config
|
||||
base_model_prefix = ""
|
||||
main_input_name = "input_values"
|
||||
_supports_sdpa = True
|
||||
_supports_flash_attn_2 = True
|
||||
_no_split_modules = ["ConvNextBlock", "ConformerBlock"]
|
||||
|
||||
def __init__(self, config):
|
||||
super().__init__(config)
|
||||
if config._attn_implementation not in {"sdpa", "flash_attention_2"}:
|
||||
raise ValueError("MERT2 supports attn_implementation='sdpa' or 'flash_attention_2'.")
|
||||
self.feature_extractor = MERT2MelFrontend(config)
|
||||
channels = [config.num_mel_bins] + config.subsampling_channels
|
||||
self.subsampling_module = nn.Sequential(*[
|
||||
ConvNextBlock(channels[i], channels[i + 1], (1, 2, 2)[i], config.subsampling_depths[i], config.subsampling_layer_norm_eps)
|
||||
for i in range(3)
|
||||
])
|
||||
self.layers = nn.ModuleList([ConformerBlock(config) for _ in range(config.num_hidden_layers)])
|
||||
self.embed_positions = RotaryEmbedding(config)
|
||||
self.post_init()
|
||||
|
||||
def _init_weights(self, module):
|
||||
if isinstance(module, (nn.Linear, nn.Conv1d)):
|
||||
nn.init.normal_(module.weight, mean=0.0, std=self.config.initializer_range)
|
||||
if module.bias is not None:
|
||||
nn.init.zeros_(module.bias)
|
||||
elif isinstance(module, nn.LayerNorm):
|
||||
nn.init.ones_(module.weight)
|
||||
nn.init.zeros_(module.bias)
|
||||
|
||||
def _get_feat_extract_output_lengths(self, input_lengths):
|
||||
return input_lengths // self.config.inputs_to_logits_ratio
|
||||
|
||||
def _encode(self, input_values, output_hidden_states):
|
||||
mel = self.feature_extractor(input_values)
|
||||
input_dtype = self.subsampling_module[0].convnext_layers[0].depthwise_block[1].weight.dtype
|
||||
hidden = self.subsampling_module(mel.to(dtype=input_dtype))
|
||||
positions = self.embed_positions(hidden)
|
||||
states = [] if output_hidden_states else None
|
||||
for layer in self.layers:
|
||||
hidden = layer(hidden, positions)
|
||||
if states is not None:
|
||||
states.append(hidden)
|
||||
return hidden, tuple(states) if states is not None else None
|
||||
|
||||
def forward(self, input_values, attention_mask=None, output_hidden_states=None, return_dict=None):
|
||||
"""Return final frames, optionally all block states, and a frame mask.
|
||||
|
||||
Different valid waveform lengths are encoded separately so convolution
|
||||
and global normalization never incorporate another sample's padding.
|
||||
Outputs are zero-padded to the longest valid feature sequence.
|
||||
"""
|
||||
output_hidden_states = self.config.output_hidden_states if output_hidden_states is None else output_hidden_states
|
||||
return_dict = self.config.use_return_dict if return_dict is None else return_dict
|
||||
if input_values.ndim != 2 or not input_values.is_floating_point() or input_values.shape[0] == 0:
|
||||
raise ValueError("input_values must be a nonempty floating-point tensor of shape [batch, samples].")
|
||||
if input_values.shape[1] < self.config.minimum_input_samples:
|
||||
raise ValueError(f"Each waveform must contain at least {self.config.minimum_input_samples} samples.")
|
||||
if not torch.isfinite(input_values).all():
|
||||
raise ValueError("input_values must contain only finite samples.")
|
||||
batch, samples = input_values.shape
|
||||
if attention_mask is None:
|
||||
hidden, states = self._encode(input_values, output_hidden_states)
|
||||
feature_mask = torch.ones(hidden.shape[:2], dtype=torch.bool, device=hidden.device)
|
||||
else:
|
||||
if attention_mask.shape != input_values.shape:
|
||||
raise ValueError("attention_mask must have the same shape as input_values.")
|
||||
attention_mask = attention_mask.to(device=input_values.device)
|
||||
if not ((attention_mask == 0) | (attention_mask == 1)).all():
|
||||
raise ValueError("attention_mask must contain only zeros and ones.")
|
||||
mask = attention_mask.bool()
|
||||
lengths = mask.sum(dim=1)
|
||||
expected = torch.arange(samples, device=input_values.device)[None, :] < lengths[:, None]
|
||||
if not torch.equal(mask, expected):
|
||||
raise ValueError("attention_mask must be right padded: valid samples followed by padding.")
|
||||
if (lengths < self.config.minimum_input_samples).any():
|
||||
raise ValueError(f"Each waveform must contain at least {self.config.minimum_input_samples} valid samples.")
|
||||
feature_lengths = self._get_feat_extract_output_lengths(lengths)
|
||||
maximum = int(feature_lengths.max().item())
|
||||
feature_mask = torch.arange(maximum, device=input_values.device)[None, :] < feature_lengths[:, None]
|
||||
hidden, state_values = None, None
|
||||
for length in torch.unique(lengths, sorted=True).tolist():
|
||||
indices = torch.where(lengths == length)[0]
|
||||
group_hidden, group_states = self._encode(input_values.index_select(0, indices)[:, :length], output_hidden_states)
|
||||
padding = (0, 0, 0, maximum - group_hidden.shape[1])
|
||||
if hidden is None:
|
||||
hidden = group_hidden.new_zeros(batch, maximum, self.config.hidden_size)
|
||||
if output_hidden_states:
|
||||
state_values = [torch.zeros_like(hidden) for _ in self.layers]
|
||||
hidden = hidden.index_copy(0, indices, F.pad(group_hidden, padding))
|
||||
if state_values is not None:
|
||||
for i, value in enumerate(group_states):
|
||||
state_values[i] = state_values[i].index_copy(0, indices, F.pad(value, padding))
|
||||
states = tuple(state_values) if state_values is not None else None
|
||||
output = MERT2ModelOutput(last_hidden_state=hidden, hidden_states=states, feature_attention_mask=feature_mask)
|
||||
return output if return_dict else output.to_tuple()
|
||||
|
||||
|
||||
MERT2Model.register_for_auto_class("AutoModel")
|
||||
@@ -0,0 +1,448 @@
|
||||
"""Hugging Face SheetSage2 model with a shared MERT-v2 encoder."""
|
||||
|
||||
from dataclasses import dataclass
|
||||
import copy
|
||||
import hashlib
|
||||
from pathlib import Path
|
||||
import shutil
|
||||
from types import SimpleNamespace
|
||||
from typing import Optional, Tuple
|
||||
|
||||
import torch
|
||||
from torch import nn
|
||||
from torch.nn import functional as F
|
||||
from transformers import AutoModel, BartConfig, PreTrainedModel
|
||||
from transformers.models.bart.modeling_bart import BartDecoder
|
||||
from transformers.utils import ModelOutput
|
||||
from transformers.utils.hub import cached_file
|
||||
|
||||
from .configuration_mert2 import MERT2Config
|
||||
from .modeling_mert2 import MERT2Model
|
||||
from .configuration_sheetsage2 import SheetSage2Config
|
||||
from .tokenization_sheetsage2 import SheetSage2Tokenizer
|
||||
|
||||
|
||||
PROJECTIONS = ("query_proj", "key_proj", "value_proj", "out_proj")
|
||||
BASE_CODE_HASHES = {
|
||||
"configuration_mert2.py": "77b53ec9d7ee31a599d744fb006e812c7eeaf7390deb46e2f460cf8c17b00bd6",
|
||||
"modeling_mert2.py": "b1a3174e5649c4b26b0c90d8626f0adacfbbba111a58ed3bb72ad651945a2f5c",
|
||||
}
|
||||
|
||||
|
||||
def _hf_relative_dependencies():
|
||||
# Transformers 4.45 copies direct imports into a fresh local module cache.
|
||||
from .audio_sheetsage2 import load_audio
|
||||
from .durations_sheetsage2 import DURATION_TEMPLATES
|
||||
from .exports_sheetsage2 import export_result
|
||||
from .io_sheetsage2 import atomic_write_text
|
||||
from .labels_sheetsage2 import STRUCTURE_LABELS
|
||||
from .midi_sheetsage2 import export_playback
|
||||
from .notation_sheetsage2 import generate_abc_from_exports
|
||||
from .rendering_sheetsage2 import render_outputs
|
||||
from .schema_sheetsage2 import get_prompt_multitask_schema
|
||||
from .tensors_sheetsage2 import WindowTensorWriter
|
||||
|
||||
|
||||
def _sha256(path):
|
||||
digest = hashlib.sha256()
|
||||
with Path(path).open("rb") as stream:
|
||||
for block in iter(lambda: stream.read(8 << 20), b""):
|
||||
digest.update(block)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
@dataclass
|
||||
class SheetSage2EncoderOutput(ModelOutput):
|
||||
"""Frame features. Block states exclude the separately returned input state."""
|
||||
|
||||
encoder_last_hidden_state: Optional[torch.FloatTensor] = None
|
||||
backbone_last_hidden_state: Optional[torch.FloatTensor] = None
|
||||
mixed_hidden_state: Optional[torch.FloatTensor] = None
|
||||
input_hidden_state: Optional[torch.FloatTensor] = None
|
||||
backbone_hidden_states: Optional[Tuple[torch.FloatTensor, ...]] = None
|
||||
feature_attention_mask: Optional[torch.BoolTensor] = None
|
||||
|
||||
@property
|
||||
def last_hidden_state(self):
|
||||
return self.encoder_last_hidden_state
|
||||
|
||||
|
||||
@dataclass
|
||||
class SheetSage2Output(ModelOutput):
|
||||
logits: Optional[torch.FloatTensor] = None
|
||||
past_key_values: Optional[Tuple] = None
|
||||
decoder_hidden_states: Optional[Tuple[torch.FloatTensor, ...]] = None
|
||||
encoder_last_hidden_state: Optional[torch.FloatTensor] = None
|
||||
backbone_last_hidden_state: Optional[torch.FloatTensor] = None
|
||||
mixed_hidden_state: Optional[torch.FloatTensor] = None
|
||||
input_hidden_state: Optional[torch.FloatTensor] = None
|
||||
backbone_hidden_states: Optional[Tuple[torch.FloatTensor, ...]] = None
|
||||
feature_attention_mask: Optional[torch.BoolTensor] = None
|
||||
|
||||
|
||||
class AttentionAdapter(nn.Module):
|
||||
def __init__(self, width, rank):
|
||||
super().__init__()
|
||||
self.lora_A = nn.Linear(width, rank, bias=False)
|
||||
self.lora_B = nn.Linear(rank, width, bias=False)
|
||||
|
||||
|
||||
class EncoderAdapters(nn.Module):
|
||||
def __init__(self, config):
|
||||
super().__init__()
|
||||
self.layers = nn.ModuleList()
|
||||
for _ in range(config.backbone_config["num_hidden_layers"]):
|
||||
layer = nn.Module()
|
||||
layer.attn = nn.Module()
|
||||
for name in PROJECTIONS:
|
||||
setattr(layer.attn, name, AttentionAdapter(config.backbone_config["hidden_size"], config.lora_rank))
|
||||
self.layers.append(layer)
|
||||
|
||||
|
||||
class SheetSage2Model(PreTrainedModel):
|
||||
"""Audio-to-symbolic model, with optional frame and decoder representations.
|
||||
|
||||
``from_pretrained`` loads the pinned MERT-v2 parent and merges attention
|
||||
adapters in float32. ``save_pretrained`` writes an independent merged model.
|
||||
``forward`` returns raw vocabulary logits; grammar-masked generation scores
|
||||
are available separately through ``generate(output_scores=True)``.
|
||||
"""
|
||||
|
||||
config_class = SheetSage2Config
|
||||
base_model_prefix = ""
|
||||
main_input_name = "input_values"
|
||||
_supports_sdpa = True
|
||||
_tied_weights_keys = ["decoder.embed_tokens.weight", "output_projection.weight"]
|
||||
_no_split_modules = ["ConformerBlock", "BartDecoderLayer"]
|
||||
|
||||
def __init__(self, config):
|
||||
super().__init__(config)
|
||||
self.hparams = SimpleNamespace(input_audio_length=config.input_audio_length, time_hz=config.time_hz)
|
||||
self.max_output_seq_len = config.max_output_seq_len
|
||||
self.tokenizer = SheetSage2Tokenizer(
|
||||
config.input_audio_length, config.time_hz, config.tokenizer_schema_version,
|
||||
expected_fingerprint=config.tokenizer_fingerprint,
|
||||
)
|
||||
if self.tokenizer.n_tokens != config.vocab_size:
|
||||
raise ValueError("Tokenizer vocabulary size does not match the model.")
|
||||
if config.weights_format == "adapter":
|
||||
self.encoder = None
|
||||
self.adapter = EncoderAdapters(config)
|
||||
else:
|
||||
ec = MERT2Config(**config.backbone_config)
|
||||
ec._attn_implementation = config.encoder_attn_implementation
|
||||
self.encoder = MERT2Model(ec)
|
||||
self.adapter = None
|
||||
self.layer_weight = nn.Parameter(torch.zeros(config.backbone_config["num_hidden_layers"] + 1))
|
||||
self.encoder_projection = nn.Linear(config.backbone_config["hidden_size"], config.hidden_size)
|
||||
dc = BartConfig(
|
||||
vocab_size=config.vocab_size, d_model=config.hidden_size,
|
||||
decoder_layers=config.decoder_layers, decoder_attention_heads=config.num_attention_heads,
|
||||
decoder_ffn_dim=config.intermediate_size, max_position_embeddings=config.max_output_seq_len,
|
||||
dropout=config.decoder_dropout, attention_dropout=config.decoder_dropout,
|
||||
activation_dropout=config.decoder_dropout, activation_function="gelu",
|
||||
pad_token_id=config.pad_token_id, bos_token_id=config.bos_token_id,
|
||||
eos_token_id=config.eos_token_id, is_encoder_decoder=True, use_cache=True,
|
||||
)
|
||||
dc._attn_implementation = "sdpa"
|
||||
self.token_embedding = nn.Embedding(config.vocab_size, config.hidden_size, padding_idx=config.pad_token_id)
|
||||
self.decoder = BartDecoder(dc, embed_tokens=self.token_embedding)
|
||||
self.decoder.gradient_checkpointing_disable()
|
||||
self.output_projection = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
|
||||
self.output_projection.weight = self.token_embedding.weight
|
||||
self.lora_merged = config.weights_format == "merged"
|
||||
self.post_init()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.token_embedding
|
||||
|
||||
def set_input_embeddings(self, value):
|
||||
self.token_embedding = value
|
||||
self.decoder.embed_tokens.weight = value.weight
|
||||
|
||||
def tie_weights(self):
|
||||
super().tie_weights()
|
||||
if hasattr(self, "decoder") and hasattr(self, "token_embedding"):
|
||||
self.decoder.embed_tokens.weight = self.token_embedding.weight
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.output_projection
|
||||
|
||||
def set_output_embeddings(self, value):
|
||||
self.output_projection = value
|
||||
|
||||
def _init_weights(self, module):
|
||||
if isinstance(module, (nn.Linear, nn.Embedding)):
|
||||
nn.init.normal_(module.weight, mean=0.0, std=0.02)
|
||||
if getattr(module, "bias", None) is not None:
|
||||
nn.init.zeros_(module.bias)
|
||||
if isinstance(module, nn.Embedding) and module.padding_idx is not None:
|
||||
module.weight.data[module.padding_idx].zero_()
|
||||
elif isinstance(module, nn.LayerNorm):
|
||||
nn.init.ones_(module.weight)
|
||||
nn.init.zeros_(module.bias)
|
||||
|
||||
@torch.no_grad()
|
||||
def merge_lora(self):
|
||||
if self.lora_merged:
|
||||
return self
|
||||
if self.encoder is None or self.adapter is None:
|
||||
raise RuntimeError("Load both the MERT-v2 parent and adapters before merging.")
|
||||
scale = self.config.lora_alpha / self.config.lora_rank
|
||||
with torch.autocast("cpu", enabled=False):
|
||||
for layer, adapter in zip(self.encoder.layers, self.adapter.layers):
|
||||
for name in PROJECTIONS:
|
||||
projection = getattr(layer.attn, name)
|
||||
update = getattr(adapter.attn, name)
|
||||
values = (projection.weight, update.lora_A.weight, update.lora_B.weight)
|
||||
if any(value.device.type != "cpu" or value.dtype != torch.float32 for value in values):
|
||||
raise ValueError("Merge adapters on CPU with float32 parameters.")
|
||||
projection.weight.add_((update.lora_B.weight @ update.lora_A.weight) * scale)
|
||||
self.adapter = None
|
||||
self.lora_merged = True
|
||||
self.config.weights_format = "merged"
|
||||
return self
|
||||
|
||||
@classmethod
|
||||
def from_pretrained(cls, pretrained_model_name_or_path, *model_args, **kwargs):
|
||||
base_path = kwargs.pop("base_model_path", None)
|
||||
requested_dtype = kwargs.pop("torch_dtype", torch.float32)
|
||||
target_device = kwargs.pop("device_map", None)
|
||||
backend = kwargs.pop("attn_implementation", None)
|
||||
return_loading_info = kwargs.pop("output_loading_info", False)
|
||||
config = kwargs.pop("config", None)
|
||||
hub_keys = ("cache_dir", "force_download", "local_files_only", "token", "revision", "subfolder")
|
||||
hub_args = {name: kwargs[name] for name in hub_keys if name in kwargs}
|
||||
if config is None:
|
||||
config = cls.config_class.from_pretrained(pretrained_model_name_or_path, **hub_args)
|
||||
if backend is not None:
|
||||
config.encoder_attn_implementation = backend
|
||||
if config.encoder_attn_implementation not in {"sdpa", "flash_attention_2"}:
|
||||
raise ValueError("Select attn_implementation='sdpa' or 'flash_attention_2'.")
|
||||
if isinstance(target_device, dict):
|
||||
if set(target_device) != {""}:
|
||||
raise ValueError("Use a single device for SheetSage2: device_map={'': 'cuda:0'}.")
|
||||
target_device = target_device[""]
|
||||
if target_device == "auto":
|
||||
target_device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
if requested_dtype == "auto":
|
||||
requested_dtype = torch.float32
|
||||
if isinstance(requested_dtype, str):
|
||||
requested_dtype = getattr(torch, requested_dtype, None)
|
||||
if requested_dtype not in {None, torch.float32, torch.bfloat16, torch.float16}:
|
||||
raise ValueError("Use torch_dtype float32, bfloat16, float16, or 'auto'.")
|
||||
# Adapters must be loaded and merged before any reduced-precision cast.
|
||||
model, loading_info = super().from_pretrained(
|
||||
pretrained_model_name_or_path, *model_args, config=config,
|
||||
torch_dtype=torch.float32, attn_implementation="sdpa", output_loading_info=True, **kwargs,
|
||||
)
|
||||
if any(loading_info.get(name) for name in ("missing_keys", "unexpected_keys", "mismatched_keys", "error_msgs")):
|
||||
raise ValueError(f"Incomplete or incompatible SheetSage2 weights: {loading_info}")
|
||||
if config.weights_format == "adapter":
|
||||
parent = str(base_path or config.base_model_name_or_path)
|
||||
parent_hub_args = {name: hub_args[name] for name in ("cache_dir", "force_download", "local_files_only", "token") if name in hub_args}
|
||||
parent_hub_args["revision"] = config.base_model_revision
|
||||
expected_files = dict(BASE_CODE_HASHES, **{"model.safetensors": config.base_model_sha256})
|
||||
for filename, expected in expected_files.items():
|
||||
resolved = cached_file(parent, filename, **parent_hub_args)
|
||||
if _sha256(resolved) != expected:
|
||||
raise ValueError(f"MERT-v2 parent integrity check failed: {filename}")
|
||||
model.encoder = AutoModel.from_pretrained(
|
||||
parent, trust_remote_code=True, code_revision=config.base_model_revision,
|
||||
torch_dtype=torch.float32, attn_implementation=config.encoder_attn_implementation,
|
||||
**parent_hub_args,
|
||||
)
|
||||
for name, expected in config.backbone_config.items():
|
||||
if name in ("hidden_size", "intermediate_size", "num_hidden_layers", "num_attention_heads",
|
||||
"sampling_rate", "hop_length", "n_fft", "win_length", "num_mel_bins",
|
||||
"conv_depthwise_kernel_size", "rotary_embedding_base", "subsampling_channels",
|
||||
"subsampling_depths", "layer_norm_eps", "subsampling_layer_norm_eps", "variant"):
|
||||
if getattr(model.encoder.config, name) != expected:
|
||||
raise ValueError(f"MERT-v2 parent architecture mismatch: {name}")
|
||||
model.merge_lora()
|
||||
if requested_dtype is not None:
|
||||
model.to(dtype=requested_dtype)
|
||||
if target_device is not None:
|
||||
model.to(target_device)
|
||||
# Resource lookup follows the caller's cache and offline settings. These
|
||||
# transient options are never written into config or model weights.
|
||||
model._hub_resource_options = {name: hub_args[name] for name in ("cache_dir", "local_files_only", "token") if name in hub_args}
|
||||
resource_args = dict(model._hub_resource_options, local_files_only=True,
|
||||
revision=config._commit_hash or hub_args.get("revision"),
|
||||
subfolder=hub_args.get("subfolder", ""),
|
||||
_raise_exceptions_for_missing_entries=False)
|
||||
resource_config = cached_file(str(pretrained_model_name_or_path), "config.json", **resource_args)
|
||||
model._source_snapshot = Path(resource_config).parent if resource_config else Path(pretrained_model_name_or_path)
|
||||
model.eval().requires_grad_(False)
|
||||
return (model, loading_info) if return_loading_info else model
|
||||
|
||||
def save_pretrained(self, save_directory, *args, **kwargs):
|
||||
if not self.lora_merged or self.encoder is None:
|
||||
raise ValueError("Load and merge the model before saving a standalone snapshot.")
|
||||
original_config = self.config
|
||||
source = original_config._name_or_path
|
||||
self.config = copy.deepcopy(original_config)
|
||||
self.config.weights_format = "merged"
|
||||
self.config._name_or_path = ""
|
||||
self.config.backbone_config.pop("_name_or_path", None)
|
||||
try:
|
||||
result = super().save_pretrained(save_directory, *args, **kwargs)
|
||||
from .processing_sheetsage2 import SheetSage2Processor
|
||||
SheetSage2Processor.from_model_config(self.config).save_pretrained(save_directory)
|
||||
finally:
|
||||
self.config = original_config
|
||||
source_dir = getattr(self, "_source_snapshot", Path(source))
|
||||
if source and not source_dir.is_dir():
|
||||
from huggingface_hub import snapshot_download
|
||||
try:
|
||||
source_dir = Path(snapshot_download(source, revision=original_config._commit_hash,
|
||||
cache_dir=getattr(self, "_hub_resource_options", {}).get("cache_dir"),
|
||||
local_files_only=True))
|
||||
except (OSError, ValueError):
|
||||
source_dir = None
|
||||
if source and source_dir is not None and source_dir.is_dir():
|
||||
destination = Path(save_directory)
|
||||
for name in ("infer.py", "render.py", "setup_render.py", "requirements.txt", "requirements-render.txt",
|
||||
"LICENSE", "THIRD_PARTY_NOTICES.md"):
|
||||
if (source_dir / name).is_file() and (source_dir / name).resolve() != (destination / name).resolve():
|
||||
shutil.copy2(source_dir / name, destination / name)
|
||||
assets = source_dir / "render_assets"
|
||||
if (assets / "manifest.json").is_file() and assets.resolve() != (destination / "render_assets").resolve():
|
||||
from .rendering_sheetsage2 import _verify_assets
|
||||
try:
|
||||
_verify_assets(assets, audio=True)
|
||||
except (FileNotFoundError, ValueError):
|
||||
pass
|
||||
else:
|
||||
shutil.copytree(assets, destination / "render_assets", dirs_exist_ok=True)
|
||||
return result
|
||||
|
||||
def _prepare_audio(self, input_values, attention_mask=None):
|
||||
if self.encoder is None or not self.lora_merged:
|
||||
raise RuntimeError("Load the model with from_pretrained before inference.")
|
||||
if input_values.ndim != 2 or input_values.shape[0] < 1 or not input_values.is_floating_point():
|
||||
raise ValueError("input_values must be floating-point [batch, samples].")
|
||||
if not torch.isfinite(input_values).all():
|
||||
raise ValueError("Audio must contain only finite samples.")
|
||||
batch, samples = input_values.shape
|
||||
minimum = self.encoder.config.minimum_input_samples
|
||||
window = round(self.config.input_audio_length * self.config.sampling_rate)
|
||||
if samples < minimum:
|
||||
raise ValueError(f"Each waveform must contain at least {minimum} samples.")
|
||||
if samples > window:
|
||||
raise ValueError("Audio exceeds one model window; use transcribe for whole songs.")
|
||||
if attention_mask is None:
|
||||
lengths = torch.full((batch,), samples, dtype=torch.long, device=input_values.device)
|
||||
else:
|
||||
if attention_mask.shape != input_values.shape:
|
||||
raise ValueError("attention_mask must have the same shape as input_values.")
|
||||
mask = attention_mask.to(device=input_values.device)
|
||||
if not ((mask == 0) | (mask == 1)).all():
|
||||
raise ValueError("attention_mask must contain zeros and ones.")
|
||||
mask = mask.bool()
|
||||
lengths = mask.sum(1)
|
||||
expected = torch.arange(samples, device=input_values.device)[None] < lengths[:, None]
|
||||
if not torch.equal(mask, expected) or (lengths < minimum).any():
|
||||
raise ValueError("attention_mask must identify a nonempty right-padded waveform.")
|
||||
input_values = input_values.masked_fill(~mask, 0)
|
||||
# The encoder attends to the complete fixed window, including this silence.
|
||||
input_values = F.pad(input_values.float(), (0, window - samples))
|
||||
stride = self.encoder.config.inputs_to_logits_ratio
|
||||
if window % stride:
|
||||
input_values = F.pad(input_values, (0, stride - window % stride))
|
||||
return input_values, lengths
|
||||
|
||||
def get_audio_features(self, input_values, attention_mask=None, output_hidden_states=None, return_dict=True):
|
||||
output_hidden_states = self.config.output_hidden_states if output_hidden_states is None else output_hidden_states
|
||||
waveform, lengths = self._prepare_audio(input_values, attention_mask)
|
||||
mel = self.encoder.feature_extractor(waveform)
|
||||
weight_dtype = self.encoder.subsampling_module[0].convnext_layers[0].depthwise_block[1].weight.dtype
|
||||
hidden = self.encoder.subsampling_module(mel.to(dtype=weight_dtype))
|
||||
input_hidden = hidden if output_hidden_states else None
|
||||
weights = torch.softmax(self.layer_weight, dim=0)
|
||||
mixed = hidden * weights[0]
|
||||
positions = self.encoder.embed_positions(hidden)
|
||||
states = [] if output_hidden_states else None
|
||||
for weight, layer in zip(weights[1:], self.encoder.layers):
|
||||
hidden = layer(hidden, positions)
|
||||
mixed = mixed + hidden * weight
|
||||
if states is not None:
|
||||
states.append(hidden)
|
||||
memory = self.encoder_projection(mixed)
|
||||
stride = self.encoder.config.inputs_to_logits_ratio
|
||||
frame_mask = torch.arange(memory.shape[1], device=memory.device)[None] < ((lengths + stride - 1) // stride)[:, None]
|
||||
output = SheetSage2EncoderOutput(
|
||||
encoder_last_hidden_state=memory, backbone_last_hidden_state=hidden,
|
||||
mixed_hidden_state=mixed, input_hidden_state=input_hidden,
|
||||
backbone_hidden_states=tuple(states) if states is not None else None,
|
||||
feature_attention_mask=frame_mask,
|
||||
)
|
||||
return output if return_dict else output.to_tuple()
|
||||
|
||||
def encode(self, audio):
|
||||
return self.get_audio_features(audio).last_hidden_state
|
||||
|
||||
def _decode(self, memory, decoder_input_ids, use_cache=False, past_key_values=None, output_hidden_states=False):
|
||||
attention_mask = None if past_key_values is not None else decoder_input_ids != self.tokenizer.pad_token
|
||||
output = self.decoder(
|
||||
input_ids=decoder_input_ids, attention_mask=attention_mask,
|
||||
encoder_hidden_states=memory, encoder_attention_mask=None,
|
||||
past_key_values=past_key_values, use_cache=use_cache,
|
||||
output_hidden_states=output_hidden_states, return_dict=True,
|
||||
)
|
||||
return self.output_projection(output.last_hidden_state), output
|
||||
|
||||
def decode(self, memory, decoder_input_ids, use_cache=False, past_key_values=None):
|
||||
logits, output = self._decode(memory, decoder_input_ids, use_cache, past_key_values)
|
||||
return logits, output.past_key_values
|
||||
|
||||
def forward(self, input_values=None, decoder_input_ids=None, attention_mask=None,
|
||||
encoder_outputs=None, past_key_values=None, use_cache=None,
|
||||
output_hidden_states=None, return_dict=None):
|
||||
output_hidden_states = self.config.output_hidden_states if output_hidden_states is None else output_hidden_states
|
||||
use_cache = self.config.use_cache if use_cache is None else use_cache
|
||||
if decoder_input_ids is None:
|
||||
raise ValueError("decoder_input_ids is required for logits; use generate to transcribe audio.")
|
||||
if encoder_outputs is None:
|
||||
if input_values is None:
|
||||
raise ValueError("Provide input_values or encoder_outputs.")
|
||||
encoder_outputs = self.get_audio_features(input_values, attention_mask, output_hidden_states)
|
||||
if torch.is_tensor(encoder_outputs):
|
||||
encoder_outputs = SheetSage2EncoderOutput(encoder_last_hidden_state=encoder_outputs)
|
||||
logits, decoded = self._decode(encoder_outputs.last_hidden_state, decoder_input_ids,
|
||||
use_cache, past_key_values, output_hidden_states)
|
||||
output = SheetSage2Output(
|
||||
logits=logits, past_key_values=decoded.past_key_values,
|
||||
decoder_hidden_states=decoded.hidden_states,
|
||||
encoder_last_hidden_state=encoder_outputs.last_hidden_state,
|
||||
backbone_last_hidden_state=encoder_outputs.backbone_last_hidden_state if output_hidden_states else None,
|
||||
mixed_hidden_state=encoder_outputs.mixed_hidden_state if output_hidden_states else None,
|
||||
input_hidden_state=encoder_outputs.input_hidden_state if output_hidden_states else None,
|
||||
backbone_hidden_states=encoder_outputs.backbone_hidden_states if output_hidden_states else None,
|
||||
feature_attention_mask=encoder_outputs.feature_attention_mask,
|
||||
)
|
||||
return_dict = self.config.use_return_dict if return_dict is None else return_dict
|
||||
return output if return_dict else output.to_tuple()
|
||||
|
||||
def generate(self, input_values, **kwargs):
|
||||
"""Generate grammar-constrained symbolic tokens with autoregressive caching."""
|
||||
from .generation_sheetsage2 import generate
|
||||
return generate(self, input_values, **kwargs)
|
||||
|
||||
def transcribe(self, audio, output_dir=None, *, melody_only=False, **kwargs):
|
||||
"""Return transcription in memory; set output_dir to also save files.
|
||||
|
||||
Accepts a path, encoded audio bytes, binary stream, or waveform with
|
||||
sampling_rate. Returns ABC text, MIDI bytes, timed events, and optional
|
||||
per-window CPU tensors. With output_dir, optional tensors are saved
|
||||
instead of retained in memory. Set melody_only=True to retain both vocal
|
||||
and instrumental melodies while omitting chords from ABC and playback;
|
||||
raw predicted annotations remain available. The default keeps full
|
||||
transcription. See the model card for rendering options.
|
||||
"""
|
||||
from .pipeline_sheetsage2 import transcribe
|
||||
return transcribe(self, audio, output_dir=output_dir, melody_only=melody_only, **kwargs)
|
||||
|
||||
|
||||
SheetSage2ForConditionalGeneration = SheetSage2Model
|
||||
SheetSage2Model.register_for_auto_class("AutoModel")
|
||||
@@ -0,0 +1,259 @@
|
||||
"""Whole-song inference with cached overlap prefixes and right-hand audio context."""
|
||||
from pathlib import Path
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
|
||||
import torch
|
||||
|
||||
from .io_sheetsage2 import atomic_write_text
|
||||
from .audio_sheetsage2 import SAMPLE_RATE, load_audio, slice_audio
|
||||
from .generation_sheetsage2 import (
|
||||
FULL_TASK_PROMPTS, build_overlap_prefix_tokens, constrained_prompt_generate,
|
||||
decode_generated_tokens, event_time_map, stitched_window_events, write_window_tokens,
|
||||
)
|
||||
from .exports_sheetsage2 import export_result
|
||||
|
||||
|
||||
def _tensor_files(output_dir):
|
||||
"""Inventory only files owned by the optional tensor exporter."""
|
||||
files = set()
|
||||
tensor_dir = output_dir / "tensors"
|
||||
if tensor_dir.is_symlink():
|
||||
return files
|
||||
for directory in tensor_dir.glob("window-*"):
|
||||
if re.fullmatch(r"window-\d{4,}", directory.name) and directory.is_dir() and not directory.is_symlink():
|
||||
for path in directory.iterdir():
|
||||
if path.is_file() and re.fullmatch(r"(?:index\.json|audio\.safetensors|(?:tokens|decoder)-\d{4,}\.safetensors)", path.name):
|
||||
files.add(path)
|
||||
return files
|
||||
|
||||
|
||||
def sliding_window_plan(duration, window_seconds=300.0, overlap_seconds=200.0, lookahead_seconds=100.0):
|
||||
if not duration > 0 or not window_seconds > 0:
|
||||
raise ValueError("Duration and window length must be positive")
|
||||
if not 0 <= lookahead_seconds <= overlap_seconds < window_seconds:
|
||||
raise ValueError("Require 0 <= lookahead <= overlap < window length")
|
||||
hop = window_seconds - overlap_seconds
|
||||
start, accepted = 0.0, 0.0
|
||||
result = []
|
||||
while True:
|
||||
last = start + window_seconds >= duration - 1e-6
|
||||
accept_end = duration if last else start + window_seconds - lookahead_seconds
|
||||
result.append(dict(start=start, end=min(duration, start + window_seconds),
|
||||
accept_start=accepted, accept_end=accept_end, prefix_end=accepted,
|
||||
generation_stop=None if last else window_seconds - lookahead_seconds))
|
||||
if last:
|
||||
return result
|
||||
accepted = accept_end
|
||||
start = min(start + hop, duration - window_seconds)
|
||||
|
||||
|
||||
class Transcriber:
|
||||
def __init__(self, model, dtype="bf16"):
|
||||
self.model = model
|
||||
self.device = next(model.parameters()).device
|
||||
self.dtype = {"bf16": torch.bfloat16, "fp32": None}[dtype]
|
||||
|
||||
@torch.inference_mode()
|
||||
def analyze(self, audio_path, output_dir=None, *, prompts=FULL_TASK_PROMPTS,
|
||||
sampling_rate=None, max_seconds=None, preset="default",
|
||||
overlap_seconds=None, lookahead_seconds=None, progress=None,
|
||||
export_logits=False, export_scores=False, export_embeddings=False,
|
||||
output_hidden_states=False, melody_only=False):
|
||||
def report(stage, **fields):
|
||||
if progress:
|
||||
progress(dict(stage=stage, **fields))
|
||||
|
||||
if not isinstance(melody_only, bool):
|
||||
raise ValueError("melody_only must be True or False")
|
||||
if preset not in ("default", "paper"):
|
||||
raise ValueError("preset must be default or paper")
|
||||
defaults = (200.0, 100.0) if preset == "default" else (100.0, 0.0)
|
||||
overlap_seconds = defaults[0] if overlap_seconds is None else overlap_seconds
|
||||
lookahead_seconds = defaults[1] if lookahead_seconds is None else lookahead_seconds
|
||||
if preset == "paper" and (overlap_seconds, lookahead_seconds) != defaults:
|
||||
raise ValueError("paper preset fixes overlap=100 and lookahead=0")
|
||||
started = time.monotonic()
|
||||
output_dir = Path(output_dir) if output_dir is not None else None
|
||||
if output_dir is not None:
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
previous_tensors = _tensor_files(output_dir) if output_dir is not None else set()
|
||||
current_tensors = set()
|
||||
tensor_results = []
|
||||
prompts = self.model.tokenizer.normalize_prompts(prompts)
|
||||
if "timestamp" not in prompts:
|
||||
raise ValueError("timestamp is required to export timed annotations")
|
||||
report("audio")
|
||||
audio = load_audio(audio_path, sampling_rate=sampling_rate, max_seconds=max_seconds, preset=preset)
|
||||
duration = len(audio) / SAMPLE_RATE
|
||||
window_length = float(self.model.hparams.input_audio_length)
|
||||
plan = sliding_window_plan(duration, window_length, overlap_seconds, lookahead_seconds)
|
||||
if self.device.type == "cuda":
|
||||
torch.cuda.reset_peak_memory_stats(self.device)
|
||||
stitched, records, warnings = [], [], []
|
||||
for index, window in enumerate(plan):
|
||||
report("encoding", window=index + 1, windows=len(plan), start=window["start"])
|
||||
segment = slice_audio(audio, window["start"], window_length)[None].to(self.device)
|
||||
prefix, base = None, 0
|
||||
if index:
|
||||
_, prefix, base = build_overlap_prefix_tokens(
|
||||
stitched, self.model.tokenizer, prompts, window["start"], window["prefix_end"])
|
||||
if prefix is not None:
|
||||
if preset == "paper" and len(prefix) >= self.model.max_output_seq_len - 16:
|
||||
warnings.append(f"Window {index + 1}: overlap prefix exceeded the context")
|
||||
prefix = None
|
||||
elif preset != "paper" and len(prefix) >= self.model.max_output_seq_len - 128:
|
||||
raise ValueError("Overlap prefix fills the context; reduce overlap_seconds")
|
||||
tick = time.monotonic()
|
||||
memory = None
|
||||
tensor_record = None
|
||||
if export_embeddings or output_hidden_states or export_logits or export_scores:
|
||||
from .tensors_sheetsage2 import WindowTensorWriter
|
||||
tensor_record = WindowTensorWriter(output_dir, index, window, export_logits, export_scores)
|
||||
if export_embeddings or output_hidden_states:
|
||||
memory = tensor_record.audio_features(self.model, segment, self.dtype, output_hidden_states)
|
||||
tokens = constrained_prompt_generate(
|
||||
self.model, segment, prompts, self.model.max_output_seq_len,
|
||||
prefix_tokens=prefix, autocast_dtype=self.dtype,
|
||||
stop_time_seconds=(None if preset == "paper" else
|
||||
window["generation_stop"] if window["generation_stop"] is not None else
|
||||
min(duration - window["start"], window_length)),
|
||||
memory=memory,
|
||||
step_callback=tensor_record.capture if tensor_record is not None and (export_logits or export_scores) else None,
|
||||
progress_callback=lambda n: report("decoding", window=index + 1, windows=len(plan), tokens=n),
|
||||
)
|
||||
if tensor_record is not None:
|
||||
if export_embeddings:
|
||||
tensor_record.decoder_features(self.model, segment, tokens, self.dtype, memory)
|
||||
tensor_data = tensor_record.finish()
|
||||
if tensor_data is not None:
|
||||
tensor_results.append(tensor_data)
|
||||
if tensor_record.directory is not None:
|
||||
current_tensors.update(tensor_record.directory / name for name in tensor_record.files)
|
||||
current_tensors.add(tensor_record.directory / "index.json")
|
||||
if len(tokens) > self.model.max_output_seq_len:
|
||||
warnings.append(f"Window {index + 1} reached the token limit; inspect its token coverage")
|
||||
decoded, warning = decode_generated_tokens(self.model.tokenizer, tokens, Path(audio_path).stem if isinstance(audio_path, (str, Path)) else "audio", index)
|
||||
if warning:
|
||||
warnings.append(warning["error"])
|
||||
lookup = event_time_map(decoded, window_length)
|
||||
accepted = stitched_window_events(
|
||||
decoded, lookup, window["start"], window["accept_start"], window["accept_end"],
|
||||
duration, index, global_subbeat_base=base or 0,
|
||||
)
|
||||
stitched.extend(accepted)
|
||||
if preset == "paper":
|
||||
stitched.sort(key=lambda e: (float(e.get("time", 0)), int(e.get("window_index", 0)),
|
||||
int(e.get("source_subbeat", e["subbeat"]))))
|
||||
record = dict(window, window_index=index, prefix_tokens=0 if prefix is None else len(prefix),
|
||||
tokens=tokens, events=len(decoded["events"]), accepted_events=len(accepted),
|
||||
elapsed_seconds=time.monotonic() - tick)
|
||||
records.append(record)
|
||||
report("window_complete", window=index + 1, windows=len(plan), tokens=len(tokens))
|
||||
if preset != "paper":
|
||||
stitched.sort(key=lambda e: (e["time"], e["global_subbeat"]))
|
||||
decoded = dict(schema_version=self.model.tokenizer.schema_version,
|
||||
prompts=list(prompts), events=stitched, has_eos=True)
|
||||
report("notation")
|
||||
exported = export_result(decoded, self.model.tokenizer, output_dir, duration,
|
||||
paper=preset == "paper", melody_only=melody_only)
|
||||
payload = exported.pop("payload")
|
||||
if output_dir is not None:
|
||||
write_window_tokens(output_dir / "tokens.txt", records, self.model.tokenizer)
|
||||
atomic_write_text(output_dir / "tokens.json", json.dumps([
|
||||
dict(r, tokens=r["tokens"].tolist()) for r in records
|
||||
]))
|
||||
result = dict(
|
||||
audio=Path(audio_path).name if isinstance(audio_path, (str, Path)) else "audio",
|
||||
duration_seconds=duration, prompts=list(prompts), preset=preset, melody_only=melody_only,
|
||||
dtype="bf16" if self.dtype and self.device.type == "cuda" else "fp32",
|
||||
window_seconds=window_length, overlap_seconds=overlap_seconds, lookahead_seconds=lookahead_seconds,
|
||||
windows=[dict(r, tokens=len(r["tokens"])) for r in records],
|
||||
elapsed_seconds=time.monotonic() - started, warnings=warnings,
|
||||
peak_gpu_mib=torch.cuda.max_memory_allocated(self.device) / 1024**2 if self.device.type == "cuda" else 0,
|
||||
**exported,
|
||||
)
|
||||
if output_dir is not None:
|
||||
atomic_write_text(output_dir / "result.json", json.dumps(result, indent=2))
|
||||
for path in previous_tensors - current_tensors:
|
||||
path.unlink(missing_ok=True)
|
||||
for directory in {path.parent for path in previous_tensors - current_tensors}:
|
||||
if directory.is_dir() and not any(directory.iterdir()):
|
||||
directory.rmdir()
|
||||
report("complete", **exported)
|
||||
return dict(result, num_events=result["events"], **payload,
|
||||
tokens=[r["tokens"] for r in records], tensors=tensor_results,
|
||||
_metadata=result)
|
||||
|
||||
|
||||
@torch.inference_mode()
|
||||
def transcribe(model, audio, output_dir=None, *, render_audio=False, render_score=False,
|
||||
render_parts=("mix",), dtype="bf16", melody_only=False, **kwargs):
|
||||
"""Transcribe a path, encoded audio bytes, binary stream, or waveform.
|
||||
|
||||
Arrays use channels-first layout and require sampling_rate. With
|
||||
output_dir=None, all audio/results stay in memory: ABC is text, MIDI is
|
||||
bytes, events are a list, and optional features are CPU tensors grouped by
|
||||
window. A directory saves the standard output files and writes optional
|
||||
tensors to disk instead of keeping them in memory. Optional
|
||||
rendering also returns bytes/text when no output directory is given.
|
||||
melody_only=True omits chords from ABC and MIDI playback, retaining both
|
||||
melody voices and the original decoded events/LAB annotations. If the
|
||||
requested melody-only ABC cannot be built, raises RuntimeError with the
|
||||
completed transcription available as error.result.
|
||||
"""
|
||||
if not isinstance(melody_only, bool):
|
||||
raise ValueError("melody_only must be True or False")
|
||||
output_dir = Path(output_dir) if output_dir is not None else None
|
||||
if render_audio or render_score:
|
||||
from .rendering_sheetsage2 import render_outputs, render_memory, validate_render_options
|
||||
formats, render_parts = validate_render_options(audio=render_audio, score=render_score, parts=render_parts)
|
||||
result = Transcriber(model, dtype=dtype).analyze(audio, output_dir, melody_only=melody_only, **kwargs)
|
||||
metadata = result.pop("_metadata", None)
|
||||
if metadata is None:
|
||||
metadata = dict(result)
|
||||
if render_audio or render_score:
|
||||
try:
|
||||
assets = Path(__file__).with_name("render_assets")
|
||||
if not assets.is_dir():
|
||||
source = getattr(model, "_source_snapshot", Path(model.config._name_or_path))
|
||||
if (source / "render_assets").is_dir():
|
||||
assets = source / "render_assets"
|
||||
else:
|
||||
from huggingface_hub import snapshot_download
|
||||
assets = Path(snapshot_download(model.config._name_or_path,
|
||||
revision=model.config._commit_hash, allow_patterns=["render_assets/**"],
|
||||
**getattr(model, "_hub_resource_options", {}))) / "render_assets"
|
||||
# A missing score does not prevent a requested MIDI piano preview.
|
||||
rendered = None
|
||||
if render_audio or not result.get("abc_error"):
|
||||
options = dict(audio=render_audio, score=() if result.get("abc_error") else formats,
|
||||
parts=render_parts, assets_dir=assets)
|
||||
if output_dir is None:
|
||||
rendered = render_memory(midi=result["midi"], abc=result["abc"],
|
||||
duration=result["duration_seconds"], **options)
|
||||
else:
|
||||
rendered = render_outputs(input_dir=output_dir, output_dir=output_dir, **options)
|
||||
result["rendered"] = rendered
|
||||
if output_dir is not None:
|
||||
metadata["rendered"] = rendered
|
||||
if formats and result.get("abc_error"):
|
||||
raise ValueError(f"ABC unavailable: {result['abc_error']}")
|
||||
except Exception as exc:
|
||||
result["render_error"] = str(exc)
|
||||
metadata["render_error"] = str(exc)
|
||||
if output_dir is not None:
|
||||
atomic_write_text(output_dir / "result.json", json.dumps(metadata, indent=2))
|
||||
location = f"saved to {output_dir.resolve()}" if output_dir is not None else "completed in memory"
|
||||
error = RuntimeError(f"Transcription {location}; rendering failed: {exc}")
|
||||
error.result = result
|
||||
raise error from exc
|
||||
if output_dir is not None:
|
||||
atomic_write_text(output_dir / "result.json", json.dumps(metadata, indent=2))
|
||||
if melody_only and not result.get("abc"):
|
||||
reason = result.get("abc_error") or "No ABC score was produced"
|
||||
error = RuntimeError(f"Melody-only ABC unavailable: {reason}")
|
||||
error.result = result
|
||||
raise error
|
||||
return result
|
||||
@@ -0,0 +1,109 @@
|
||||
"""Waveform preparation and symbolic decoding for SheetSage2."""
|
||||
|
||||
import numbers
|
||||
|
||||
import torch
|
||||
import torchaudio
|
||||
from transformers import ProcessorMixin
|
||||
from transformers.feature_extraction_utils import BatchFeature
|
||||
|
||||
from .tokenization_sheetsage2 import SheetSage2Tokenizer
|
||||
|
||||
|
||||
def _hf_relative_dependencies():
|
||||
# Keep standalone AutoProcessor loading compatible with Transformers 4.45.
|
||||
from .durations_sheetsage2 import DURATION_TEMPLATES
|
||||
from .labels_sheetsage2 import STRUCTURE_LABELS
|
||||
from .schema_sheetsage2 import get_prompt_multitask_schema
|
||||
|
||||
|
||||
class SheetSage2Processor(ProcessorMixin):
|
||||
"""Prepare mono waveforms without amplitude normalization.
|
||||
|
||||
Pass one mono waveform, a batch tensor, or a list of mono waveforms. Use
|
||||
``model.transcribe`` for files and audio longer than one model window.
|
||||
"""
|
||||
|
||||
attributes = []
|
||||
valid_kwargs = ["sampling_rate", "window_seconds", "time_hz", "schema_version", "tokenizer_fingerprint"]
|
||||
|
||||
def __init__(self, sampling_rate=24000, window_seconds=300.0, time_hz=100,
|
||||
schema_version="v1", tokenizer_fingerprint="5ba3325af0344c7f", **kwargs):
|
||||
super().__init__(**{key: value for key, value in kwargs.items() if key == "chat_template"})
|
||||
self.sampling_rate = int(sampling_rate)
|
||||
self.window_seconds = float(window_seconds)
|
||||
self.time_hz = int(time_hz)
|
||||
self.schema_version = str(schema_version)
|
||||
self.tokenizer_fingerprint = str(tokenizer_fingerprint)
|
||||
self.tokenizer = SheetSage2Tokenizer(
|
||||
self.window_seconds, self.time_hz, self.schema_version,
|
||||
expected_fingerprint=self.tokenizer_fingerprint,
|
||||
)
|
||||
|
||||
@property
|
||||
def model_input_names(self):
|
||||
return ["input_values", "attention_mask"]
|
||||
|
||||
@classmethod
|
||||
def from_model_config(cls, config):
|
||||
return cls(config.sampling_rate, config.input_audio_length, config.time_hz,
|
||||
config.tokenizer_schema_version, config.tokenizer_fingerprint)
|
||||
|
||||
def __call__(self, audio, sampling_rate=None, padding=True, return_tensors="pt",
|
||||
return_attention_mask=True):
|
||||
source_rate = self.sampling_rate if sampling_rate is None else int(sampling_rate)
|
||||
if source_rate <= 0:
|
||||
raise ValueError("sampling_rate must be positive.")
|
||||
if isinstance(audio, (str, bytes)):
|
||||
raise ValueError("Pass waveform samples here; use model.transcribe for audio files.")
|
||||
if isinstance(audio, (list, tuple)) and audio and not isinstance(audio[0], numbers.Number):
|
||||
waveforms = [torch.as_tensor(value) for value in audio]
|
||||
else:
|
||||
value = torch.as_tensor(audio)
|
||||
waveforms = list(value) if value.ndim == 2 else [value]
|
||||
if not waveforms:
|
||||
raise ValueError("Provide at least one waveform.")
|
||||
prepared = []
|
||||
maximum = round(self.window_seconds * self.sampling_rate)
|
||||
for waveform in waveforms:
|
||||
if waveform.ndim != 1 or not waveform.is_floating_point() or not torch.isfinite(waveform).all():
|
||||
raise ValueError("Each waveform must be a one-dimensional finite floating-point array.")
|
||||
waveform = waveform.to(dtype=torch.float32)
|
||||
if source_rate != self.sampling_rate:
|
||||
waveform = torchaudio.functional.resample(waveform, source_rate, self.sampling_rate)
|
||||
if waveform.numel() < 1025:
|
||||
raise ValueError("Each waveform must contain at least 1025 samples at 24 kHz.")
|
||||
if waveform.numel() > maximum:
|
||||
raise ValueError("Audio exceeds one model window; use model.transcribe for whole songs.")
|
||||
prepared.append(waveform)
|
||||
if len({str(value.device) for value in prepared}) != 1:
|
||||
raise ValueError("All waveforms in a batch must be on the same device.")
|
||||
if padding is False:
|
||||
length = max(value.numel() for value in prepared)
|
||||
if any(value.numel() != length for value in prepared):
|
||||
raise ValueError("Use padding=True for waveforms of different lengths.")
|
||||
elif padding is True or padding == "max_length":
|
||||
length = maximum
|
||||
elif padding == "longest":
|
||||
length = max(value.numel() for value in prepared)
|
||||
else:
|
||||
raise ValueError("padding must be True, False, 'max_length', or 'longest'.")
|
||||
lengths = torch.tensor([value.numel() for value in prepared], device=prepared[0].device)
|
||||
values = torch.stack([torch.nn.functional.pad(value, (0, length - value.numel())) for value in prepared])
|
||||
data = {"input_values": values}
|
||||
if return_attention_mask:
|
||||
data["attention_mask"] = (torch.arange(length, device=values.device)[None] < lengths[:, None]).long()
|
||||
if return_tensors == "np":
|
||||
data = {key: value.cpu().numpy() for key, value in data.items()}
|
||||
elif return_tensors not in {None, "pt"}:
|
||||
raise ValueError("return_tensors must be 'pt', 'np', or None.")
|
||||
return BatchFeature(data=data)
|
||||
|
||||
def decode(self, token_ids, strict=True):
|
||||
return self.tokenizer.decode_sequence(token_ids, strict=strict)
|
||||
|
||||
def batch_decode(self, sequences, strict=True):
|
||||
return [self.decode(tokens, strict=strict) for tokens in sequences]
|
||||
|
||||
|
||||
SheetSage2Processor.register_for_auto_class()
|
||||
@@ -0,0 +1,11 @@
|
||||
{
|
||||
"processor_class": "SheetSage2Processor",
|
||||
"sampling_rate": 24000,
|
||||
"window_seconds": 300.0,
|
||||
"time_hz": 100,
|
||||
"schema_version": "v1",
|
||||
"tokenizer_fingerprint": "5ba3325af0344c7f",
|
||||
"auto_map": {
|
||||
"AutoProcessor": "processing_sheetsage2.SheetSage2Processor"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,45 @@
|
||||
"""Render existing SheetSage2 MIDI and ABC outputs without loading a model."""
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
SCRIPT_DIR = Path(__file__).absolute().parent
|
||||
sys.path.insert(0, str(SCRIPT_DIR))
|
||||
|
||||
from rendering_sheetsage2 import render_outputs
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--input", help="Transcription output directory.")
|
||||
parser.add_argument("--midi", help="MIDI file with original note timing.")
|
||||
parser.add_argument("--abc", help="ABC score file.")
|
||||
parser.add_argument("--output", help="Output directory; defaults to --input or rendered/.")
|
||||
parser.add_argument("--audio", action="store_true", help="Render piano WAV.")
|
||||
parser.add_argument("--score", default="", help="Comma-separated pdf,svg,png.")
|
||||
parser.add_argument("--parts", default="mix", help="Comma-separated mix,melody,vocal,instrumental,chords, or all (audio only).")
|
||||
parser.add_argument("--duration", type=float, help="Minimum audio duration in seconds; preserves trailing silence.")
|
||||
args = parser.parse_args()
|
||||
audio = args.audio
|
||||
score = args.score
|
||||
if not audio and not score:
|
||||
if args.input:
|
||||
audio, score = True, "pdf"
|
||||
else:
|
||||
audio, score = bool(args.midi), "pdf" if args.abc else ""
|
||||
if not args.input and not args.midi and not args.abc:
|
||||
parser.error("Provide --input, --midi, or --abc.")
|
||||
try:
|
||||
result = render_outputs(args.input, midi=args.midi, abc=args.abc,
|
||||
output_dir=args.output, audio=audio, score=score,
|
||||
parts=args.parts, duration=args.duration)
|
||||
except (ValueError, OSError, RuntimeError) as exc:
|
||||
print(f"Rendering failed: {exc}", file=sys.stderr)
|
||||
return 1
|
||||
print(json.dumps(result, indent=2))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,306 @@
|
||||
"""Optional offline piano and staff rendering. No model or GPU is loaded."""
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import hashlib
|
||||
import io
|
||||
import json
|
||||
import math
|
||||
import mimetypes
|
||||
from pathlib import Path
|
||||
import re
|
||||
import wave
|
||||
|
||||
|
||||
PARTS = ("mix", "melody", "vocal", "instrumental", "chords")
|
||||
FORMATS = ("pdf", "svg", "png")
|
||||
|
||||
|
||||
def _options(value, allowed, label):
|
||||
values = value.split(",") if isinstance(value, str) else list(value)
|
||||
values = tuple(dict.fromkeys(x.strip().lower() for x in values if x.strip()))
|
||||
unknown = set(values) - set(allowed)
|
||||
if unknown:
|
||||
raise ValueError(f"Unknown {label}: {', '.join(sorted(unknown))}. Choose from {', '.join(allowed)}.")
|
||||
return values
|
||||
|
||||
|
||||
def validate_render_options(*, audio=False, score=(), parts=("mix",)):
|
||||
"""Validate output choices without importing a model or browser runtime.
|
||||
|
||||
Return normalized ``(score_formats, audio_parts)`` tuples. ``score=True``
|
||||
selects PDF; no requested output is valid for an inference-only call.
|
||||
"""
|
||||
score = ("pdf",) if score is True else () if score is False or score is None else score
|
||||
formats = _options(score, FORMATS, "score format")
|
||||
requested = _options(parts, PARTS + ("all",), "audio part")
|
||||
if "all" in requested:
|
||||
requested = PARTS
|
||||
if audio and not requested:
|
||||
raise ValueError("Choose at least one audio part.")
|
||||
return formats, requested
|
||||
|
||||
|
||||
def _tracks(midi):
|
||||
try:
|
||||
import pretty_midi
|
||||
except ImportError as exc:
|
||||
raise RuntimeError("Install rendering support first: python setup_render.py") from exc
|
||||
try:
|
||||
source = io.BytesIO(bytes(midi)) if isinstance(midi, (bytes, bytearray, memoryview)) else str(midi)
|
||||
parsed = pretty_midi.PrettyMIDI(source)
|
||||
except Exception as exc:
|
||||
name = midi.name if isinstance(midi, Path) else "in-memory MIDI"
|
||||
raise ValueError(f"Cannot read MIDI: {name}: {exc}") from exc
|
||||
result = []
|
||||
for instrument in parsed.instruments:
|
||||
if instrument.is_drum:
|
||||
raise ValueError("Piano rendering does not support drum tracks.")
|
||||
notes = []
|
||||
for note in instrument.notes:
|
||||
if not all(math.isfinite(x) for x in (note.start, note.end)) or note.start < 0 or note.end <= note.start:
|
||||
raise ValueError("MIDI note times must be finite, nonnegative, and increasing.")
|
||||
if not 21 <= note.pitch <= 109:
|
||||
raise ValueError(f"Piano samples cover MIDI pitches 21–109; received {note.pitch}.")
|
||||
notes.append(dict(pitch=int(note.pitch), start=float(note.start), end=float(note.end), velocity=int(note.velocity)))
|
||||
result.append(dict(name=instrument.name, notes=notes))
|
||||
return result, float(parsed.get_end_time())
|
||||
|
||||
|
||||
def _select(tracks, part):
|
||||
def role(track):
|
||||
name = track["name"].strip().lower()
|
||||
if "chord" in name:
|
||||
return "chords"
|
||||
if "vocal" in name:
|
||||
return "vocal"
|
||||
if name in {"ins", "instrument", "instrumental"} or "instrumental" in name:
|
||||
return "instrumental"
|
||||
return "melody"
|
||||
if part == "mix":
|
||||
return tracks
|
||||
if part == "melody":
|
||||
return [track for track in tracks if role(track) != "chords"]
|
||||
return [track for track in tracks if role(track) == part]
|
||||
|
||||
|
||||
def _verify_assets(directory, *, audio):
|
||||
manifest = directory / "manifest.json"
|
||||
if not manifest.is_file():
|
||||
raise FileNotFoundError("Rendering assets are missing. Download render_assets/ from the model repository.")
|
||||
inventory = json.loads(manifest.read_text(encoding="utf-8"))["files"]
|
||||
for relative, digest in inventory.items():
|
||||
if not audio and relative.startswith("soundfonts/"):
|
||||
continue
|
||||
path = directory / relative
|
||||
if not path.is_file() or hashlib.sha256(path.read_bytes()).hexdigest() != digest:
|
||||
raise ValueError(f"Rendering asset is missing or corrupt: {relative}")
|
||||
return set(inventory)
|
||||
|
||||
|
||||
def _metadata_duration(directory):
|
||||
for name, key in (("result.json", "duration_seconds"), ("playback.json", "duration")):
|
||||
path = directory / name
|
||||
if path.is_file():
|
||||
value = json.loads(path.read_text(encoding="utf-8")).get(key)
|
||||
if value is not None:
|
||||
return float(value)
|
||||
return None
|
||||
|
||||
|
||||
def render_outputs(input_dir=None, *, midi=None, abc=None, output_dir=None,
|
||||
audio=False, score=(), parts=("mix",), duration=None,
|
||||
assets_dir=None):
|
||||
"""Render existing outputs, preserving MIDI note times and canonical ABC.
|
||||
|
||||
``audio=True`` writes piano WAV; ``score`` selects pdf/svg/png. ``parts``
|
||||
selects audio stems; ``all`` requests every available stem. SVG and PNG
|
||||
contain one A4 page per file; PDF contains every page. No network request
|
||||
is permitted during rendering. Assets and the browser must be installed.
|
||||
"""
|
||||
formats, requested = validate_render_options(audio=audio, score=score, parts=parts)
|
||||
if not audio and not formats:
|
||||
raise ValueError("Request audio or at least one score format.")
|
||||
directory = Path(input_dir).expanduser().resolve() if input_dir else None
|
||||
if directory is not None and not directory.is_dir():
|
||||
raise FileNotFoundError(f"Input directory does not exist: {directory}")
|
||||
destination = Path(output_dir or directory or "rendered").expanduser().resolve()
|
||||
if midi is None and directory:
|
||||
midi = directory / ("transcription.mid" if (directory / "transcription.mid").exists() else "melody.mid")
|
||||
if abc is None and directory:
|
||||
abc = directory / "score.abc"
|
||||
if duration is None and directory:
|
||||
duration = _metadata_duration(directory)
|
||||
if duration is not None and (not math.isfinite(duration) or duration < 0):
|
||||
raise ValueError("Duration must be finite and nonnegative.")
|
||||
tracks, seconds = [], 0.0
|
||||
if audio:
|
||||
if midi is None or not Path(midi).is_file():
|
||||
raise FileNotFoundError("Audio rendering needs an existing MIDI file (--midi or --input).")
|
||||
tracks, end = _tracks(Path(midi))
|
||||
seconds = max(end, duration or 0)
|
||||
if seconds <= 0:
|
||||
raise ValueError("An empty MIDI needs a positive --duration.")
|
||||
abc_text = None
|
||||
if formats:
|
||||
if abc is None or not Path(abc).is_file():
|
||||
raise FileNotFoundError("Score rendering needs an existing ABC file (--abc or --input).")
|
||||
abc_text = Path(abc).read_text(encoding="utf-8")
|
||||
if not abc_text.strip():
|
||||
raise ValueError("ABC input is empty.")
|
||||
return _render(tracks, seconds, abc_text, audio=audio, formats=formats,
|
||||
requested=requested, destination=destination, assets_dir=assets_dir)
|
||||
|
||||
|
||||
def render_memory(*, midi=None, abc=None, audio=False, score=(), parts=("mix",),
|
||||
duration=None, assets_dir=None):
|
||||
"""Render MIDI bytes and ABC text without writing input or result files.
|
||||
|
||||
Return WAV bytes in ``audio[part]`` and page lists in ``score[format]``:
|
||||
PDF and PNG values are bytes, SVG values are strings. A PDF list contains
|
||||
one complete document. The optional browser runtime may use its own profile.
|
||||
"""
|
||||
formats, requested = validate_render_options(audio=audio, score=score, parts=parts)
|
||||
if not audio and not formats:
|
||||
raise ValueError("Request audio or at least one score format.")
|
||||
if duration is not None and (not math.isfinite(duration) or duration < 0):
|
||||
raise ValueError("Duration must be finite and nonnegative.")
|
||||
tracks, seconds = [], 0.0
|
||||
if audio:
|
||||
if not isinstance(midi, (bytes, bytearray, memoryview)):
|
||||
raise ValueError("Audio rendering needs MIDI bytes.")
|
||||
tracks, end = _tracks(midi)
|
||||
seconds = max(end, duration or 0)
|
||||
if seconds <= 0:
|
||||
raise ValueError("An empty MIDI needs a positive duration.")
|
||||
if formats and (not isinstance(abc, str) or not abc.strip()):
|
||||
raise ValueError("Score rendering needs nonempty ABC text.")
|
||||
return _render(tracks, seconds, abc, audio=audio, formats=formats,
|
||||
requested=requested, assets_dir=assets_dir)
|
||||
|
||||
|
||||
def _render(tracks, seconds, abc_text, *, audio, formats, requested,
|
||||
destination=None, assets_dir=None):
|
||||
assets = Path(assets_dir) if assets_dir else Path(__file__).with_name("render_assets")
|
||||
assets = assets.resolve()
|
||||
asset_files = _verify_assets(assets, audio=audio)
|
||||
try:
|
||||
from playwright.sync_api import sync_playwright, Error as BrowserError
|
||||
except ImportError as exc:
|
||||
raise RuntimeError("Install rendering support first: python setup_render.py") from exc
|
||||
|
||||
output = {"audio": {}, "score": {}, "warnings": []}
|
||||
blocked = []
|
||||
previous_pages = []
|
||||
if destination is not None:
|
||||
destination.mkdir(parents=True, exist_ok=True)
|
||||
for path in destination.iterdir():
|
||||
match = re.fullmatch(r"score_(\d{3,})\.(svg|png)", path.name)
|
||||
if match and match[2] in formats and path.is_file() and not path.is_symlink():
|
||||
previous_pages.append((int(match[1]), path))
|
||||
with sync_playwright() as playwright:
|
||||
try:
|
||||
browser = playwright.chromium.launch(headless=True, args=["--autoplay-policy=no-user-gesture-required", "--disable-gpu"])
|
||||
except Exception as exc:
|
||||
raise RuntimeError("Could not start the renderer. Run python setup_render.py; on Linux with missing system libraries, use --with-deps.") from exc
|
||||
try:
|
||||
context = browser.new_context(viewport={"width": 794, "height": 1123}, device_scale_factor=2, service_workers="block", offline=True)
|
||||
def route_handler(route):
|
||||
from urllib.parse import unquote, urlsplit
|
||||
url = urlsplit(route.request.url)
|
||||
if url.scheme != "https" or url.netloc != "render.invalid":
|
||||
blocked.append(route.request.url)
|
||||
route.abort()
|
||||
return
|
||||
if url.path == "/":
|
||||
route.fulfill(status=200, content_type="text/html", body="<!doctype html><html><head><meta charset='utf-8'><style>html,body{margin:0;background:white}.score-page{display:block;break-after:page;width:210mm;height:297mm}.score-page:last-child{break-after:auto}@page{size:A4;margin:0}</style></head><body></body></html>")
|
||||
return
|
||||
relative = unquote(url.path.lstrip("/"))
|
||||
file = assets / relative
|
||||
# HF snapshots link verified assets into the shared blob cache.
|
||||
# Whitelist logical names instead of rejecting those symlinks.
|
||||
if relative not in asset_files or not file.is_file():
|
||||
blocked.append(url.path)
|
||||
route.abort()
|
||||
return
|
||||
route.fulfill(status=200, path=str(file), content_type=mimetypes.guess_type(file.name)[0] or "application/octet-stream")
|
||||
context.route("**/*", route_handler)
|
||||
page = context.new_page()
|
||||
page.goto("https://render.invalid/")
|
||||
page.add_script_tag(url="https://render.invalid/abcjs-basic-min.js")
|
||||
page.add_script_tag(url="https://render.invalid/renderer.js")
|
||||
if audio:
|
||||
for part in requested:
|
||||
selected = _select(tracks, part)
|
||||
if not selected and part not in {"mix", "melody"}:
|
||||
output["warnings"].append(f"No {part} track is available; skipped.")
|
||||
continue
|
||||
info = page.evaluate("renderAudio", {"tracks": selected, "duration": seconds})
|
||||
target = destination / f"piano_{part}.wav" if destination is not None else None
|
||||
temporary = target.with_suffix(".wav.partial") if target is not None else None
|
||||
buffer = io.BytesIO() if target is None else None
|
||||
try:
|
||||
with wave.open(str(temporary) if temporary is not None else buffer, "wb") as stream:
|
||||
stream.setnchannels(info["channels"])
|
||||
stream.setsampwidth(2)
|
||||
stream.setframerate(info["sample_rate"])
|
||||
for start in range(0, info["frames"], 32768):
|
||||
encoded = page.evaluate("audioBlock", {"start": start, "count": 32768})
|
||||
stream.writeframesraw(base64.b64decode(encoded))
|
||||
if temporary is not None:
|
||||
temporary.replace(target)
|
||||
output["audio"][part] = str(target) if target is not None else buffer.getvalue()
|
||||
finally:
|
||||
if temporary is not None:
|
||||
temporary.unlink(missing_ok=True)
|
||||
page.evaluate("window.renderedAudio = null")
|
||||
if formats:
|
||||
page.evaluate("""async ({font, license}) => {
|
||||
window.renderFontData = font;
|
||||
window.renderFontLicense = license;
|
||||
const face = new FontFace('SheetSageSans', 'url(data:font/ttf;base64,' + font + ')');
|
||||
document.fonts.add(await face.load());
|
||||
}""", {"font": base64.b64encode((assets / "DejaVuSans.ttf").read_bytes()).decode("ascii"),
|
||||
"license": (assets / "LICENSE.font").read_text(encoding="utf-8")})
|
||||
info = page.evaluate("renderScore", abc_text)
|
||||
for kind in formats:
|
||||
output["score"][kind] = []
|
||||
if "pdf" in formats:
|
||||
if destination is None:
|
||||
output["score"]["pdf"] = [page.pdf(format="A4", print_background=True, prefer_css_page_size=True)]
|
||||
else:
|
||||
target = destination / "score.pdf"
|
||||
temporary = target.with_suffix(".pdf.partial")
|
||||
page.pdf(path=str(temporary), format="A4", print_background=True, prefer_css_page_size=True)
|
||||
temporary.replace(target)
|
||||
output["score"]["pdf"] = [str(target)]
|
||||
for index, svg in enumerate(info["pages"], 1):
|
||||
if "svg" in formats:
|
||||
if destination is None:
|
||||
output["score"]["svg"].append(svg)
|
||||
else:
|
||||
target = destination / f"score_{index:03}.svg"
|
||||
target.write_text(svg, encoding="utf-8")
|
||||
output["score"]["svg"].append(str(target))
|
||||
if "png" in formats:
|
||||
element = page.locator(".score-page").nth(index-1)
|
||||
if destination is None:
|
||||
output["score"]["png"].append(element.screenshot(type="png"))
|
||||
else:
|
||||
target = destination / f"score_{index:03}.png"
|
||||
element.screenshot(path=str(target))
|
||||
output["score"]["png"].append(str(target))
|
||||
output["warnings"].extend(info["warnings"])
|
||||
if blocked:
|
||||
raise RuntimeError(f"Rendering attempted to load unavailable assets: {', '.join(blocked)}")
|
||||
except BrowserError as exc:
|
||||
raise RuntimeError(str(exc).split("Call log:", 1)[0].strip()) from exc
|
||||
finally:
|
||||
browser.close()
|
||||
# A failed render leaves existing pages intact. Only remove obsolete pages
|
||||
# in the requested formats, within this renderer's numbered-file namespace.
|
||||
if formats:
|
||||
for number, path in previous_pages:
|
||||
if number > len(info["pages"]):
|
||||
path.unlink(missing_ok=True)
|
||||
return output
|
||||
@@ -0,0 +1,4 @@
|
||||
playwright==1.58.0
|
||||
pretty_midi==0.2.10
|
||||
setuptools==78.1.1
|
||||
mir_eval
|
||||
@@ -0,0 +1,11 @@
|
||||
torch==2.8.0
|
||||
torchaudio==2.8.0
|
||||
transformers==4.45.2
|
||||
huggingface-hub==0.36.0
|
||||
safetensors==0.5.3
|
||||
numpy==1.24.3
|
||||
scipy==1.13.1
|
||||
mir_eval==0.8.2
|
||||
pretty_midi==0.2.10
|
||||
mido==1.3.3
|
||||
setuptools==78.1.1
|
||||
@@ -0,0 +1,122 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PromptTaskSpec:
|
||||
name: str
|
||||
sampling_group: str
|
||||
output_field: str
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TokenBlockSpec:
|
||||
name: str
|
||||
labels: tuple[str, ...]
|
||||
output_field: str | None = None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PromptMultitaskSchema:
|
||||
version: str
|
||||
tasks: tuple[PromptTaskSpec, ...]
|
||||
event_field_order: tuple[str, ...]
|
||||
appended_token_blocks: tuple[TokenBlockSpec, ...] = ()
|
||||
|
||||
|
||||
V1_TASKS = (
|
||||
PromptTaskSpec("timestamp", "timestamp", "timestamp"),
|
||||
PromptTaskSpec("downbeat_meter", "rhythm", "rhythm"),
|
||||
PromptTaskSpec("structure", "structure", "structure"),
|
||||
PromptTaskSpec("key", "key", "key"),
|
||||
PromptTaskSpec("chord_majmin", "chord", "chord"),
|
||||
PromptTaskSpec("chord_full", "chord", "chord"),
|
||||
PromptTaskSpec("melody_vocal", "melody", "melody"),
|
||||
PromptTaskSpec("melody_full", "melody", "melody"),
|
||||
)
|
||||
|
||||
|
||||
PROMPT_MULTITASK_SCHEMAS = {
|
||||
"v1": PromptMultitaskSchema(
|
||||
version="v1",
|
||||
tasks=V1_TASKS,
|
||||
event_field_order=(
|
||||
"timestamp",
|
||||
"rhythm",
|
||||
"structure",
|
||||
"key",
|
||||
"chord",
|
||||
"melody",
|
||||
),
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def _validate_schema(schema):
|
||||
task_names = tuple(task.name for task in schema.tasks)
|
||||
if len(task_names) != len(set(task_names)):
|
||||
raise ValueError(f"Duplicate task name in schema {schema.version!r}")
|
||||
field_names = tuple(schema.event_field_order)
|
||||
if len(field_names) != len(set(field_names)):
|
||||
raise ValueError(f"Duplicate output field in schema {schema.version!r}")
|
||||
missing_fields = {
|
||||
task.output_field for task in schema.tasks if task.output_field not in field_names
|
||||
}
|
||||
if missing_fields:
|
||||
raise ValueError(
|
||||
f"Tasks in schema {schema.version!r} use unknown output fields: "
|
||||
f"{tuple(sorted(missing_fields))}"
|
||||
)
|
||||
block_names = tuple(block.name for block in schema.appended_token_blocks)
|
||||
if len(block_names) != len(set(block_names)):
|
||||
raise ValueError(f"Duplicate token block in schema {schema.version!r}")
|
||||
for block in schema.appended_token_blocks:
|
||||
if not block.labels or len(block.labels) != len(set(block.labels)):
|
||||
raise ValueError(
|
||||
f"Token block {block.name!r} must contain unique labels"
|
||||
)
|
||||
|
||||
|
||||
def register_prompt_multitask_schema(schema):
|
||||
"""Register a frozen schema version without permitting redefinition."""
|
||||
if not isinstance(schema, PromptMultitaskSchema):
|
||||
raise TypeError("schema must be a PromptMultitaskSchema")
|
||||
_validate_schema(schema)
|
||||
existing = PROMPT_MULTITASK_SCHEMAS.get(schema.version)
|
||||
if existing is not None and existing != schema:
|
||||
raise ValueError(f"Schema {schema.version!r} is already registered")
|
||||
PROMPT_MULTITASK_SCHEMAS[schema.version] = schema
|
||||
return schema
|
||||
|
||||
|
||||
def extend_prompt_multitask_schema(
|
||||
base_version,
|
||||
version,
|
||||
tasks=(),
|
||||
event_fields=(),
|
||||
token_blocks=(),
|
||||
):
|
||||
"""Append tasks and vocabulary blocks while preserving every base token id."""
|
||||
base = get_prompt_multitask_schema(base_version)
|
||||
schema = PromptMultitaskSchema(
|
||||
version=str(version),
|
||||
tasks=base.tasks + tuple(tasks),
|
||||
event_field_order=base.event_field_order + tuple(event_fields),
|
||||
appended_token_blocks=base.appended_token_blocks + tuple(token_blocks),
|
||||
)
|
||||
return register_prompt_multitask_schema(schema)
|
||||
|
||||
|
||||
def get_prompt_multitask_schema(version):
|
||||
version = str(version)
|
||||
try:
|
||||
return PROMPT_MULTITASK_SCHEMAS[version]
|
||||
except KeyError as exc:
|
||||
raise ValueError(
|
||||
f"Unknown prompt multitask schema {version!r}; "
|
||||
f"available={tuple(PROMPT_MULTITASK_SCHEMAS)}"
|
||||
) from exc
|
||||
|
||||
|
||||
_validate_schema(PROMPT_MULTITASK_SCHEMAS["v1"])
|
||||
@@ -0,0 +1,37 @@
|
||||
"""Install the optional renderer and its matching headless browser."""
|
||||
from pathlib import Path
|
||||
import argparse
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
SCRIPT_DIR = Path(__file__).absolute().parent
|
||||
sys.path.insert(0, str(SCRIPT_DIR))
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--with-deps", action="store_true", help="Also install Chromium OS libraries (may require sudo).")
|
||||
args = parser.parse_args()
|
||||
subprocess.run([sys.executable, "-m", "pip", "install", "-r", str(SCRIPT_DIR / "requirements-render.txt")], check=True)
|
||||
command = [sys.executable, "-m", "playwright", "install", "--only-shell", "chromium"]
|
||||
if args.with_deps:
|
||||
command.append("--with-deps")
|
||||
subprocess.run(command, check=True)
|
||||
check = """from playwright.sync_api import sync_playwright
|
||||
with sync_playwright() as p:
|
||||
browser = p.chromium.launch(headless=True, args=['--disable-gpu'])
|
||||
context = browser.new_context(offline=True)
|
||||
page = context.new_page()
|
||||
assert page.evaluate('typeof OfflineAudioContext') == 'function'
|
||||
browser.close()
|
||||
"""
|
||||
result = subprocess.run([sys.executable, "-c", check], capture_output=True, text=True)
|
||||
if result.returncode:
|
||||
print("Chromium could not start. On Linux, run python setup_render.py --with-deps to install its system libraries.", file=sys.stderr)
|
||||
return 1
|
||||
print("Rendering is ready. Example: python render.py --input output --audio --score pdf")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,99 @@
|
||||
"""Aligned audio features and token predictions, in memory or on disk."""
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import torch
|
||||
from safetensors.torch import save_file
|
||||
|
||||
from .generation_sheetsage2 import inference_autocast
|
||||
|
||||
|
||||
class WindowTensorWriter:
|
||||
def __init__(self, output_dir, index, window, logits, scores):
|
||||
self.directory = Path(output_dir) / "tensors" / f"window-{index:04d}" if output_dir is not None else None
|
||||
if self.directory is not None:
|
||||
self.directory.mkdir(parents=True, exist_ok=True)
|
||||
self.window = window
|
||||
self.logits, self.scores = logits, scores
|
||||
self.pending, self.files = [], []
|
||||
self.data = {"window_index": index, "token_batches": [], "decoder_batches": []} if self.directory is None else None
|
||||
|
||||
def _store(self, name, payload, group):
|
||||
if self.directory is not None:
|
||||
save_file(payload, str(self.directory / name))
|
||||
self.files.append(name)
|
||||
elif group == "audio":
|
||||
self.data["audio"] = payload
|
||||
else:
|
||||
self.data[group].append(payload)
|
||||
|
||||
def audio_features(self, model, audio, dtype, hidden_states):
|
||||
with inference_autocast(audio.device, dtype):
|
||||
features = model.get_audio_features(audio, output_hidden_states=hidden_states)
|
||||
payload = {}
|
||||
for name, value in features.items():
|
||||
if torch.is_tensor(value):
|
||||
payload[name] = value.detach().cpu().contiguous().clone()
|
||||
elif isinstance(value, tuple):
|
||||
for index, layer in enumerate(value):
|
||||
if torch.is_tensor(layer):
|
||||
payload[f"{name}.{index}"] = layer.detach().cpu().contiguous().clone()
|
||||
memory = features.encoder_last_hidden_state
|
||||
frames = memory.shape[1]
|
||||
valid_seconds = self.window["end"] - self.window["start"]
|
||||
payload["frame_times"] = torch.arange(frames, dtype=torch.float64) / 25 + self.window["start"]
|
||||
payload["feature_attention_mask"] = (torch.arange(frames) / 25 < valid_seconds)[None]
|
||||
self._store("audio.safetensors", payload, "audio")
|
||||
return memory
|
||||
|
||||
def capture(self, position, ids, logits, masked):
|
||||
# Transcription processes one window at a time; keep at most 32 rows.
|
||||
row = {"token_position": torch.tensor(position, dtype=torch.long)}
|
||||
if self.logits:
|
||||
row["logits"] = logits[0].detach().cpu().contiguous()
|
||||
if self.scores:
|
||||
row["scores"] = masked[0].detach().cpu().contiguous()
|
||||
self.pending.append(row)
|
||||
if len(self.pending) == 32:
|
||||
self.flush()
|
||||
|
||||
def flush(self):
|
||||
if not self.pending:
|
||||
return
|
||||
first = int(self.pending[0]["token_position"])
|
||||
name = f"tokens-{first:04d}.safetensors"
|
||||
self._store(name, {key: torch.stack([row[key] for row in self.pending]) for key in self.pending[0]},
|
||||
"token_batches")
|
||||
self.pending.clear()
|
||||
|
||||
def decoder_features(self, model, audio, tokens, dtype, memory=None):
|
||||
with inference_autocast(audio.device, dtype):
|
||||
if memory is None:
|
||||
memory = model.encode(audio)
|
||||
cache = None
|
||||
for start in range(0, len(tokens) - 1, 64):
|
||||
ids = tokens[start:min(start + 64, len(tokens) - 1)][None].to(audio.device)
|
||||
output = model.decoder(
|
||||
input_ids=ids,
|
||||
attention_mask=ids.ne(model.tokenizer.pad_token) if cache is None else None,
|
||||
encoder_hidden_states=memory, encoder_attention_mask=None,
|
||||
past_key_values=cache, use_cache=True, return_dict=True,
|
||||
)
|
||||
cache = output.past_key_values
|
||||
name = f"decoder-{start:04d}.safetensors"
|
||||
self._store(name, {"decoder_hidden_state": output.last_hidden_state.detach().cpu().contiguous(),
|
||||
"input_token_ids": ids.cpu(),
|
||||
"token_positions": torch.arange(start, start + ids.shape[1])},
|
||||
"decoder_batches")
|
||||
|
||||
def finish(self):
|
||||
self.flush()
|
||||
metadata = dict(self.window, frame_rate=25)
|
||||
if self.directory is not None:
|
||||
metadata["files"] = self.files
|
||||
metadata["token_position_convention"] = "logits predict the token at token_position; decoder embeddings represent input_token_ids"
|
||||
if self.directory is not None:
|
||||
(self.directory / "index.json").write_text(json.dumps(metadata, indent=2) + "\n")
|
||||
else:
|
||||
self.data["metadata"] = metadata
|
||||
return self.data
|
||||
@@ -0,0 +1,732 @@
|
||||
import re
|
||||
import hashlib
|
||||
import json
|
||||
|
||||
import mir_eval.chord
|
||||
import numpy as np
|
||||
|
||||
from .durations_sheetsage2 import DURATION_TEMPLATES, duration_boundaries
|
||||
from .labels_sheetsage2 import STRUCTURE_LABELS
|
||||
from .schema_sheetsage2 import get_prompt_multitask_schema
|
||||
|
||||
|
||||
PROMPT_ORDER = tuple(task.name for task in get_prompt_multitask_schema("v1").tasks)
|
||||
|
||||
|
||||
CHROMATIC_SHARPS = (
|
||||
"C",
|
||||
"C#",
|
||||
"D",
|
||||
"D#",
|
||||
"E",
|
||||
"F",
|
||||
"F#",
|
||||
"G",
|
||||
"G#",
|
||||
"A",
|
||||
"A#",
|
||||
"B",
|
||||
)
|
||||
|
||||
FULL_CHORD_QUALITIES = (
|
||||
"maj",
|
||||
"min",
|
||||
"dim",
|
||||
"aug",
|
||||
"maj7",
|
||||
"min7",
|
||||
"7",
|
||||
"hdim7",
|
||||
"dim7",
|
||||
"minmaj7",
|
||||
"sus2",
|
||||
"sus4",
|
||||
"sus4(b7)",
|
||||
"maj6",
|
||||
"min6",
|
||||
)
|
||||
|
||||
FULL_CHORD_INVERSIONS = {
|
||||
"maj": ("/2", "/3", "/5"),
|
||||
"min": ("/2", "/b3", "/5"),
|
||||
"maj7": ("/3", "/5", "/7"),
|
||||
"min7": ("/b3", "/5", "/b7"),
|
||||
"7": ("/3", "/5", "/b7"),
|
||||
}
|
||||
|
||||
|
||||
def build_full_chord_vocabulary():
|
||||
labels = ["N"]
|
||||
for quality in FULL_CHORD_QUALITIES:
|
||||
inversions = FULL_CHORD_INVERSIONS.get(quality, ())
|
||||
for root in CHROMATIC_SHARPS:
|
||||
for inversion in (*inversions, ""):
|
||||
labels.append(f"{root}:{quality}{inversion}")
|
||||
return tuple(labels)
|
||||
|
||||
|
||||
FULL_CHORD_VOCABULARY = build_full_chord_vocabulary()
|
||||
|
||||
|
||||
class SheetSage2Tokenizer:
|
||||
"""Typed prompt and event vocabulary for prompt-conditioned transcription."""
|
||||
|
||||
meter_numerators = tuple(range(1, 33))
|
||||
meter_denominators = (1, 2, 4, 8, 16, 32)
|
||||
n_eighth_positions = 256
|
||||
max_subbeat_shift = 256
|
||||
prompt_capacity = 256
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
audio_length_seconds=300.0,
|
||||
time_hz=100,
|
||||
schema_version="v1",
|
||||
expected_fingerprint=None,
|
||||
):
|
||||
self.audio_length_seconds = float(audio_length_seconds)
|
||||
self.time_hz = int(time_hz)
|
||||
self.schema = get_prompt_multitask_schema(schema_version)
|
||||
self.schema_version = self.schema.version
|
||||
self.task_specs = self.schema.tasks
|
||||
self.event_field_order = self.schema.event_field_order
|
||||
self.n_time_tokens = int(round(self.audio_length_seconds * self.time_hz))
|
||||
if self.n_time_tokens <= 0:
|
||||
raise ValueError("audio_length_seconds must produce at least one time token")
|
||||
|
||||
self.pad_token = 0
|
||||
self.sos_token = 1
|
||||
self.bos_token = self.sos_token
|
||||
self.eos_token = 2
|
||||
self.out_token = 3
|
||||
|
||||
self.prompt_token_start = 4
|
||||
self.prompt_names = tuple(task.name for task in self.task_specs)
|
||||
if len(self.prompt_names) > self.prompt_capacity:
|
||||
raise ValueError(
|
||||
f"Schema {self.schema_version} has {len(self.prompt_names)} prompts, "
|
||||
f"exceeding immutable capacity {self.prompt_capacity}"
|
||||
)
|
||||
self.prompt_to_id = {
|
||||
name: self.prompt_token_start + index
|
||||
for index, name in enumerate(self.prompt_names)
|
||||
}
|
||||
self.prompt_token_end = self.prompt_token_start + self.prompt_capacity
|
||||
|
||||
self.subbeat_shift_token_start = self.prompt_token_end
|
||||
self.n_subbeat_shift_tokens = self.max_subbeat_shift + 1
|
||||
self.subbeat_shift_token_end = (
|
||||
self.subbeat_shift_token_start + self.n_subbeat_shift_tokens
|
||||
)
|
||||
|
||||
self.time_token_start = self.subbeat_shift_token_end
|
||||
self.time_token_end = self.time_token_start + self.n_time_tokens
|
||||
|
||||
self.meter_pairs = tuple(
|
||||
(numerator, denominator)
|
||||
for numerator in self.meter_numerators
|
||||
for denominator in self.meter_denominators
|
||||
)
|
||||
self.meter_to_id = {meter: index for index, meter in enumerate(self.meter_pairs)}
|
||||
self.meter_token_start = self.time_token_end
|
||||
self.meter_token_end = self.meter_token_start + len(self.meter_pairs)
|
||||
|
||||
self.eighth_position_token_start = self.meter_token_end
|
||||
self.eighth_position_token_end = (
|
||||
self.eighth_position_token_start + self.n_eighth_positions
|
||||
)
|
||||
|
||||
self.structure_labels = tuple(STRUCTURE_LABELS)
|
||||
self.structure_token_start = self.eighth_position_token_end
|
||||
self.structure_token_end = self.structure_token_start + len(self.structure_labels)
|
||||
|
||||
self.key_token_start = self.structure_token_end
|
||||
self.n_key_tokens = 24
|
||||
self.key_token_end = self.key_token_start + self.n_key_tokens
|
||||
|
||||
self.majmin_chord_labels = (
|
||||
"N",
|
||||
*(f"{root}:maj" for root in CHROMATIC_SHARPS),
|
||||
*(f"{root}:min" for root in CHROMATIC_SHARPS),
|
||||
)
|
||||
self.majmin_chord_token_start = self.key_token_end
|
||||
self.majmin_chord_token_end = (
|
||||
self.majmin_chord_token_start + len(self.majmin_chord_labels)
|
||||
)
|
||||
|
||||
self.full_chord_labels = FULL_CHORD_VOCABULARY
|
||||
self.full_chord_to_id = {
|
||||
label: index for index, label in enumerate(self.full_chord_labels)
|
||||
}
|
||||
self.full_chord_token_start = self.majmin_chord_token_end
|
||||
self.full_chord_token_end = (
|
||||
self.full_chord_token_start + len(self.full_chord_labels)
|
||||
)
|
||||
|
||||
self.pitch_token_start = self.full_chord_token_end
|
||||
self.n_pitch_tokens = 256
|
||||
self.pitch_token_end = self.pitch_token_start + self.n_pitch_tokens
|
||||
|
||||
self.duration_templates = DURATION_TEMPLATES
|
||||
self.duration_boundaries = duration_boundaries
|
||||
self.duration_token_start = self.pitch_token_end
|
||||
self.n_duration_tokens = len(self.duration_templates)
|
||||
self.duration_token_end = self.duration_token_start + self.n_duration_tokens
|
||||
self.appended_token_blocks = {}
|
||||
next_token = self.duration_token_end
|
||||
for block in self.schema.appended_token_blocks:
|
||||
start = next_token
|
||||
end = start + len(block.labels)
|
||||
self.appended_token_blocks[block.name] = {
|
||||
"start": start,
|
||||
"end": end,
|
||||
"labels": tuple(block.labels),
|
||||
"output_field": block.output_field or block.name,
|
||||
}
|
||||
next_token = end
|
||||
self.n_tokens = next_token
|
||||
|
||||
self._full_chord_templates = self._build_full_chord_templates()
|
||||
self.vocab_fingerprint = self._compute_fingerprint()
|
||||
if (
|
||||
expected_fingerprint is not None
|
||||
and str(expected_fingerprint) != self.vocab_fingerprint
|
||||
):
|
||||
raise ValueError(
|
||||
f"Tokenizer fingerprint mismatch for schema {self.schema_version}: "
|
||||
f"expected {expected_fingerprint}, got {self.vocab_fingerprint}"
|
||||
)
|
||||
|
||||
def _compute_fingerprint(self):
|
||||
payload = {
|
||||
"schema_version": self.schema_version,
|
||||
"audio_length_seconds": self.audio_length_seconds,
|
||||
"time_hz": self.time_hz,
|
||||
"prompt_capacity": self.prompt_capacity,
|
||||
"prompt_names": self.prompt_names,
|
||||
"event_field_order": self.event_field_order,
|
||||
"meter_pairs": self.meter_pairs,
|
||||
"structure_labels": self.structure_labels,
|
||||
"majmin_chord_labels": self.majmin_chord_labels,
|
||||
"full_chord_labels": self.full_chord_labels,
|
||||
"duration_templates": tuple(int(value) for value in self.duration_templates),
|
||||
"appended_token_blocks": tuple(
|
||||
(name, block["labels"], block["output_field"])
|
||||
for name, block in self.appended_token_blocks.items()
|
||||
),
|
||||
"n_tokens": self.n_tokens,
|
||||
}
|
||||
encoded = json.dumps(payload, sort_keys=True, separators=(",", ":")).encode(
|
||||
"utf-8"
|
||||
)
|
||||
return hashlib.sha256(encoded).hexdigest()[:16]
|
||||
|
||||
def get_config(self):
|
||||
return {
|
||||
"schema_version": self.schema_version,
|
||||
"audio_length_seconds": self.audio_length_seconds,
|
||||
"time_hz": self.time_hz,
|
||||
"vocab_fingerprint": self.vocab_fingerprint,
|
||||
"n_tokens": self.n_tokens,
|
||||
}
|
||||
|
||||
def normalize_prompts(self, prompts):
|
||||
names = []
|
||||
seen = set()
|
||||
for prompt in prompts:
|
||||
name = str(prompt).strip()
|
||||
if name.startswith("<|") and name.endswith("|>"):
|
||||
name = name[2:-2]
|
||||
if name not in self.prompt_to_id:
|
||||
raise ValueError(f"Unknown prompt: {prompt!r}")
|
||||
if name not in seen:
|
||||
names.append(name)
|
||||
seen.add(name)
|
||||
names.sort(key=self.prompt_names.index)
|
||||
selected_groups = {}
|
||||
task_by_name = {task.name: task for task in self.task_specs}
|
||||
for name in names:
|
||||
group = task_by_name[name].sampling_group
|
||||
previous = selected_groups.get(group)
|
||||
if previous is not None:
|
||||
raise ValueError(
|
||||
f"Prompts {previous!r} and {name!r} are mutually exclusive "
|
||||
f"within sampling group {group!r}"
|
||||
)
|
||||
selected_groups[group] = name
|
||||
if not names:
|
||||
raise ValueError("At least one task prompt is required")
|
||||
return tuple(names)
|
||||
|
||||
def prompt_to_token(self, prompt):
|
||||
return self.prompt_to_id[self.normalize_prompts((prompt,))[0]]
|
||||
|
||||
def token_to_prompt(self, token):
|
||||
token = int(token)
|
||||
for prompt, prompt_token in self.prompt_to_id.items():
|
||||
if token == prompt_token:
|
||||
return prompt
|
||||
raise ValueError(f"token {token} is not a prompt token")
|
||||
|
||||
def prompt_prefix(self, prompts):
|
||||
prompts = self.normalize_prompts(prompts)
|
||||
return [
|
||||
self.sos_token,
|
||||
*(self.prompt_to_id[prompt] for prompt in prompts),
|
||||
self.out_token,
|
||||
]
|
||||
|
||||
def subbeat_shift_to_tokens(self, shift):
|
||||
shift = int(shift)
|
||||
if shift < 0:
|
||||
raise ValueError("subbeat shift must be non-negative")
|
||||
tokens = []
|
||||
while shift > self.max_subbeat_shift:
|
||||
tokens.append(self.subbeat_shift_token_start + self.max_subbeat_shift)
|
||||
shift -= self.max_subbeat_shift
|
||||
tokens.append(self.subbeat_shift_token_start + shift)
|
||||
return tokens
|
||||
|
||||
def token_to_subbeat_shift(self, token):
|
||||
token = int(token)
|
||||
if not self.subbeat_shift_token_start <= token < self.subbeat_shift_token_end:
|
||||
raise ValueError(f"token {token} is not a subbeat shift token")
|
||||
return token - self.subbeat_shift_token_start
|
||||
|
||||
def time_id_to_token(self, time_id):
|
||||
time_id = int(time_id)
|
||||
if not 0 <= time_id < self.n_time_tokens:
|
||||
raise ValueError(f"time id {time_id} is outside [0, {self.n_time_tokens})")
|
||||
return self.time_token_start + time_id
|
||||
|
||||
def token_to_time_id(self, token):
|
||||
token = int(token)
|
||||
if not self.time_token_start <= token < self.time_token_end:
|
||||
raise ValueError(f"token {token} is not a time token")
|
||||
return token - self.time_token_start
|
||||
|
||||
def meter_to_token(self, numerator, denominator):
|
||||
meter = (int(numerator), int(denominator))
|
||||
if meter not in self.meter_to_id:
|
||||
raise ValueError(f"Unsupported meter {meter[0]}/{meter[1]}")
|
||||
return self.meter_token_start + self.meter_to_id[meter]
|
||||
|
||||
def eighth_position_to_token(self, position):
|
||||
position = int(position)
|
||||
if not 0 <= position < self.n_eighth_positions:
|
||||
raise ValueError(
|
||||
f"eighth-note position {position} is outside [0, {self.n_eighth_positions})"
|
||||
)
|
||||
return self.eighth_position_token_start + position
|
||||
|
||||
def structure_to_token(self, label):
|
||||
label = str(label)
|
||||
if label not in self.structure_labels:
|
||||
raise ValueError(f"Unknown structure label: {label!r}")
|
||||
return self.structure_token_start + self.structure_labels.index(label)
|
||||
|
||||
def key_to_token(self, key_label, pitch_shift=0):
|
||||
tonic, mode = str(key_label).split(":", 1)
|
||||
tonic_id = mir_eval.chord.pitch_class_to_semitone(tonic)
|
||||
if tonic_id < 0:
|
||||
raise ValueError(f"Invalid key tonic: {key_label!r}")
|
||||
tonic_id = (int(tonic_id) + int(pitch_shift)) % 12
|
||||
minor_modes = {"minor", "dorian", "phrygian", "locrian"}
|
||||
mode_id = 1 if mode.lower() in minor_modes else 0
|
||||
return self.key_token_start + mode_id * 12 + tonic_id
|
||||
|
||||
@staticmethod
|
||||
def _canonical_chord_parts(chord_label, pitch_shift=0):
|
||||
chord_label = str(chord_label).strip()
|
||||
if chord_label in {"N", "X", ""}:
|
||||
return "N", None, None
|
||||
root, suffix = chord_label.split(":", 1)
|
||||
root_id = mir_eval.chord.pitch_class_to_semitone(root)
|
||||
if root_id < 0:
|
||||
return "N", None, None
|
||||
root_id = (int(root_id) + int(pitch_shift)) % 12
|
||||
canonical = f"{CHROMATIC_SHARPS[root_id]}:{suffix}"
|
||||
quality = re.split(r"/", suffix, maxsplit=1)[0]
|
||||
return canonical, root_id, quality
|
||||
|
||||
def chord_majmin_to_token(self, chord_label, pitch_shift=0):
|
||||
canonical, root_id, quality = self._canonical_chord_parts(
|
||||
chord_label,
|
||||
pitch_shift,
|
||||
)
|
||||
if canonical == "N":
|
||||
return self.majmin_chord_token_start
|
||||
|
||||
try:
|
||||
original_root, chroma, _ = mir_eval.chord.encode(str(chord_label))
|
||||
relative = mir_eval.chord.rotate_bitmap_to_root(chroma, original_root)
|
||||
has_minor_third = bool(relative[3] > 0)
|
||||
has_major_third = bool(relative[4] > 0)
|
||||
except Exception:
|
||||
has_minor_third = quality.startswith(("min", "dim", "hdim"))
|
||||
has_major_third = not has_minor_third
|
||||
|
||||
if has_major_third and not has_minor_third:
|
||||
quality_id = 0
|
||||
elif has_minor_third and not has_major_third:
|
||||
quality_id = 1
|
||||
elif quality.startswith(("min", "dim", "hdim")):
|
||||
quality_id = 1
|
||||
elif has_major_third:
|
||||
quality_id = 0
|
||||
else:
|
||||
return self.majmin_chord_token_start
|
||||
return self.majmin_chord_token_start + 1 + quality_id * 12 + root_id
|
||||
|
||||
@staticmethod
|
||||
def _chord_template(chord_label):
|
||||
root, chroma, bass = mir_eval.chord.encode(chord_label)
|
||||
root_chroma = np.zeros(12, dtype=np.float32)
|
||||
root_chroma[root] = 1.0
|
||||
relative_chroma = mir_eval.chord.rotate_bitmap_to_root(chroma, root).astype(
|
||||
np.float32
|
||||
)
|
||||
bass_chroma = np.zeros(12, dtype=np.float32)
|
||||
bass_chroma[(bass + root) % 12] = 1.0
|
||||
return np.concatenate([root_chroma, relative_chroma, bass_chroma])
|
||||
|
||||
def _build_full_chord_templates(self):
|
||||
templates = [np.zeros(36, dtype=np.float32)]
|
||||
templates.extend(
|
||||
self._chord_template(label) for label in self.full_chord_labels[1:]
|
||||
)
|
||||
return np.stack(templates)
|
||||
|
||||
def chord_full_to_token(self, chord_label, pitch_shift=0):
|
||||
canonical, _root_id, _quality = self._canonical_chord_parts(
|
||||
chord_label,
|
||||
pitch_shift,
|
||||
)
|
||||
chord_id = self.full_chord_to_id.get(canonical)
|
||||
if chord_id is None:
|
||||
if canonical == "N":
|
||||
chord_id = 0
|
||||
else:
|
||||
try:
|
||||
target = self._chord_template(canonical)
|
||||
distances = np.abs(self._full_chord_templates - target[None]).sum(
|
||||
axis=1
|
||||
)
|
||||
chord_id = int(np.argmin(distances))
|
||||
except Exception:
|
||||
chord_id = 0
|
||||
return self.full_chord_token_start + chord_id
|
||||
|
||||
def pitch_to_token(self, pitch, track=0, full_melody=False):
|
||||
pitch = int(pitch)
|
||||
track = int(track)
|
||||
if not 0 <= pitch < 128:
|
||||
raise ValueError(f"MIDI pitch {pitch} is outside [0, 128)")
|
||||
if track not in (0, 1):
|
||||
raise ValueError(f"melody track {track} is outside [0, 2)")
|
||||
pitch_id = pitch + (128 if full_melody and track == 1 else 0)
|
||||
return self.pitch_token_start + pitch_id
|
||||
|
||||
def duration_bin_to_token(self, duration_bin):
|
||||
duration_bin = int(duration_bin)
|
||||
if not 0 <= duration_bin < self.n_duration_tokens:
|
||||
raise ValueError(
|
||||
f"duration bin {duration_bin} is outside [0, {self.n_duration_tokens})"
|
||||
)
|
||||
return self.duration_token_start + duration_bin
|
||||
|
||||
def token_to_duration_bin(self, token):
|
||||
token = int(token)
|
||||
if not self.duration_token_start <= token < self.duration_token_end:
|
||||
raise ValueError(f"token {token} is not a duration token")
|
||||
return token - self.duration_token_start
|
||||
|
||||
def appended_label_to_token(self, block_name, label):
|
||||
block = self.appended_token_blocks.get(str(block_name))
|
||||
if block is None:
|
||||
raise ValueError(f"unknown appended token block: {block_name!r}")
|
||||
try:
|
||||
index = block["labels"].index(str(label))
|
||||
except ValueError as exc:
|
||||
raise ValueError(
|
||||
f"unknown label {label!r} for appended block {block_name!r}"
|
||||
) from exc
|
||||
return block["start"] + index
|
||||
|
||||
def token_to_appended_label(self, token):
|
||||
token = int(token)
|
||||
token_type = self.token_type(token)
|
||||
block = self.appended_token_blocks.get(token_type)
|
||||
if block is None:
|
||||
raise ValueError(f"token {token} is not from an appended token block")
|
||||
return token_type, block["labels"][token - block["start"]]
|
||||
|
||||
def _token_output_field(self, token_type):
|
||||
built_in = {
|
||||
"time": "timestamp",
|
||||
"meter": "rhythm",
|
||||
"eighth_position": "rhythm",
|
||||
"structure": "structure",
|
||||
"key": "key",
|
||||
"chord_majmin": "chord",
|
||||
"chord_full": "chord",
|
||||
"pitch": "melody",
|
||||
"duration": "melody",
|
||||
}
|
||||
if token_type in built_in:
|
||||
return built_in[token_type]
|
||||
block = self.appended_token_blocks.get(token_type)
|
||||
return None if block is None else block["output_field"]
|
||||
|
||||
def _decode_field(self, field, field_tokens, prompts):
|
||||
token_types = [self.token_type(token) for token in field_tokens]
|
||||
if field == "timestamp":
|
||||
return self.token_to_time_id(field_tokens[0]) / self.time_hz
|
||||
if field == "rhythm":
|
||||
rhythm = {}
|
||||
for token, token_type in zip(field_tokens, token_types):
|
||||
if token_type == "meter":
|
||||
rhythm["meter"] = self.meter_pairs[token - self.meter_token_start]
|
||||
elif token_type == "eighth_position":
|
||||
rhythm["eighth_position"] = token - self.eighth_position_token_start
|
||||
return rhythm
|
||||
if field == "structure":
|
||||
return self.structure_labels[field_tokens[0] - self.structure_token_start]
|
||||
if field == "key":
|
||||
key_id = field_tokens[0] - self.key_token_start
|
||||
mode = "minor" if key_id >= 12 else "major"
|
||||
return f"{CHROMATIC_SHARPS[key_id % 12]}:{mode}"
|
||||
if field == "chord":
|
||||
if token_types[0] == "chord_majmin":
|
||||
return self.majmin_chord_labels[
|
||||
field_tokens[0] - self.majmin_chord_token_start
|
||||
]
|
||||
return self.full_chord_labels[
|
||||
field_tokens[0] - self.full_chord_token_start
|
||||
]
|
||||
if field == "melody":
|
||||
notes = []
|
||||
index = 0
|
||||
while index < len(field_tokens):
|
||||
pitch_id = field_tokens[index] - self.pitch_token_start
|
||||
duration_bin = 0
|
||||
if (
|
||||
index + 1 < len(field_tokens)
|
||||
and self.token_type(field_tokens[index + 1]) == "duration"
|
||||
):
|
||||
duration_bin = self.token_to_duration_bin(field_tokens[index + 1])
|
||||
index += 2
|
||||
else:
|
||||
index += 1
|
||||
notes.append(
|
||||
{
|
||||
"pitch": pitch_id % 128,
|
||||
"track": int(pitch_id >= 128),
|
||||
"duration_bin": duration_bin,
|
||||
"duration_steps": int(self.duration_templates[duration_bin]),
|
||||
}
|
||||
)
|
||||
return notes
|
||||
values = []
|
||||
for token in field_tokens:
|
||||
block_name, label = self.token_to_appended_label(token)
|
||||
values.append({"block": block_name, "label": label})
|
||||
return values[0] if len(values) == 1 else values
|
||||
|
||||
def decode_sequence(self, tokens, strict=True):
|
||||
"""Parse one prompt-conditioned sequence into timed, typed events."""
|
||||
if hasattr(tokens, "detach"):
|
||||
tokens = tokens.detach().cpu().tolist()
|
||||
tokens = [int(token) for token in tokens]
|
||||
while tokens and tokens[-1] == self.pad_token:
|
||||
tokens.pop()
|
||||
if not tokens or tokens[0] != self.sos_token:
|
||||
raise ValueError("sequence must begin with <|sos|>")
|
||||
|
||||
try:
|
||||
out_index = tokens.index(self.out_token, 1)
|
||||
except ValueError as exc:
|
||||
raise ValueError("sequence is missing <|out|>") from exc
|
||||
prompts = tuple(self.token_to_prompt(token) for token in tokens[1:out_index])
|
||||
if strict and self.normalize_prompts(prompts) != prompts:
|
||||
raise ValueError("prompt tokens are not in canonical schema order")
|
||||
active_fields = {
|
||||
task.output_field for task in self.task_specs if task.name in prompts
|
||||
}
|
||||
|
||||
events = []
|
||||
position = out_index + 1
|
||||
current_step = 0
|
||||
saw_eos = False
|
||||
while position < len(tokens):
|
||||
token = tokens[position]
|
||||
if token == self.eos_token:
|
||||
saw_eos = True
|
||||
position += 1
|
||||
break
|
||||
if self.token_type(token) != "subbeat_shift":
|
||||
raise ValueError(f"event at token index {position} has no subbeat shift")
|
||||
shift = 0
|
||||
while (
|
||||
position < len(tokens)
|
||||
and self.token_type(tokens[position]) == "subbeat_shift"
|
||||
):
|
||||
shift += self.token_to_subbeat_shift(tokens[position])
|
||||
position += 1
|
||||
current_step += shift
|
||||
|
||||
tokens_by_field = {field: [] for field in self.event_field_order}
|
||||
while position < len(tokens):
|
||||
token = tokens[position]
|
||||
token_type = self.token_type(token)
|
||||
if token_type == "subbeat_shift" or token == self.eos_token:
|
||||
break
|
||||
field = self._token_output_field(token_type)
|
||||
if field is None or field not in tokens_by_field:
|
||||
raise ValueError(
|
||||
f"token {token} ({token_type}) has no field in schema "
|
||||
f"{self.schema_version}"
|
||||
)
|
||||
if strict and field not in active_fields:
|
||||
raise ValueError(
|
||||
f"token {token} belongs to inactive output field {field!r}"
|
||||
)
|
||||
tokens_by_field[field].append(token)
|
||||
position += 1
|
||||
|
||||
tokens_by_field = {
|
||||
field: values for field, values in tokens_by_field.items() if values
|
||||
}
|
||||
if not tokens_by_field:
|
||||
if strict:
|
||||
raise ValueError(f"empty event at subbeat {current_step}")
|
||||
continue
|
||||
if strict:
|
||||
for field, values in tokens_by_field.items():
|
||||
types = [self.token_type(value) for value in values]
|
||||
if field == "timestamp" and types != ["time"]:
|
||||
raise ValueError("timestamp event must contain exactly one time token")
|
||||
if field == "rhythm":
|
||||
if types not in (["eighth_position"], ["meter", "eighth_position"]):
|
||||
raise ValueError(f"invalid rhythm payload: {types}")
|
||||
if field in {"structure", "key", "chord"} and len(values) != 1:
|
||||
raise ValueError(f"field {field!r} must contain exactly one token")
|
||||
if field == "melody":
|
||||
index = 0
|
||||
while index < len(types):
|
||||
if types[index] != "pitch":
|
||||
raise ValueError(
|
||||
"melody payload must contain pitch tokens with optional duration"
|
||||
)
|
||||
if index + 1 < len(types) and types[index + 1] == "duration":
|
||||
index += 2
|
||||
else:
|
||||
index += 1
|
||||
events.append(
|
||||
{
|
||||
"subbeat": current_step,
|
||||
"tokens_by_field": tokens_by_field,
|
||||
"values": {
|
||||
field: self._decode_field(field, values, prompts)
|
||||
for field, values in tokens_by_field.items()
|
||||
},
|
||||
}
|
||||
)
|
||||
|
||||
if strict and not saw_eos:
|
||||
raise ValueError("sequence is missing <|eos|>")
|
||||
if strict and position != len(tokens):
|
||||
raise ValueError("non-padding tokens follow <|eos|>")
|
||||
return {
|
||||
"schema_version": self.schema_version,
|
||||
"prompts": prompts,
|
||||
"events": events,
|
||||
"has_eos": saw_eos,
|
||||
}
|
||||
|
||||
def encode_decoded_sequence(self, decoded):
|
||||
"""Re-encode the lossless representation returned by decode_sequence."""
|
||||
prompts = self.normalize_prompts(decoded["prompts"])
|
||||
output = self.prompt_prefix(prompts)
|
||||
previous_step = 0
|
||||
for event in decoded["events"]:
|
||||
step = int(event["subbeat"])
|
||||
if step < previous_step:
|
||||
raise ValueError("events must be sorted by non-decreasing subbeat")
|
||||
output.extend(self.subbeat_shift_to_tokens(step - previous_step))
|
||||
previous_step = step
|
||||
tokens_by_field = event["tokens_by_field"]
|
||||
for field in self.event_field_order:
|
||||
output.extend(int(token) for token in tokens_by_field.get(field, ()))
|
||||
if decoded.get("has_eos", True):
|
||||
output.append(self.eos_token)
|
||||
return output
|
||||
|
||||
def token_type(self, token):
|
||||
token = int(token)
|
||||
for name, start, end in (
|
||||
("subbeat_shift", self.subbeat_shift_token_start, self.subbeat_shift_token_end),
|
||||
("time", self.time_token_start, self.time_token_end),
|
||||
("meter", self.meter_token_start, self.meter_token_end),
|
||||
("eighth_position", self.eighth_position_token_start, self.eighth_position_token_end),
|
||||
("structure", self.structure_token_start, self.structure_token_end),
|
||||
("key", self.key_token_start, self.key_token_end),
|
||||
("chord_majmin", self.majmin_chord_token_start, self.majmin_chord_token_end),
|
||||
("chord_full", self.full_chord_token_start, self.full_chord_token_end),
|
||||
("pitch", self.pitch_token_start, self.pitch_token_end),
|
||||
("duration", self.duration_token_start, self.duration_token_end),
|
||||
):
|
||||
if start <= token < end:
|
||||
return name
|
||||
if token in self.prompt_to_id.values():
|
||||
return "prompt"
|
||||
for name, block in self.appended_token_blocks.items():
|
||||
if block["start"] <= token < block["end"]:
|
||||
return name
|
||||
if token == self.pad_token:
|
||||
return "pad"
|
||||
if token == self.sos_token:
|
||||
return "sos"
|
||||
if token == self.eos_token:
|
||||
return "eos"
|
||||
if token == self.out_token:
|
||||
return "out"
|
||||
raise ValueError(f"token {token} is outside vocabulary size {self.n_tokens}")
|
||||
|
||||
def describe(self, token):
|
||||
token = int(token)
|
||||
token_type = self.token_type(token)
|
||||
if token_type == "prompt":
|
||||
prompt_by_id = {value: key for key, value in self.prompt_to_id.items()}
|
||||
return f"<|{prompt_by_id[token]}|>"
|
||||
if token_type == "subbeat_shift":
|
||||
return f"<subbeat_shift_{token - self.subbeat_shift_token_start}>"
|
||||
if token_type == "time":
|
||||
time_id = token - self.time_token_start
|
||||
return f"<time_{time_id / self.time_hz:.2f}s>"
|
||||
if token_type == "meter":
|
||||
numerator, denominator = self.meter_pairs[token - self.meter_token_start]
|
||||
return f"<meter_{numerator}/{denominator}>"
|
||||
if token_type == "eighth_position":
|
||||
return f"<eighth_pos_{token - self.eighth_position_token_start}>"
|
||||
if token_type == "structure":
|
||||
return f"<structure_{self.structure_labels[token - self.structure_token_start]}>"
|
||||
if token_type == "key":
|
||||
key_id = token - self.key_token_start
|
||||
mode = "minor" if key_id >= 12 else "major"
|
||||
return f"<key_{CHROMATIC_SHARPS[key_id % 12]}:{mode}>"
|
||||
if token_type == "chord_majmin":
|
||||
return f"<chord_majmin_{self.majmin_chord_labels[token - self.majmin_chord_token_start]}>"
|
||||
if token_type == "chord_full":
|
||||
return f"<chord_full_{self.full_chord_labels[token - self.full_chord_token_start]}>"
|
||||
if token_type == "pitch":
|
||||
pitch = token - self.pitch_token_start
|
||||
track = 1 if pitch >= 128 else 0
|
||||
return f"<pitch_{pitch % 128}_track_{track}>"
|
||||
if token_type == "duration":
|
||||
return f"<duration_{token - self.duration_token_start}>"
|
||||
if token_type in self.appended_token_blocks:
|
||||
block = self.appended_token_blocks[token_type]
|
||||
label = block["labels"][token - block["start"]]
|
||||
return f"<{token_type}_{label}>"
|
||||
return f"<|{token_type}|>"
|
||||
@@ -0,0 +1,15 @@
|
||||
# Third-party code notices
|
||||
|
||||
The Oobleck VAE and SnakeBeta implementation in `modeling_vae.py` is derived
|
||||
from stable-audio-tools commit `a6ae0cdf8b2eb1567a4b42ceadddec3712d99d45`.
|
||||
The module hierarchy, weight normalization and activation equations preserve
|
||||
the checkpoint's original inference implementation.
|
||||
|
||||
- Oobleck / stable-audio-tools: Copyright (c) 2023 Stability AI, MIT.
|
||||
Full text: `licenses/stable-audio-tools-MIT.txt`.
|
||||
- SnakeBeta / BigVGAN: Copyright (c) 2022 NVIDIA CORPORATION, MIT.
|
||||
Full text: `licenses/SnakeBeta-NVIDIA-MIT.txt`.
|
||||
|
||||
These notices cover the identified source code and retain its original licenses.
|
||||
The YuE2 model checkpoint weights are separately licensed under CC BY-NC 4.0;
|
||||
see MODEL_LICENSE for the scope and full terms. This does not relicense third-party code.
|
||||
@@ -0,0 +1,38 @@
|
||||
{
|
||||
"return_dict": true,
|
||||
"output_hidden_states": false,
|
||||
"torchscript": false,
|
||||
"dtype": "bfloat16",
|
||||
"tie_word_embeddings": false,
|
||||
"is_encoder_decoder": false,
|
||||
"is_decoder": false,
|
||||
"add_cross_attention": false,
|
||||
"architectures": [
|
||||
"YuE2ForCausalLM"
|
||||
],
|
||||
"bos_token_id": null,
|
||||
"pad_token_id": null,
|
||||
"eos_token_id": null,
|
||||
"decoder_start_token_id": null,
|
||||
"transformers_version": "4.57.6",
|
||||
"auto_map": {
|
||||
"AutoConfig": "modeling_yue2.YuE2Config",
|
||||
"AutoModelForCausalLM": "modeling_yue2.YuE2ForCausalLM"
|
||||
},
|
||||
"model_type": "yue2",
|
||||
"hidden_size": 2048,
|
||||
"num_hidden_layers": 28,
|
||||
"num_attention_heads": 16,
|
||||
"num_key_value_heads": 8,
|
||||
"head_dim": 128,
|
||||
"intermediate_size": 6144,
|
||||
"vocab_size": 184704,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_theta": 1000000,
|
||||
"max_position_embeddings": 24576,
|
||||
"latent_type": "vae",
|
||||
"latent_dim": 64,
|
||||
"max_latent_frames": 24576,
|
||||
"timestep_shift": 1.0,
|
||||
"output_attentions": false
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
{
|
||||
"bos_token_id": 151643,
|
||||
"eos_token_id": 151852,
|
||||
"pad_token_id": 151643,
|
||||
"do_sample": true,
|
||||
"temperature": 1.0,
|
||||
"top_p": 0.95,
|
||||
"top_k": 100,
|
||||
"max_new_tokens": 9000,
|
||||
"use_cache": true
|
||||
}
|
||||
@@ -0,0 +1,705 @@
|
||||
"""YuE2 AR–NAR Mixture-of-Transformers, with checkpoint-compatible names.
|
||||
|
||||
This module is self contained for Transformers ``trust_remote_code`` loading.
|
||||
It imports no CUDA extension and implements the released model architecture.
|
||||
``generate`` returns token IDs; the package pipeline supplies song generation.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import List, Optional, Tuple, Union
|
||||
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch.nn.functional as F
|
||||
from transformers import GenerationMixin, PretrainedConfig, PreTrainedModel
|
||||
from transformers.cache_utils import DynamicCache
|
||||
from transformers.modeling_outputs import CausalLMOutputWithPast
|
||||
|
||||
|
||||
def sdpa(query, key, value, *, attn_mask=None, is_causal=False):
|
||||
"""Use native grouped-query attention, including a portable MPS fallback."""
|
||||
grouped = query.shape[1] != key.shape[1]
|
||||
if grouped and query.device.type == "mps":
|
||||
# PyTorch's MPS attention does not implement enable_gqa on every release.
|
||||
groups = query.shape[1] // key.shape[1]
|
||||
key = key.repeat_interleave(groups, dim=1)
|
||||
value = value.repeat_interleave(groups, dim=1)
|
||||
grouped = False
|
||||
return F.scaled_dot_product_attention(
|
||||
query, key, value, attn_mask=attn_mask, is_causal=is_causal,
|
||||
enable_gqa=grouped,
|
||||
)
|
||||
|
||||
|
||||
def _causal_mask(attention_mask, cache_position, key_length, batch_size):
|
||||
"""Physical cache slots are causal; RoPE positions may exclude padding."""
|
||||
device = cache_position.device
|
||||
visible = torch.arange(key_length, device=device)[None, :] <= cache_position[:, None]
|
||||
visible = visible[None, None].expand(batch_size, 1, -1, -1)
|
||||
if attention_mask is None:
|
||||
return visible
|
||||
mask = attention_mask.to(device=device)
|
||||
if mask.ndim == 2:
|
||||
if mask.shape[0] != batch_size or mask.shape[1] > key_length:
|
||||
raise ValueError("attention_mask must cover the batch and used cache slots")
|
||||
# Static cache has unused capacity after the supplied 2D padding mask.
|
||||
if mask.shape[1] < key_length:
|
||||
mask = F.pad(mask, (0, key_length - mask.shape[1]), value=0)
|
||||
return visible & mask[:, None, None, :].bool()
|
||||
if mask.ndim != 4 or mask.shape[-2:] != visible.shape[-2:]:
|
||||
raise ValueError("Expected a 2D padding mask or a matching 4D attention mask")
|
||||
if mask.dtype == torch.bool:
|
||||
return visible & mask
|
||||
return mask.masked_fill(~visible, float("-inf"))
|
||||
|
||||
# ══════════════════════════════════════════════════════════════════════════════
|
||||
# Config
|
||||
# ══════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
|
||||
class YuE2Config(PretrainedConfig):
|
||||
model_type = "yue2"
|
||||
|
||||
_hf_fields = frozenset({
|
||||
"model_type", "architectures", "auto_map", "transformers_version",
|
||||
"dtype", "torch_dtype", "return_dict", "output_hidden_states",
|
||||
"output_attentions", "use_cache", "tie_word_embeddings", "torchscript",
|
||||
"is_decoder", "is_encoder_decoder", "add_cross_attention",
|
||||
"bos_token_id", "eos_token_id", "pad_token_id", "decoder_start_token_id",
|
||||
"attn_implementation",
|
||||
})
|
||||
|
||||
def to_dict(self):
|
||||
return {key: value for key, value in super().to_dict().items()
|
||||
if key in self._hf_fields or key in self._inference_fields}
|
||||
|
||||
_inference_fields = frozenset(['hidden_size', 'num_hidden_layers', 'num_attention_heads', 'num_key_value_heads', 'head_dim', 'intermediate_size', 'vocab_size', 'rms_norm_eps', 'rope_theta', 'max_position_embeddings', 'tie_word_embeddings', 'latent_type', 'latent_dim', 'max_latent_frames', 'timestep_shift'])
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
hidden_size: int = 2048,
|
||||
num_hidden_layers: int = 28,
|
||||
num_attention_heads: int = 16,
|
||||
num_key_value_heads: int = 8,
|
||||
head_dim: int = 128,
|
||||
intermediate_size: int = 6144,
|
||||
vocab_size: int = 184704,
|
||||
rms_norm_eps: float = 1e-6,
|
||||
rope_theta: float = 1000000.0,
|
||||
max_position_embeddings: int = 24576,
|
||||
tie_word_embeddings: bool = False,
|
||||
# Acoustic inference architecture
|
||||
latent_type: str = "vae",
|
||||
latent_dim: int = 64,
|
||||
max_latent_frames: int = 24576,
|
||||
timestep_shift: float = 1.0,
|
||||
**kwargs,
|
||||
):
|
||||
if latent_type != "vae":
|
||||
raise ValueError("YuE2 inference supports only latent_type='vae'")
|
||||
# Serialize only the documented model and Transformers configuration.
|
||||
kwargs = {key: value for key, value in kwargs.items() if key in self._hf_fields}
|
||||
super().__init__(tie_word_embeddings=tie_word_embeddings, **kwargs)
|
||||
self.hidden_size = hidden_size
|
||||
self.num_hidden_layers = num_hidden_layers
|
||||
self.num_attention_heads = num_attention_heads
|
||||
self.num_key_value_heads = num_key_value_heads
|
||||
self.head_dim = head_dim
|
||||
self.intermediate_size = intermediate_size
|
||||
self.vocab_size = vocab_size
|
||||
self.rms_norm_eps = rms_norm_eps
|
||||
self.rope_theta = rope_theta
|
||||
self.max_position_embeddings = max_position_embeddings
|
||||
self.latent_type = latent_type
|
||||
self.latent_dim = latent_dim
|
||||
self.max_latent_frames = max_latent_frames
|
||||
self.timestep_shift = timestep_shift
|
||||
|
||||
|
||||
# ══════════════════════════════════════════════════════════════════════════════
|
||||
# Building blocks
|
||||
# ══════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
|
||||
class RMSNorm(nn.Module):
|
||||
def __init__(self, dim: int, eps: float = 1e-6):
|
||||
super().__init__()
|
||||
self.weight = nn.Parameter(torch.ones(dim))
|
||||
self.eps = eps
|
||||
|
||||
def forward(self, x: torch.Tensor) -> torch.Tensor:
|
||||
return x * torch.rsqrt(x.float().pow(2).mean(-1, keepdim=True) + self.eps).to(x.dtype) * self.weight
|
||||
|
||||
|
||||
class RotaryEmbedding(nn.Module):
|
||||
def __init__(self, head_dim: int, base: float = 1000000.0):
|
||||
super().__init__()
|
||||
self.head_dim = head_dim
|
||||
self.base = base
|
||||
self._inv_freq: Optional[torch.Tensor] = None
|
||||
|
||||
def forward(self, position_ids: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
if self._inv_freq is None or self._inv_freq.device != position_ids.device:
|
||||
self._inv_freq = 1.0 / (self.base ** (
|
||||
torch.arange(0, self.head_dim, 2, dtype=torch.float32, device=position_ids.device) / self.head_dim
|
||||
))
|
||||
pos = position_ids.float().unsqueeze(-1)
|
||||
angles = pos * self._inv_freq
|
||||
return angles.cos(), angles.sin()
|
||||
|
||||
|
||||
def _apply_rotary(x: torch.Tensor, cos: torch.Tensor, sin: torch.Tensor) -> torch.Tensor:
|
||||
half = x.shape[-1] // 2
|
||||
x1, x2 = x[..., :half], x[..., half:]
|
||||
cos, sin = cos.to(x.dtype), sin.to(x.dtype)
|
||||
return torch.cat([x1 * cos - x2 * sin, x2 * cos + x1 * sin], dim=-1)
|
||||
|
||||
|
||||
class Attention(nn.Module):
|
||||
def __init__(self, config: YuE2Config):
|
||||
super().__init__()
|
||||
self.num_heads = config.num_attention_heads
|
||||
self.num_kv_heads = config.num_key_value_heads
|
||||
self.head_dim = config.head_dim
|
||||
self.num_kv_groups = self.num_heads // self.num_kv_heads
|
||||
|
||||
self.q_proj = nn.Linear(config.hidden_size, self.num_heads * self.head_dim, bias=False)
|
||||
self.k_proj = nn.Linear(config.hidden_size, self.num_kv_heads * self.head_dim, bias=False)
|
||||
self.v_proj = nn.Linear(config.hidden_size, self.num_kv_heads * self.head_dim, bias=False)
|
||||
self.o_proj = nn.Linear(self.num_heads * self.head_dim, config.hidden_size, bias=False)
|
||||
self.q_norm = RMSNorm(self.head_dim, config.rms_norm_eps)
|
||||
self.k_norm = RMSNorm(self.head_dim, config.rms_norm_eps)
|
||||
|
||||
def project_qkv(
|
||||
self, x: torch.Tensor, cos: torch.Tensor, sin: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
"""Project, normalize, and apply RoPE. No SDPA, no KV cache, no O proj.
|
||||
|
||||
Returns Q [B,T,num_heads,hd], K [B,T,num_kv_heads,hd], V [B,T,num_kv_heads,hd].
|
||||
"""
|
||||
B, T, _ = x.shape
|
||||
q = self.q_proj(x).view(B, T, self.num_heads, self.head_dim)
|
||||
k = self.k_proj(x).view(B, T, self.num_kv_heads, self.head_dim)
|
||||
v = self.v_proj(x).view(B, T, self.num_kv_heads, self.head_dim)
|
||||
q, k = self.q_norm(q), self.k_norm(k)
|
||||
rc, rs = cos.unsqueeze(2), sin.unsqueeze(2)
|
||||
q = _apply_rotary(q, rc, rs)
|
||||
k = _apply_rotary(k, rc, rs)
|
||||
return q, k, v
|
||||
|
||||
def forward(
|
||||
self,
|
||||
x: torch.Tensor,
|
||||
cos: torch.Tensor,
|
||||
sin: torch.Tensor,
|
||||
past_key_value: Optional[DynamicCache] = None,
|
||||
layer_idx: int = 0,
|
||||
attention_mask: Optional[torch.Tensor] = None,
|
||||
cache_position: Optional[torch.Tensor] = None,
|
||||
) -> torch.Tensor:
|
||||
B, T, _ = x.shape
|
||||
q, k, v = self.project_qkv(x, cos, sin)
|
||||
|
||||
q, k, v = q.transpose(1, 2), k.transpose(1, 2), v.transpose(1, 2)
|
||||
|
||||
if past_key_value is not None:
|
||||
k, v = past_key_value.update(k, v, layer_idx, {"cache_position": cache_position})
|
||||
|
||||
if attention_mask is not None:
|
||||
out = sdpa(q, k, v, attn_mask=attention_mask[..., :k.shape[2]])
|
||||
else:
|
||||
out = sdpa(q, k, v, is_causal=(T > 1 and k.shape[2] == T))
|
||||
return self.o_proj(out.transpose(1, 2).reshape(B, T, -1))
|
||||
|
||||
|
||||
class MLP(nn.Module):
|
||||
def __init__(self, config: YuE2Config):
|
||||
super().__init__()
|
||||
self.gate_proj = nn.Linear(config.hidden_size, config.intermediate_size, bias=False)
|
||||
self.up_proj = nn.Linear(config.hidden_size, config.intermediate_size, bias=False)
|
||||
self.down_proj = nn.Linear(config.intermediate_size, config.hidden_size, bias=False)
|
||||
|
||||
def forward(self, x: torch.Tensor) -> torch.Tensor:
|
||||
return self.down_proj(F.silu(self.gate_proj(x)) * self.up_proj(x))
|
||||
|
||||
|
||||
class DecoderLayer(nn.Module):
|
||||
"""Transformer layer with full MoT: dual attention projections + dual MLP."""
|
||||
|
||||
def __init__(self, config: YuE2Config):
|
||||
super().__init__()
|
||||
# AR attention path
|
||||
self.input_layernorm = RMSNorm(config.hidden_size, config.rms_norm_eps)
|
||||
self.self_attn = Attention(config)
|
||||
# NAR attention path (separate Q/K/V/O + layernorms)
|
||||
self.nar_input_layernorm = RMSNorm(config.hidden_size, config.rms_norm_eps)
|
||||
self.nar_self_attn = Attention(config)
|
||||
# AR MLP path
|
||||
self.post_attention_layernorm = RMSNorm(config.hidden_size, config.rms_norm_eps)
|
||||
self.mlp = MLP(config)
|
||||
# NAR MLP path
|
||||
self.nar_pre_mlp_layernorm = RMSNorm(config.hidden_size, config.rms_norm_eps)
|
||||
self.nar_mlp = MLP(config)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
x: torch.Tensor,
|
||||
cos: torch.Tensor,
|
||||
sin: torch.Tensor,
|
||||
past_key_value: Optional[DynamicCache] = None,
|
||||
layer_idx: int = 0,
|
||||
attention_mask: Optional[torch.Tensor] = None,
|
||||
ar_mask: Optional[torch.Tensor] = None,
|
||||
cache_position: Optional[torch.Tensor] = None,
|
||||
) -> torch.Tensor:
|
||||
if ar_mask is not None:
|
||||
mask_3d = ar_mask.unsqueeze(-1) # [B, S, 1]
|
||||
mask_4d = ar_mask.unsqueeze(-1).unsqueeze(-1) # [B, S, 1, 1]
|
||||
|
||||
# Per-type input layernorm
|
||||
ln_ar = self.input_layernorm(x)
|
||||
ln_nar = self.nar_input_layernorm(x)
|
||||
|
||||
# Per-type QKV projection (both process all tokens)
|
||||
q_ar, k_ar, v_ar = self.self_attn.project_qkv(ln_ar, cos, sin)
|
||||
q_nar, k_nar, v_nar = self.nar_self_attn.project_qkv(ln_nar, cos, sin)
|
||||
|
||||
# Merge Q/K/V per-position: AR positions use AR projections, NAR use NAR
|
||||
query = torch.where(mask_4d, q_ar, q_nar) # [B, S, num_heads, hd]
|
||||
# K/V have num_kv_heads (fewer), same mask broadcast works
|
||||
key = torch.where(mask_4d, k_ar, k_nar) # [B, S, num_kv_heads, hd]
|
||||
value = torch.where(mask_4d, v_ar, v_nar)
|
||||
|
||||
# Transpose to [B, H, S, D] for SDPA
|
||||
B, S = x.shape[:2]
|
||||
query = query.transpose(1, 2)
|
||||
key = key.transpose(1, 2)
|
||||
value = value.transpose(1, 2)
|
||||
|
||||
# Shared attention with hybrid mask
|
||||
if attention_mask is not None and attention_mask.dtype != torch.bool:
|
||||
attention_mask = attention_mask.to(query.dtype)
|
||||
core_out = sdpa(query, key, value, attn_mask=attention_mask)
|
||||
core_out = core_out.transpose(1, 2).reshape(B, S, -1)
|
||||
|
||||
# Per-type O projection, merge by mask
|
||||
o_ar = self.self_attn.o_proj(core_out)
|
||||
o_nar = self.nar_self_attn.o_proj(core_out)
|
||||
h = torch.where(mask_3d, o_ar, o_nar)
|
||||
x = x + h
|
||||
|
||||
# Per-type MLP
|
||||
ar_out = self.mlp(self.post_attention_layernorm(x))
|
||||
nar_out = self.nar_mlp(self.nar_pre_mlp_layernorm(x))
|
||||
mlp_out = torch.where(mask_3d, ar_out, nar_out)
|
||||
else:
|
||||
# AR-only mode (generation): use AR path only
|
||||
h = self.self_attn(self.input_layernorm(x), cos, sin, past_key_value, layer_idx,
|
||||
attention_mask, cache_position)
|
||||
x = x + h
|
||||
mlp_out = self.mlp(self.post_attention_layernorm(x))
|
||||
|
||||
x = x + mlp_out
|
||||
return x
|
||||
|
||||
|
||||
# ══════════════════════════════════════════════════════════════════════════════
|
||||
# NAR auxiliary modules
|
||||
# ══════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
|
||||
class TimestepEmbedder(nn.Module):
|
||||
"""Sinusoidal timestep → MLP → hidden_size (same as modules.py)."""
|
||||
|
||||
def __init__(self, hidden_size: int, frequency_embedding_size: int = 256):
|
||||
super().__init__()
|
||||
self.mlp = nn.Sequential(
|
||||
nn.Linear(frequency_embedding_size, hidden_size),
|
||||
nn.SiLU(),
|
||||
nn.Linear(hidden_size, hidden_size),
|
||||
)
|
||||
self.frequency_embedding_size = frequency_embedding_size
|
||||
|
||||
def forward(self, t):
|
||||
half = self.frequency_embedding_size // 2
|
||||
freqs = torch.exp(
|
||||
-math.log(10000) * torch.arange(half, device=t.device, dtype=torch.float32) / half
|
||||
)
|
||||
args = t.float().unsqueeze(-1) * freqs.unsqueeze(0)
|
||||
emb = torch.cat([torch.cos(args), torch.sin(args)], dim=-1)
|
||||
return self.mlp(emb.to(next(self.parameters()).dtype))
|
||||
|
||||
|
||||
class AudioPositionEmbedding(nn.Module):
|
||||
"""Non-learnable 1D sinusoidal PE for audio latent frames."""
|
||||
|
||||
def __init__(self, max_frames: int, hidden_size: int):
|
||||
super().__init__()
|
||||
pe = torch.zeros(max_frames, hidden_size)
|
||||
position = torch.arange(0, max_frames, dtype=torch.float32).unsqueeze(1)
|
||||
div_term = torch.exp(
|
||||
torch.arange(0, hidden_size, 2, dtype=torch.float32) * (-math.log(10000.0) / hidden_size)
|
||||
)
|
||||
pe[:, 0::2] = torch.sin(position * div_term)
|
||||
pe[:, 1::2] = torch.cos(position * div_term)
|
||||
self.register_buffer("pe", pe)
|
||||
|
||||
def forward(self, position_ids):
|
||||
return self.pe[position_ids]
|
||||
|
||||
|
||||
# ══════════════════════════════════════════════════════════════════════════════
|
||||
# Static KV Cache
|
||||
# ══════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
|
||||
class StaticKVCache:
|
||||
"""Bounded, append-only cache for the explicit single-request AR loop.
|
||||
|
||||
Returns views of the used prefix and never reallocates/copies its history.
|
||||
Standard HF ``generate`` also supports Transformers' own StaticCache.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self, num_layers: int, batch_size: int, num_kv_heads: int,
|
||||
max_seq_len: int, head_dim: int, dtype: torch.dtype, device: torch.device,
|
||||
):
|
||||
self.num_layers = num_layers
|
||||
self.max_seq_len = max_seq_len
|
||||
self._seen_tokens = 0
|
||||
self.key_cache: List[torch.Tensor] = [
|
||||
torch.zeros(batch_size, num_kv_heads, max_seq_len, head_dim, dtype=dtype, device=device)
|
||||
for _ in range(num_layers)
|
||||
]
|
||||
self.value_cache: List[torch.Tensor] = [
|
||||
torch.zeros(batch_size, num_kv_heads, max_seq_len, head_dim, dtype=dtype, device=device)
|
||||
for _ in range(num_layers)
|
||||
]
|
||||
|
||||
def get_seq_length(self, layer_idx=0) -> int:
|
||||
return self._seen_tokens
|
||||
|
||||
def update(self, key_states, value_states, layer_idx, cache_kwargs=None):
|
||||
T = key_states.shape[2]
|
||||
pos = self._seen_tokens
|
||||
end = pos + T
|
||||
if end > self.max_seq_len:
|
||||
raise ValueError(f"KV cache capacity {self.max_seq_len} exceeded by {end}; generation was not shortened")
|
||||
self.key_cache[layer_idx][:, :, pos:end] = key_states
|
||||
self.value_cache[layer_idx][:, :, pos:end] = value_states
|
||||
if layer_idx == self.num_layers - 1:
|
||||
self._seen_tokens = end
|
||||
return self.key_cache[layer_idx][:, :, :end], self.value_cache[layer_idx][:, :, :end]
|
||||
|
||||
def reset(self):
|
||||
self._seen_tokens = 0
|
||||
|
||||
def reorder_cache(self, beam_idx):
|
||||
self.key_cache = [v.index_select(0, beam_idx.to(v.device)) for v in self.key_cache]
|
||||
self.value_cache = [v.index_select(0, beam_idx.to(v.device)) for v in self.value_cache]
|
||||
|
||||
|
||||
# ══════════════════════════════════════════════════════════════════════════════
|
||||
# Model
|
||||
# ══════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
|
||||
class Backbone(nn.Module):
|
||||
"""Transformer backbone with MoT dual MLP."""
|
||||
|
||||
def __init__(self, config: YuE2Config):
|
||||
super().__init__()
|
||||
self.embed_tokens = nn.Embedding(config.vocab_size, config.hidden_size)
|
||||
self.layers = nn.ModuleList([DecoderLayer(config) for _ in range(config.num_hidden_layers)])
|
||||
self.norm = RMSNorm(config.hidden_size, config.rms_norm_eps)
|
||||
self.rotary_emb = RotaryEmbedding(config.head_dim, config.rope_theta)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
input_ids: Optional[torch.LongTensor] = None,
|
||||
position_ids: Optional[torch.LongTensor] = None,
|
||||
past_key_values=None,
|
||||
use_cache: bool = True,
|
||||
attention_mask: Optional[torch.Tensor] = None,
|
||||
ar_mask: Optional[torch.Tensor] = None,
|
||||
inputs_embeds: Optional[torch.Tensor] = None,
|
||||
cache_position: Optional[torch.Tensor] = None,
|
||||
) -> Tuple[torch.Tensor, ...]:
|
||||
if inputs_embeds is not None:
|
||||
x = inputs_embeds
|
||||
else:
|
||||
x = self.embed_tokens(input_ids)
|
||||
cos, sin = self.rotary_emb(position_ids)
|
||||
|
||||
if use_cache and past_key_values is None:
|
||||
past_key_values = DynamicCache()
|
||||
|
||||
for i, layer in enumerate(self.layers):
|
||||
x = layer(x, cos, sin, past_key_values if use_cache else None,
|
||||
layer_idx=i, attention_mask=attention_mask, ar_mask=ar_mask,
|
||||
cache_position=cache_position)
|
||||
|
||||
return self.norm(x), past_key_values
|
||||
|
||||
|
||||
class YuE2PreTrainedModel(PreTrainedModel):
|
||||
config_class = YuE2Config
|
||||
base_model_prefix = "model"
|
||||
supports_gradient_checkpointing = True
|
||||
_no_split_modules = ["DecoderLayer"]
|
||||
_supports_sdpa = True
|
||||
|
||||
def _init_weights(self, module):
|
||||
if isinstance(module, nn.Linear):
|
||||
nn.init.normal_(module.weight, std=0.01)
|
||||
if module.bias is not None:
|
||||
nn.init.zeros_(module.bias)
|
||||
elif isinstance(module, nn.Embedding):
|
||||
nn.init.normal_(module.weight, std=0.01)
|
||||
|
||||
|
||||
class YuE2ForCausalLM(YuE2PreTrainedModel, GenerationMixin):
|
||||
"""YuE2 model: AR causal LM (generate) + NAR flow matching (ODE)."""
|
||||
|
||||
def __init__(self, config: YuE2Config):
|
||||
super().__init__(config)
|
||||
self.model = Backbone(config)
|
||||
self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
|
||||
|
||||
# NAR auxiliary
|
||||
self.llm2vae = nn.Linear(config.hidden_size, config.latent_dim)
|
||||
self.vae2llm = nn.Linear(config.latent_dim, config.hidden_size)
|
||||
self.time_embedder = TimestepEmbedder(config.hidden_size)
|
||||
self.latent_pos_embed = AudioPositionEmbedding(config.max_latent_frames, config.hidden_size)
|
||||
|
||||
self.post_init()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.model.embed_tokens
|
||||
|
||||
def set_input_embeddings(self, value):
|
||||
self.model.embed_tokens = value
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_head
|
||||
|
||||
def set_output_embeddings(self, new_embeddings):
|
||||
self.lm_head = new_embeddings
|
||||
|
||||
# ── AR forward (standard causal LM, KV cached) ───────────────────
|
||||
|
||||
def forward(
|
||||
self,
|
||||
input_ids: Optional[torch.LongTensor] = None,
|
||||
attention_mask: Optional[torch.Tensor] = None,
|
||||
position_ids: Optional[torch.LongTensor] = None,
|
||||
past_key_values=None,
|
||||
inputs_embeds: Optional[torch.FloatTensor] = None,
|
||||
labels: Optional[torch.LongTensor] = None,
|
||||
use_cache: Optional[bool] = None,
|
||||
cache_position: Optional[torch.LongTensor] = None,
|
||||
logits_to_keep: Union[int, torch.Tensor] = 0,
|
||||
return_dict: Optional[bool] = None,
|
||||
**kwargs,
|
||||
) -> Union[Tuple, CausalLMOutputWithPast]:
|
||||
if (input_ids is None) == (inputs_embeds is None):
|
||||
raise ValueError("Supply exactly one of input_ids or inputs_embeds")
|
||||
use_cache = use_cache if use_cache is not None else getattr(self.config, "use_cache", True)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
tensor = input_ids if input_ids is not None else inputs_embeds
|
||||
batch_size, seq_len = tensor.shape[:2]
|
||||
if not seq_len:
|
||||
raise ValueError("Input must contain at least one token")
|
||||
device = tensor.device
|
||||
past_len = past_key_values.get_seq_length() if past_key_values is not None and use_cache else 0
|
||||
if cache_position is None:
|
||||
cache_position = torch.arange(past_len, past_len + seq_len, device=device)
|
||||
else:
|
||||
cache_position = cache_position.to(device=device, dtype=torch.long)
|
||||
if cache_position.ndim != 1 or cache_position.numel() != seq_len:
|
||||
raise ValueError("cache_position must identify each current token's physical cache slot")
|
||||
if position_ids is None:
|
||||
if attention_mask is not None and attention_mask.ndim == 2:
|
||||
position_ids = attention_mask.long().cumsum(-1) - 1
|
||||
position_ids.masked_fill_(attention_mask == 0, 0)
|
||||
position_ids = position_ids[:, -seq_len:].to(device)
|
||||
else:
|
||||
position_ids = cache_position[None]
|
||||
else:
|
||||
position_ids = position_ids.to(device=device, dtype=torch.long)
|
||||
if position_ids.shape[-1] != seq_len:
|
||||
raise ValueError("position_ids must cover the current input tokens")
|
||||
|
||||
key_length = past_len + seq_len
|
||||
if use_cache and past_key_values is not None and hasattr(past_key_values, "get_max_cache_shape"):
|
||||
capacity = past_key_values.get_max_cache_shape()
|
||||
if capacity is not None and capacity > 0:
|
||||
key_length = capacity
|
||||
# No explicit mask is needed for unpadded prefill or single-token dynamic
|
||||
# decode. Chunked prefill needs bottom-right causal alignment; a full
|
||||
# static cache additionally needs to hide all unfilled slots.
|
||||
needs_mask = attention_mask is not None or key_length != past_len + seq_len or (past_len > 0 and seq_len > 1)
|
||||
causal_mask = _causal_mask(attention_mask, cache_position, key_length, batch_size) if needs_mask else None
|
||||
hidden_states, past_key_values = self.model(
|
||||
input_ids=input_ids, position_ids=position_ids,
|
||||
past_key_values=past_key_values, use_cache=use_cache,
|
||||
attention_mask=causal_mask, inputs_embeds=inputs_embeds,
|
||||
cache_position=cache_position,
|
||||
)
|
||||
|
||||
if isinstance(logits_to_keep, int):
|
||||
if logits_to_keep < 0:
|
||||
raise ValueError("logits_to_keep must be nonnegative")
|
||||
selected = hidden_states[:, -logits_to_keep:, :] if logits_to_keep else hidden_states
|
||||
else:
|
||||
selected = hidden_states[:, logits_to_keep.to(device), :]
|
||||
if labels is not None and selected.shape[1] != hidden_states.shape[1]:
|
||||
raise ValueError("Loss computation requires logits_to_keep=0")
|
||||
logits = self.lm_head(selected)
|
||||
|
||||
loss = None
|
||||
if labels is not None:
|
||||
shift_logits = logits[..., :-1, :].contiguous()
|
||||
shift_labels = labels[..., 1:].contiguous()
|
||||
loss = F.cross_entropy(shift_logits.view(-1, shift_logits.size(-1)), shift_labels.view(-1))
|
||||
|
||||
if not return_dict:
|
||||
output = (logits, past_key_values) if use_cache else (logits,)
|
||||
return ((loss,) + output) if loss is not None else output
|
||||
return CausalLMOutputWithPast(loss=loss, logits=logits, past_key_values=past_key_values if use_cache else None)
|
||||
|
||||
def prepare_inputs_for_generation(
|
||||
self, input_ids, past_key_values=None, attention_mask=None,
|
||||
inputs_embeds=None, cache_position=None, position_ids=None, **kwargs,
|
||||
):
|
||||
"""Keep physical cache slots separate from padding-aware RoPE positions."""
|
||||
past_len = past_key_values.get_seq_length() if past_key_values is not None else 0
|
||||
if cache_position is None:
|
||||
total = inputs_embeds.shape[1] if inputs_embeds is not None and past_len == 0 else input_ids.shape[1]
|
||||
count = max(total - past_len, 1) if past_len else total
|
||||
cache_position = torch.arange(past_len, past_len + count, device=input_ids.device)
|
||||
count = cache_position.numel()
|
||||
use_embeds = inputs_embeds is not None and past_len == 0
|
||||
if use_embeds:
|
||||
current_ids, current_embeds = None, inputs_embeds[:, -count:]
|
||||
else:
|
||||
current_ids, current_embeds = input_ids[:, -count:].contiguous(), None
|
||||
if position_ids is None and attention_mask is not None and attention_mask.ndim == 2:
|
||||
position_ids = attention_mask.long().cumsum(-1) - 1
|
||||
position_ids.masked_fill_(attention_mask == 0, 0)
|
||||
if position_ids is not None:
|
||||
position_ids = position_ids[:, -count:].contiguous()
|
||||
return {
|
||||
"input_ids": current_ids, "inputs_embeds": current_embeds,
|
||||
"past_key_values": past_key_values, "attention_mask": attention_mask,
|
||||
"position_ids": position_ids, "cache_position": cache_position,
|
||||
"use_cache": kwargs.get("use_cache", True),
|
||||
"logits_to_keep": kwargs.get("logits_to_keep", 1),
|
||||
}
|
||||
|
||||
# ── NAR velocity (flow matching, no KV cache) ────────────────────
|
||||
|
||||
def _shift_t_value(self, t_value: float, device: torch.device, dtype: torch.dtype) -> torch.Tensor:
|
||||
t_sig = torch.sigmoid(torch.tensor(t_value, dtype=dtype, device=device))
|
||||
shift = self.config.timestep_shift
|
||||
return shift * t_sig / (1 + (shift - 1) * t_sig)
|
||||
|
||||
@torch.no_grad()
|
||||
def nar_velocity(
|
||||
self,
|
||||
tokens: torch.LongTensor,
|
||||
ar_mask: torch.BoolTensor,
|
||||
nar_mask: torch.BoolTensor,
|
||||
nar_content_mask: torch.BoolTensor,
|
||||
x_t: torch.Tensor,
|
||||
t_value: float,
|
||||
nar_cond_end: int = 0,
|
||||
) -> torch.Tensor:
|
||||
"""Compute v_theta(x_t, t) — flow-matching velocity field.
|
||||
|
||||
Args:
|
||||
tokens: [1, S] full sequence (AR + NAR tokens)
|
||||
ar_mask: [1, S] True for AR positions
|
||||
nar_mask: [1, S] True for NAR positions
|
||||
nar_content_mask: [1, S] True for actual latent positions (not LATENT_START/END)
|
||||
x_t: [T_lat, D] current ODE state
|
||||
t_value: raw timestep (will be sigmoid-shifted)
|
||||
nar_cond_end: if > 0, NAR only sees positions < nar_cond_end (text-only mode)
|
||||
Returns:
|
||||
v_pred: [T_lat, D] predicted velocity
|
||||
"""
|
||||
device = tokens.device
|
||||
dtype = next(self.parameters()).dtype
|
||||
B, S = tokens.shape
|
||||
|
||||
# 1. Token embeddings
|
||||
token_emb = self.model.embed_tokens(tokens) # [B, S, H]
|
||||
|
||||
# 2. Build latent hidden for ALL NAR positions (START + content + END)
|
||||
# Training injects vae2llm(x_t) + time_emb + pos_emb at ALL NAR positions,
|
||||
# including LATENT_START (clean=0) and LATENT_END (clean=0).
|
||||
# NAR position IDs via cumsum: START=0, content=[1..T_lat], END=T_lat+1.
|
||||
t_shifted = self._shift_t_value(t_value, device, dtype)
|
||||
T_lat = x_t.shape[0]
|
||||
|
||||
nar_indices = nar_mask[0].nonzero(as_tuple=True)[0] # all NAR positions
|
||||
content_indices = nar_content_mask[0].nonzero(as_tuple=True)[0]
|
||||
N_nar = nar_indices.shape[0] # START + T_lat + END
|
||||
|
||||
# Build x_t for all NAR positions: zeros for START/END, actual x_t for content
|
||||
x_nar = torch.zeros(N_nar, x_t.shape[1], device=device, dtype=dtype)
|
||||
x_nar[1:1 + T_lat] = x_t.to(dtype) # content frames at positions [1, T_lat]
|
||||
|
||||
latent_hidden_nar = self.vae2llm(x_nar.unsqueeze(0)) # [1, N_nar, H]
|
||||
|
||||
# Timestep embedding (same t for all NAR positions)
|
||||
time_emb = self.time_embedder(t_shifted.expand(N_nar)).unsqueeze(0)
|
||||
latent_hidden_nar = latent_hidden_nar + time_emb
|
||||
|
||||
# Position embedding: cumsum-style [0, 1, 2, ..., N_nar-1]
|
||||
pos_ids = torch.arange(N_nar, device=device).clamp(max=self.config.max_latent_frames - 1)
|
||||
pos_emb = self.latent_pos_embed(pos_ids).unsqueeze(0)
|
||||
latent_hidden_nar = latent_hidden_nar + pos_emb
|
||||
|
||||
# Inject at ALL NAR positions (matching training's torch.where)
|
||||
token_emb[0, nar_indices] = latent_hidden_nar[0]
|
||||
|
||||
# 3. Build hybrid attention mask [B, 1, S, S]
|
||||
# AR→AR: causal, NAR→AR: full, NAR→NAR: bidirectional, AR→NAR: blocked
|
||||
ar_q = ar_mask.unsqueeze(2).float() # [B, S, 1]
|
||||
ar_k = ar_mask.unsqueeze(1).float() # [B, 1, S]
|
||||
nar_q = nar_mask.unsqueeze(2).float()
|
||||
nar_k = nar_mask.unsqueeze(1).float()
|
||||
causal = torch.tril(torch.ones(S, S, device=device))
|
||||
|
||||
if nar_cond_end > 0:
|
||||
# Codec dropout: NAR only sees positions < nar_cond_end (text) + NAR
|
||||
text_k = torch.zeros(1, 1, S, device=device)
|
||||
text_k[0, 0, :nar_cond_end] = 1.0
|
||||
mask = (ar_q * ar_k * causal) + (nar_q * text_k) + (nar_q * nar_k)
|
||||
else:
|
||||
mask = (ar_q * ar_k * causal) + (nar_q * ar_k) + (nar_q * nar_k)
|
||||
# Convert to additive: 0 → attend, -inf → block
|
||||
attn_mask = mask.unsqueeze(1) # [B, 1, S, S]
|
||||
attn_mask = attn_mask.masked_fill(attn_mask == 0, float("-inf")).masked_fill(attn_mask > 0, 0.0)
|
||||
|
||||
# 4. Position IDs + RoPE
|
||||
position_ids = torch.arange(S, device=device).unsqueeze(0)
|
||||
|
||||
# 5. Forward through decoder (with MoT routing)
|
||||
ar_mask_bt = ar_mask # [B, S] bool for MoT routing
|
||||
hidden_states, _ = self.model(
|
||||
inputs_embeds=token_emb, position_ids=position_ids,
|
||||
use_cache=False, attention_mask=attn_mask, ar_mask=ar_mask_bt,
|
||||
)
|
||||
|
||||
# 6. NAR head at content positions
|
||||
nar_pred = self.llm2vae(hidden_states) # [B, S, D]
|
||||
v_pred = nar_pred[0, content_indices] # [T_lat, D]
|
||||
return v_pred
|
||||
|
||||
# Keep custom code + auto_map when a local user calls save_pretrained as well
|
||||
# as when the release builder creates a Hub repository.
|
||||
YuE2Config.register_for_auto_class()
|
||||
YuE2ForCausalLM.register_for_auto_class("AutoModelForCausalLM")
|
||||
@@ -0,0 +1,9 @@
|
||||
{
|
||||
"schema": 1,
|
||||
"files": {
|
||||
"model.safetensors": {
|
||||
"bytes": 7261441640,
|
||||
"sha256": "1d55c42c1a9875c34f5d736e15078449992b044e807ce2a138e6cf289a1e59e9"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
{
|
||||
"abc": {
|
||||
"temperature": 0.7,
|
||||
"top_p": 0.9,
|
||||
"top_k": 30,
|
||||
"repetition_penalty": 1.005,
|
||||
"penalty_window": 100,
|
||||
"min_tokens": 32,
|
||||
"max_tokens": 4096
|
||||
},
|
||||
"semantic": {
|
||||
"temperature": 1.0,
|
||||
"top_p": 0.95,
|
||||
"top_k": 100,
|
||||
"repetition_penalty": 1.2,
|
||||
"penalty_window": 50,
|
||||
"min_tokens": 200,
|
||||
"max_tokens": 9000
|
||||
},
|
||||
"ode_steps": 32,
|
||||
"ode_method": "midpoint",
|
||||
"context": 24576,
|
||||
"version": "yue2-native-v1"
|
||||
}
|
||||
@@ -1,4 +1,22 @@
|
||||
from comfy_api.latest import ComfyExtension, io
|
||||
from typing_extensions import override
|
||||
|
||||
from .yue_node import YUE_SM_Model,YUE_SM_Vae, YUE_SM_Sampler,YUE_SM_Clip,YUE_SM_Cond
|
||||
|
||||
WEB_DIRECTORY = "./js"
|
||||
|
||||
class YUE_SM_Extension(ComfyExtension):
|
||||
@override
|
||||
async def get_node_list(self) -> list[type[io.ComfyNode]]:
|
||||
return [
|
||||
YUE_SM_Model,
|
||||
YUE_SM_Vae,
|
||||
YUE_SM_Sampler,
|
||||
YUE_SM_Clip,
|
||||
YUE_SM_Cond,
|
||||
]
|
||||
|
||||
async def comfy_entrypoint() -> YUE_SM_Extension: # ComfyUI calls this to load your extension and its nodes.
|
||||
return YUE_SM_Extension()
|
||||
|
||||
from .yue_node import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
|
||||
|
||||
__all__ = ['NODE_CLASS_MAPPINGS', 'NODE_DISPLAY_NAME_MAPPINGS']
|
||||
|
||||
|
After Width: | Height: | Size: 398 KiB |
|
After Width: | Height: | Size: 168 KiB |
@@ -0,0 +1,498 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:inkscape="http://www.inkscape.org/namespaces/inkscape" version="1.1" width="396" height="170.87078" viewBox="0 0 396 170.87078">
|
||||
<defs>
|
||||
<clipPath id="clip_1">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<path id="font_2_6" d="M.46191407 .33007813C.46191407 .27539063 .45760093 .22688802 .4489746 .18457031 .4403483 .14225261 .4272461 .10668945 .40966798 .07788086 .39208985 .049072267 .36979167 .02726237 .34277345 .012451172 .31575523-.0023600262 .28385417-.009765625 .24707031-.009765625 .17805989-.009765625 .12597656 .019205729 .09082031 .07714844 .055664064 .13509114 .038085939 .21940105 .038085939 .33007813 .038085939 .43847657 .055664064 .521403 .09082031 .5788574 .12597656 .6363118 .17936199 .66503909 .25097657 .66503909 .31998698 .66503909 .37239585 .6366374 .40820313 .579834 .4440104 .5230306 .46191407 .43977867 .46191407 .33007813M.3720703 .33032228C.3720703 .3771566 .3700358 .4189504 .3659668 .45570375 .3618978 .49245707 .35506187 .5234375 .34545899 .548645 .3358561 .57385256 .32299806 .593043 .30688478 .60621646 .29077149 .61938986 .27083335 .62597659 .24707031 .62597659 .22298177 .62597659 .20320638 .61938986 .18774414 .60621646 .1722819 .593043 .16015625 .57385256 .15136719 .548645 .14257813 .5234375 .13647461 .49245707 .13305664 .45570375 .12963867 .4189504 .12792969 .3771566 .12792969 .33032228 .12792969 .28348796 .12963867 .24145 .13305664 .20420838 .13647461 .16696675 .14257813 .13541667 .15136719 .109558109 .16015625 .08369955 .1722819 .06385803 .18774414 .05003357 .20320638 .036209108 .22298177 .029296875 .24707031 .029296875 .27083335 .029296875 .29077149 .036209108 .30688478 .05003357 .32299806 .06385803 .3358561 .08369955 .34545899 .109558109 .35506187 .13541667 .3618978 .16696675 .3659668 .20420838 .3700358 .24145 .3720703 .28348796 .3720703 .33032228Z"/>
|
||||
<path id="font_2_8" d="M.44482423 0H.043945314V.071777347C.076822917 .102376308 .10701498 .12980144 .13452149 .15405274 .162028 .17830403 .18676758 .20092774 .20874024 .22192383 .23071289 .24291992 .24991863 .26302085 .26635743 .28222657 .28279624 .30143229 .2964681 .3214518 .30737306 .34228517 .31827799 .3631185 .32641603 .38557945 .3317871 .40966798 .3371582 .4337565 .33984376 .4609375 .33984376 .49121095 .33984376 .5335286 .33024089 .5657552 .31103517 .5878906 .29182945 .61002609 .26041667 .62109377 .21679688 .62109377 .20703125 .62109377 .197347 .6203613 .18774414 .6188965 .17814128 .61744186 .16894531 .61549887 .16015625 .6130676 .15136719 .61064657 .14314778 .6078949 .13549805 .6048126 .12784831 .60173037 .12109375 .5985718 .115234378 .5953369L.09814453 .515625H.06591797V.6411896C.090657558 .6470286 .114990238 .6519725 .13891602 .6560211 .1628418 .66007998 .18880208 .6621094 .21679688 .6621094 .28841148 .6621094 .34220378 .6472168 .37817384 .61743167 .4141439 .5876465 .4321289 .54557296 .4321289 .49121095 .4321289 .46451823 .42862956 .43977867 .42163087 .4169922 .41463218 .39420573 .4046224 .37231446 .39160157 .35131837 .37858073 .33032228 .36263023 .3096517 .34375 .28930665 .32486979 .26896159 .3033854 .24788411 .27929688 .22607422 .25520835 .20426433 .22875977 .18107097 .19995117 .15649414 .17114258 .13191732 .14046224 .10481771 .107910159 .07519531H.44482423V0Z"/>
|
||||
<path id="font_2_11" d="M.2368164 .38378907C.27327476 .38378907 .3055013 .3801168 .3334961 .37277223 .36149089 .36542765 .38492839 .3540853 .4038086 .33874513 .4226888 .32341514 .43693034 .30391947 .4465332 .28025819 .45613609 .2565969 .4609375 .22828675 .4609375 .19532776 .4609375 .16433208 .45629884 .13627117 .44702149 .11114502 .43774415 .08601888 .4235026 .064478557 .40429688 .046524049 .38509117 .028579712 .36092124 .014709473 .3317871 .00491333 .30265299-.00487264 .26839195-.009765625 .2290039-.009765625 .19840496-.009765625 .16935222-.008051555 .1418457-.004623413 .114339198-.0011952719 .08821615 .0041097009 .06347656 .011291504L.05810547 .14941406H.09033203L.11230469 .057617189C.11816406 .05436198 .12524414 .05110677 .13354492 .047851564 .1418457 .044596357 .1508789 .041748048 .16064453 .03930664 .17041016 .036865236 .1805013 .03491211 .19091797 .033447267 .20133464 .031982423 .21142578 .03125 .2211914 .03125 .25146485 .03125 .2762858 .035079957 .2956543 .04273987 .3150228 .05039978 .33032228 .061238607 .34155274 .07525635 .3527832 .08928426 .3605143 .1060791 .3647461 .12564087 .36897788 .14520264 .37109376 .1668803 .37109376 .19067383 .37109376 .21610515 .36889649 .23827617 .36450196 .2571869 .36010743 .27609764 .35205079 .2919108 .34033204 .30462647 .32861329 .31734214 .3125814 .32687889 .29223634 .3332367 .27189128 .3395945 .24576824 .34277345 .21386719 .34277345 .19466146 .34277345 .17773438 .34147135 .16308594 .3388672 .1484375 .33626304 .13639324 .33365885 .12695313 .3310547H.080078128V.65527346H.41210938V.5800781H.12402344V.3720703C.12988281 .3733724 .13647461 .37467448 .14379883 .37597657 .15112305 .37727867 .15926107 .37849937 .16821289 .37963868 .17716472 .38077799 .18725586 .38175456 .19848633 .38256837 .2097168 .38338218 .22249349 .38378907 .2368164 .38378907Z"/>
|
||||
<path id="font_2_13" d="M.09814453 .50097659H.06591797V.65527346H.4711914V.61743167L.17919922 0H.11621094L.40283204 .5800781H.115234378L.09814453 .50097659Z"/>
|
||||
<path id="font_2_7" d="M.30615235 .0390625 .4399414 .025878907V0H.087890628V.025878907L.22216797 .0390625V.5727539L.08984375 .5253906V.5513611L.28076173 .66015627H.30615235V.0390625Z"/>
|
||||
<path id="font_2_27" d="M.067871097 .17675781H.099609378L.11669922 .088378909C.12288412 .080566409 .13151042 .073160808 .14257813 .06616211 .15364583 .05916341 .16593425 .052978517 .17944336 .047607423 .19295247 .042236329 .20719402 .03800456 .22216797 .03491211 .23714192 .03181966 .25179038 .030273438 .26611329 .030273438 .29085288 .030273438 .31233726 .03359477 .3305664 .040237428 .34879557 .04689026 .36385093 .056055707 .37573243 .067733768 .38761393 .07941183 .39648438 .093195598 .40234376 .10908508 .40820313 .12497457 .4111328 .1423238 .4111328 .16113281 .4111328 .18457031 .4061686 .2039388 .39624024 .21923828 .38631187 .23453777 .37329103 .24747722 .35717774 .25805665 .34106446 .26863609 .3227539 .2775879 .3022461 .2849121 .28173829 .29223634 .2606608 .29964195 .23901367 .3071289 .21736653 .31461589 .19628906 .32291667 .17578125 .33203126 .15527344 .34114585 .13696289 .3527018 .12084961 .36669923 .10473633 .38069663 .09171549 .3980306 .08178711 .41870118 .07185873 .43937174 .06689453 .46484376 .06689453 .4951172 .06689453 .5205078 .071777347 .5435384 .08154297 .564209 .091308597 .5848796 .10563151 .6024577 .12451172 .61694338 .14339192 .631429 .16658528 .6425781 .1940918 .6503906 .22159831 .6582031 .2529297 .6621094 .28808595 .6621094 .3203125 .6621094 .35091148 .65999349 .3798828 .6557617 .40885417 .65152999 .43554688 .64664718 .45996095 .6411133V.5048828H.42822267L.4111328 .58496096C.3968099 .5953776 .37923179 .6040039 .35839845 .61083987 .3375651 .6176758 .3141276 .62109377 .28808595 .62109377 .2639974 .62109377 .24324544 .61848959 .22583008 .61328127 .20841472 .60807296 .19401042 .60099288 .18261719 .592041 .17122396 .5830892 .16276042 .57250979 .15722656 .56030276 .1516927 .5480957 .14892578 .53499349 .14892578 .5209961 .14892578 .49983726 .15388997 .48225913 .16381836 .46826173 .17374675 .4542643 .18676758 .44230143 .20288086 .43237306 .21899414 .42244468 .23738607 .41389976 .25805665 .40673829 .2787272 .3995768 .29988609 .39217124 .3215332 .38452149 .34318034 .37687174 .3643392 .36824546 .38500978 .35864259 .40568034 .3490397 .42407228 .33683268 .44018556 .32202149 .45629884 .3072103 .46931968 .2891439 .47924806 .26782228 .48917643 .24650066 .49414063 .22021485 .49414063 .18896485 .49414063 .1586914 .48950196 .13134766 .4802246 .106933597 .47094728 .08251953 .45694987 .061604818 .43823243 .044189454 .41951499 .026774088 .39607749 .013427734 .36791993 .0041503908 .33976237-.005126953 .30664063-.009765625 .2685547-.009765625 .24837239-.009765625 .22851563-.008783977 .20898438-.0068206789 .18945313-.0048675539 .1710612-.0022583008 .1538086 .0010070801 .13655599 .004272461 .12060547 .008026123 .10595703 .012268066 .091308597 .01651001 .07861328 .020751954 .067871097 .024993897V.17675781Z"/>
|
||||
<path id="font_2_45" d="M.46191407 .23217774C.46191407 .15429688 .4444987 .09450277 .40966798 .05279541 .37483726 .011088054 .32063804-.009765625 .24707031-.009765625 .17805989-.009765625 .12597656 .010925293 .09082031 .05230713 .055664064 .093688968 .038085939 .15364583 .038085939 .23217774 .038085939 .30973307 .055664064 .3690389 .09082031 .4100952 .12597656 .45115153 .17936199 .4716797 .25097657 .4716797 .32063804 .4716797 .37320964 .45155845 .4086914 .41131593 .4441732 .3710734 .46191407 .3113607 .46191407 .23217774M.37402345 .23242188C.37402345 .2639974 .37190757 .29223634 .36767579 .31713868 .363444 .34204103 .35636393 .3630371 .34643556 .38012696 .33650718 .3972168 .32340495 .41023765 .3071289 .41918946 .29085288 .42814128 .27083335 .4326172 .24707031 .4326172 .22298177 .4326172 .203125 .42814128 .1875 .41918946 .171875 .41023765 .1595052 .3972168 .15039063 .38012696 .14127605 .3630371 .13492839 .34204103 .13134766 .31713868 .12776692 .29223634 .12597656 .2639974 .12597656 .23242188 .12597656 .20052083 .12776692 .17203777 .13134766 .14697266 .13492839 .121907558 .14127605 .10066732 .15039063 .08325195 .1595052 .065836589 .171875 .052490236 .1875 .04321289 .203125 .033935548 .22298177 .029296875 .24707031 .029296875 .27083335 .029296875 .29085288 .033935548 .3071289 .04321289 .32340495 .052490236 .33650718 .065836589 .34643556 .08325195 .35636393 .10066732 .363444 .121907558 .36767579 .14697266 .37190757 .17203777 .37402345 .20052083 .37402345 .23242188Z"/>
|
||||
<path id="font_2_44" d="M.15820313 .42236329C.1673177 .42757163 .17814128 .43310548 .19067383 .43896485 .20320638 .44482423 .2163086 .4501953 .22998047 .45507813 .24365235 .45996095 .25732423 .46394859 .2709961 .46704103 .28466798 .47013346 .29736329 .4716797 .30908204 .4716797 .32666017 .4716797 .34277345 .46923319 .35742188 .4643402 .3720703 .4594574 .38468424 .4516398 .39526368 .44088746 .4058431 .4301351 .4141439 .41612245 .42016603 .3988495 .42618815 .38157655 .42919923 .36071778 .42919923 .3362732V.034179689L.48486329 .021972657V0H.28710938V.021972657L.34814454 .034179689V.32714845C.34814454 .35416667 .34155274 .3754069 .32836915 .39086915 .31518556 .4063314 .29475913 .4140625 .26708985 .4140625 .25797526 .4140625 .24837239 .41358949 .23828125 .41264344 .22819011 .41170756 .21826172 .41060893 .2084961 .40934754 .19873047 .40808616 .1895345 .4065908 .1809082 .40486146 .1722819 .4031423 .16503906 .401652 .15917969 .40039063V.034179689L.2211914 .021972657V0H.022949219V.021972657L.078125 .034179689V.4248047L.022949219 .43701173V.45898438H.1538086L.15820313 .42236329Z"/>
|
||||
<path id="font_2_38" d="M.4248047 .31445313C.4248047 .26171876 .40901695 .22184246 .3774414 .19482422 .34586589 .16780599 .30045573 .15429688 .24121094 .15429688 .22786458 .15429688 .21443685 .1550293 .20092774 .15649414 .18741863 .15795899 .17610677 .15966797 .16699219 .1616211L.13623047 .09729004C.13720703 .09172567 .14355469 .08648682 .15527344 .08157349 .16699219 .07667033 .18164063 .07421875 .19921875 .07421875H.33496095C.3844401 .07421875 .42114259 .06347656 .44506837 .041992189 .46899415 .020507813 .48095704-.009114583 .48095704-.046875 .48095704-.06803385 .4766439-.088785808 .46801759-.10913086 .45939128-.1294759 .4453939-.14754232 .4260254-.16333008 .4066569-.17911785 .38134767-.19181316 .35009767-.20141602 .31884767-.21101888 .2804362-.21582031 .23486328-.21582031 .20003255-.21582031 .17041016-.212972 .1459961-.20727539 .12158203-.20157878 .10172526-.19368489 .08642578-.18359375 .071126308-.17350261 .060058595-.16178386 .053222658-.1484375 .04638672-.13509114 .04296875-.12060547 .04296875-.10498047 .04296875-.09423828 .045003255-.08406576 .049072267-.07446289 .053141279-.06486002 .05883789-.055582685 .06616211-.04663086 .07348633-.037679037 .08219401-.028971354 .092285159-.020507813 .102376308-.0120442709 .11328125-.0035807293 .125 .0048828127 .119140628 .00684611 .11238607 .01003011 .10473633 .014434814 .097086589 .018839518 .08984375 .02447001 .08300781 .031326295 .076171878 .038182577 .07047526 .0460968 .06591797 .05506897 .061360677 .06405131 .05908203 .07409159 .05908203 .08518982L.13623047 .17236328C.08479818 .19645183 .05908203 .24381511 .05908203 .31445313 .05908203 .34016929 .063313808 .36287437 .071777347 .38256837 .08024088 .40226237 .09236654 .41870118 .1081543 .43188478 .123942058 .44506837 .14339192 .45499674 .1665039 .46166993 .18961589 .4683431 .21582031 .4716797 .24511719 .4716797 .25423179 .4716797 .26350913 .4711914 .27294923 .47021485 .2823893 .46923829 .29125978 .46818034 .29956056 .46704103 .30786134 .4659017 .31510417 .4645996 .32128907 .46313478 .32747398 .46166993 .33203126 .46044923 .33496095 .45947267L.4428711 .5136719 .45996095 .49267579 .39208985 .42236329C.40315757 .4099935 .41137696 .39444987 .41674806 .37573243 .42211915 .35701499 .4248047 .33658854 .4248047 .31445313M.40478517-.06201172C.40478517-.04345703 .39908854-.028971354 .3876953-.018554688 .3763021-.0081380209 .35904948-.0029296876 .3359375-.0029296876H.15820313C.15136719-.0087890629 .1451009-.01554362 .1394043-.02319336 .13370769-.0308431 .12882488-.03889974 .12475586-.04736328 .12068685-.055826826 .11751302-.06437174 .115234378-.07299805 .11295573-.08162435 .111816409-.09000651 .111816409-.09814453 .111816409-.10986328 .11368815-.120524089 .11743164-.13012696 .12117513-.13972982 .12768555-.14794922 .13696289-.15478516 .14624024-.1616211 .15885417-.16699219 .17480469-.17089844 .1907552-.17480469 .21077474-.17675781 .23486328-.17675781 .26416017-.17675781 .2894694-.1739095 .31079103-.16821289 .33211265-.16251628 .34977214-.15454102 .36376954-.14428711 .37776695-.1340332 .3881022-.121907558 .3947754-.107910159 .40144859-.09391276 .40478517-.07861328 .40478517-.06201172M.2421875 .19140625C.27766929 .19140625 .30281578 .20157878 .31762696 .22192383 .33243815 .24226888 .33984376 .27311198 .33984376 .31445313 .33984376 .33496095 .33813478 .3527832 .3347168 .36791993 .33129884 .38305665 .32576499 .3955078 .31811524 .40527345 .31046549 .41503907 .30045573 .42236329 .28808595 .4272461 .27571617 .4321289 .2607422 .4345703 .24316406 .4345703 .22526042 .4345703 .21004232 .4321289 .19750977 .4272461 .18497722 .42236329 .17472331 .41503907 .16674805 .40527345 .15877278 .3955078 .1529948 .38305665 .14941406 .36791993 .14583333 .3527832 .14404297 .33496095 .14404297 .31445313 .14404297 .2939453 .14583333 .2759603 .14941406 .26049806 .1529948 .24503581 .1586914 .23217774 .1665039 .22192383 .1743164 .21166992 .18440755 .20402019 .19677735 .19897461 .20914714 .19392903 .22428386 .19140625 .2421875 .19140625Z"/>
|
||||
<path id="font_2_3" d="M0 0Z"/>
|
||||
<path id="font_2_47" d="M.39697267 .4814453H.43115235V-.17773438L.4819336-.18962097V-.21289063H.28710938V-.18962097L.35009767-.17773438V-.050598146C.35009767-.0382487 .3504232-.025009156 .35107423-.010879517 .35172526 .0032399495 .35302735 .017120362 .35498048 .030761719 .34033204 .01936849 .32218424 .009765625 .3005371 .001953125 .27888999-.005859375 .25341798-.009765625 .2241211-.009765625 .15999349-.009765625 .11263021 .010681152 .08203125 .051574708 .051432294 .09246826 .036132814 .15136719 .036132814 .22827149 .036132814 .26574708 .040364583 .29955546 .048828126 .32969667 .057291669 .35983787 .07006836 .38541667 .0871582 .4064331 .10424805 .42744956 .12589519 .4435781 .15209961 .45481874 .17830403 .46605937 .20898438 .4716797 .24414063 .4716797 .2607422 .4716797 .2787272 .47070313 .2980957 .46875 .3174642 .46679688 .33577476 .46402995 .35302735 .46044923L.39697267 .4814453M.12402344 .22825623C.12402344 .1950124 .12687175 .16657512 .13256836 .14294434 .13826497 .11931356 .14648438 .09991964 .15722656 .08476257 .16796875 .069615688 .18123372 .0585378 .19702149 .05152893 .21280925 .04452006 .23079427 .041015626 .25097657 .041015626 .25878907 .041015626 .2672526 .041341146 .2763672 .041992189 .28548179 .04264323 .29451499 .043701173 .3034668 .045166017 .3124186 .04663086 .32096354 .048339845 .32910157 .05029297 .3372396 .052246095 .34423829 .05452474 .35009767 .057128908V.42332459C.33642579 .42593894 .32063804 .42797853 .30273438 .42944337 .28483073 .4309082 .26757813 .43164063 .25097657 .43164063 .20800781 .43164063 .17610677 .41419984 .15527344 .37931825 .13444011 .3444468 .12402344 .2940928 .12402344 .22825623Z"/>
|
||||
<path id="font_2_51" d="M.15283203 .13085938C.15283203 .10384115 .15893555 .083089198 .17114258 .068603519 .18334961 .05411784 .20328777 .046875 .23095703 .046875 .2491862 .046875 .2680664 .048095704 .28759767 .05053711 .3071289 .052978517 .32600913 .056803388 .34423829 .06201172V.4248047L.27490235 .43701173V.45898438H.4248047V.034179689L.48291017 .021972657V0H.3491211L.34521485 .037109376C.33577476 .031901044 .32454429 .026529948 .31152345 .020996094 .2985026 .015462239 .2849121 .010416667 .27075196 .005859375 .2565918 .0013020834 .24235027-.0024414063 .22802735-.0053710939 .21370442-.008300781 .2006836-.009765625 .18896485-.009765625 .17138672-.009765625 .1554362-.0073242189 .14111328-.0024414063 .12679036 .0024414063 .11450195 .010253906 .10424805 .020996094 .09399414 .03173828 .08601888 .045654298 .080322269 .06274414 .07462565 .079833988 .071777347 .10058594 .071777347 .125V.4248047L.013183594 .43701173V.45898438H.15283203V.13085938Z"/>
|
||||
<path id="font_2_32" d="M.22705078 .46972657C.24788411 .46972657 .2672526 .46776835 .28515626 .46385194 .3030599 .45994569 .31852214 .45326744 .33154298 .44381715 .3445638 .43436686 .35473634 .42157493 .36206056 .40544129 .36938478 .38931785 .37304688 .3690338 .37304688 .34458924V.034179689L.43017579 .021972657V0H.30419923L.29492188 .045898439C.29003907 .041015626 .28344728 .03531901 .27514649 .028808594 .2668457 .022298178 .25683595 .016194663 .24511719 .010498047 .23339844 .004801432 .21980794 0 .2043457-.00390625 .18888347-.0078125 .17171224-.009765625 .15283203-.009765625 .13069661-.009765625 .11206055-.00634257 .09692383 .00050354006 .08178711 .00734965 .06966146 .016886393 .060546876 .02911377 .051432294 .04135132 .044921876 .055862428 .041015626 .072647098 .037109376 .08944193 .03515625 .10762533 .03515625 .12719727 .03515625 .14741008 .037597658 .16493225 .04248047 .1797638 .04736328 .19460552 .05419922 .20708211 .06298828 .2171936 .071777347 .2273051 .08211263 .23553975 .09399414 .24189759 .10587565 .24825542 .11873373 .2532247 .13256836 .25680543 .146403 .26039634 .16105144 .2628428 .17651367 .2641449 .1919759 .26545716 .20751953 .26627604 .22314453 .26660157L.2919922 .2685547V.34033204C.2919922 .3540039 .29085288 .36645509 .28857423 .37768556 .28629557 .38891603 .2824707 .39860026 .2770996 .40673829 .27172853 .4148763 .2644857 .42122398 .2553711 .42578126 .24625652 .43033854 .23486328 .4326172 .2211914 .4326172 .2055664 .4326172 .18977864 .4305013 .17382813 .42626954 .15787761 .42203776 .1438802 .4165039 .13183594 .40966798L.115234378 .35253907H.087890628V.45263673C.10904948 .457194 .13094075 .46118165 .15356446 .4645996 .17618816 .46801759 .2006836 .46972657 .22705078 .46972657M.2919922 .234375 .22802735 .23242188C.20882161 .23177083 .19222005 .229894 .17822266 .22679138 .16422527 .22368877 .15266927 .21838379 .14355469 .21087647 .13444011 .20336914 .12760417 .1930898 .123046878 .18003845 .118489589 .1669871 .11621094 .15033977 .11621094 .13009644 .11621094 .07266235 .13948567 .043945314 .18603516 .043945314 .20817058 .043945314 .22729492 .046468099 .2434082 .051513673 .25952149 .056559247 .27571617 .06298828 .2919922 .07080078V.234375Z"/>
|
||||
<path id="font_2_42" d="M.17919922 .034179689 .2578125 .021972657V0H.020019532V.021972657L.09814453 .034179689V.66015627L.020019532 .67204287V.69433596H.17919922V.034179689Z"/>
|
||||
<path id="font_2_40" d="M.1850586 .6086426C.1850586 .6014506 .18367513 .5946655 .1809082 .58828738 .17814128 .5819092 .1743164 .5762685 .1694336 .57136538 .16455078 .5664622 .15885417 .562617 .15234375 .5598297 .14583333 .5570526 .13899739 .55566409 .13183594 .55566409 .12467448 .55566409 .11791992 .5570526 .111572269 .5598297 .10522461 .562617 .099609378 .5664622 .09472656 .57136538 .08984375 .5762685 .08601888 .5819092 .08325195 .58828738 .08048502 .5946655 .07910156 .6014506 .07910156 .6086426 .07910156 .61583456 .08048502 .622701 .08325195 .62924197 .08601888 .6357829 .08984375 .64150497 .09472656 .6464081 .099609378 .6513112 .10522461 .65515139 .111572269 .65792849 .11791992 .66071578 .12467448 .6621094 .13183594 .6621094 .13899739 .6621094 .14583333 .66071578 .15234375 .65792849 .15885417 .65515139 .16455078 .6513112 .1694336 .6464081 .1743164 .64150497 .17814128 .6357829 .1809082 .62924197 .18367513 .622701 .1850586 .61583456 .1850586 .6086426M.18017578 .034179689 .25878907 .021972657V0H.020996094V.021972657L.099121097 .034179689V.4248047L.034179689 .43701173V.45898438H.18017578V.034179689Z"/>
|
||||
<path id="font_2_50" d="M.16308594-.009765625C.13183594-.009765625 .10847982-.00048828127 .09301758 .018066407 .077555339 .036621095 .06982422 .06266276 .06982422 .096191409V.41796876H.009765625V.4399414L.07080078 .45898438 .12011719 .56347659H.1508789V.45898438H.25585938V.41796876H.1508789V.10498047C.1508789 .08382162 .15568035 .067871097 .1652832 .057128908 .17488607 .04638672 .1875 .041015626 .203125 .041015626 .21516927 .041015626 .22713216 .041829427 .23901367 .04345703 .25089518 .045084638 .2618815 .046875 .27197267 .048828126V.017089844C.26708985 .013834636 .2606608 .010579427 .25268556 .0073242189 .24471028 .0040690104 .23592122 .0012207031 .22631836-.0012207031 .2167155-.0036621094 .20654297-.0056966149 .19580078-.0073242189 .1850586-.008951823 .17415364-.009765625 .16308594-.009765625Z"/>
|
||||
<path id="font_2_54" d="M.4482422 .4267578 .26904298-.028808594C.2586263-.05517578 .24804688-.0797526 .23730469-.10253906 .2265625-.12532552 .21459961-.1451009 .20141602-.16186524 .18823242-.17862956 .17317708-.19181316 .15625-.20141602 .13932292-.21101888 .119628909-.21582031 .09716797-.21582031 .08870443-.21582031 .08129883-.21565755 .07495117-.21533203 .068603519-.21500652 .06266276-.21443685 .057128908-.21362305 .05159505-.21280925 .0460612-.21191406 .040527345-.2109375 .03499349-.20996094 .028808594-.20865886 .021972657-.20703125V-.107910159H.044921876L.061035158-.15478516C.071126308-.16227214 .085123699-.16601563 .103027347-.16601563 .11702474-.16601563 .12972005-.16292317 .14111328-.15673828 .15250652-.15056356 .16300456-.14186096 .17260742-.1306305 .18221028-.119410198 .19091797-.106155399 .19873047-.09086609 .20654297-.07557678 .21402996-.05865987 .2211914-.040115358L.23388672-.004989624 .05908203 .4248047 .012207031 .43701173V.45898438H.22509766V.43701173L.15283203 .42382813 .27685548 .103027347 .39697267 .4248047 .3251953 .43701173V.45898438H.49609376V.43701173L.4482422 .4267578Z"/>
|
||||
<path id="font_2_35" d="M.35302735 .034179689C.33870445 .022786459 .32088218 .012613933 .29956056 .0036621094 .27823893-.0052897136 .25309245-.009765625 .2241211-.009765625 .09879557-.009765625 .036132814 .068603519 .036132814 .2253418 .036132814 .26345826 .040283204 .29774986 .048583986 .32821656 .056884767 .35869346 .06966146 .38460288 .08691406 .40594483 .104166667 .42728678 .12597656 .4435781 .15234375 .45481874 .17871094 .46605937 .20996094 .4716797 .24609375 .4716797 .2626953 .4716797 .28035484 .47070313 .29907228 .46875 .3177897 .46679688 .33577476 .46402995 .35302735 .46044923 .3523763 .46402995 .35188804 .46930949 .3515625 .47628785 .35123698 .48327638 .35099284 .49074809 .35083009 .498703 .35066734 .5066579 .35050456 .51453146 .3503418 .5223236 .35017906 .5301158 .35009767 .5364482 .35009767 .5413208V.66015627L.27294923 .67204287V.69433596H.43115235V.034179689L.48779298 .021972657V0H.35888673L.35302735 .034179689M.12402344 .22532654C.12402344 .19110616 .1270345 .16226197 .13305664 .13879395 .13907878 .11532593 .1476237 .096338909 .1586914 .081832889 .16975911 .067337039 .18310547 .0569102 .19873047 .05055237 .21435547 .04419454 .23177083 .041015626 .25097657 .041015626 .2705078 .041015626 .28889976 .04288737 .30615235 .04663086 .32340495 .050374349 .33805339 .05485026 .35009767 .060058595V.42332459C.33642579 .42593894 .32063804 .42797853 .30273438 .42944337 .28483073 .4309082 .26757813 .43164063 .25097657 .43164063 .2076823 .43164063 .17569988 .41419984 .1550293 .37931825 .13435872 .3444468 .12402344 .29311625 .12402344 .22532654Z"/>
|
||||
<path id="font_2_36" d="M.12695313 .23144531V.22264099C.12695313 .19881694 .12866211 .17596944 .13208008 .15409851 .13549805 .13222759 .14233399 .11288961 .15258789 .096084598 .1628418 .07927958 .1772461 .06589762 .19580078 .05593872 .21435547 .04598999 .23876953 .041015626 .26904298 .041015626 .2788086 .041015626 .2890625 .041422529 .2998047 .042236329 .31054688 .04305013 .32128907 .044108076 .33203126 .045410158 .34277345 .04671224 .3531901 .048177083 .36328126 .049804689 .3733724 .051432294 .38264976 .053222658 .39111329 .05517578V.027832032C.3836263 .022949219 .3745931 .018310547 .36401368 .013916016 .35343424 .009521484 .34179688 .005533854 .32910157 .001953125 .31640626-.0016276041 .30289714-.0044759118 .28857423-.006591797 .2742513-.008707683 .25976563-.009765625 .24511719-.009765625 .20703125-.009765625 .17488607-.0045522057 .14868164 .005874634 .12247721 .016301474 .10123698 .031778974 .08496094 .05230713 .0686849 .07283529 .056966146 .09825134 .049804689 .1285553 .04264323 .15885926 .0390625 .19372559 .0390625 .2331543 .0390625 .3133138 .055826826 .3731079 .08935547 .41253663 .12288412 .45196534 .17073567 .4716797 .23291016 .4716797 .25732423 .4716797 .28019206 .46842448 .30151368 .46191407 .3228353 .45540367 .34147135 .4444987 .35742188 .42919923 .3733724 .41389976 .38598634 .39339195 .39526368 .36767579 .40454103 .34195964 .4091797 .30989585 .4091797 .27148438V.23144531H.12695313M.23291016 .4326172C.21468099 .4326172 .19897461 .42879234 .18579102 .42114259 .17260742 .41349284 .16170247 .40266929 .15307617 .38867188 .14444988 .37467448 .13810222 .35766603 .1340332 .33764649 .12996419 .31762696 .12792969 .2952474 .12792969 .2705078H.32421876C.32421876 .2952474 .3228353 .31762696 .32006837 .33764649 .31730143 .35766603 .3124186 .37467448 .30541993 .38867188 .29842124 .40266929 .2890625 .41349284 .27734376 .42114259 .265625 .42879234 .2508138 .4326172 .23291016 .4326172Z"/>
|
||||
<path id="font_2_53" d="M.43408204 .032714845 .48779298 .02230835V0H.27978517V.02229309L.3408203 .033691408 .23486328 .195755 .110839847 .032714845 .17382813 .02230835V0H.0087890629V.022338868L.06201172 .030273438 .21289063 .22898865 .080078128 .4248047 .025878907 .43701173V.45898438H.23388672V.43701173L.17285156 .42382813 .26123048 .2922058 .36279298 .4248047 .2998047 .43701173V.45898438H.46484376V.43701173L.41210938 .4267578 .28320313 .25976563 .43408204 .032714845Z"/>
|
||||
<path id="font_2_55" d="M.2290039 .45361329C.2179362 .4441732 .20442708 .4346517 .18847656 .42504884 .17252605 .41544596 .15397136 .40559898 .1328125 .3955078V.43066407C.15494792 .44954429 .1751302 .4695638 .19335938 .49072267 .21158855 .51188156 .22753906 .5358073 .24121094 .5625H.25878907C.27246095 .5358073 .28841148 .51188156 .30664063 .49072267 .32486979 .4695638 .3450521 .44954429 .3671875 .43066407V.3955078C.34602867 .40559898 .32747398 .41544596 .31152345 .42504884 .2955729 .4346517 .2820638 .4441732 .2709961 .45361329V-.029296875H.2290039V.45361329Z"/>
|
||||
<path id="font_2_28" d="M.1538086 0V.025878907L.2578125 .0390625V.61328127H.23291016C.19026692 .61328127 .15445964 .61230978 .12548828 .6103668 .09651693 .6084239 .07600912 .6061554 .063964847 .6035614L.05078125 .5019531H.018066407V.65527346H.5942383V.5019531H.56103518L.54785159 .6035614C.5419922 .60485336 .5332845 .6059825 .5217285 .60694888 .51017257 .6079254 .49674479 .6088155 .4814453 .60961917 .46614585 .6104329 .4494629 .611084 .43139649 .61157229 .41333009 .61206057 .39485679 .6123047 .37597657 .6123047H.35205079V.0390625L.4560547 .025878907V0H.1538086Z"/>
|
||||
<path id="font_2_43" d="M.15917969 .421875C.16829427 .4271342 .17911785 .4327189 .19165039 .43862916 .20418294 .44454957 .21712239 .4499766 .23046875 .45491029 .24381511 .45984397 .25732423 .46387229 .2709961 .46699525 .28466798 .4701182 .29736329 .4716797 .30908204 .4716797 .33154298 .4716797 .35229493 .46740724 .3713379 .4588623 .39038087 .45032756 .4046224 .43669639 .4140625 .41796876 .42447917 .423879 .43701173 .43003846 .45166017 .43644715 .4663086 .44285584 .4815267 .4486847 .49731446 .45393373 .51310226 .4591929 .52872726 .46346537 .54418948 .4667511 .5596517 .47003684 .5735677 .4716797 .5859375 .4716797 .6035156 .4716797 .6194661 .46923319 .63378909 .4643402 .648112 .4594574 .6604004 .4516398 .6706543 .44088746 .6809082 .4301351 .6888835 .41612245 .6945801 .3988495 .7002767 .38157655 .703125 .36071778 .703125 .3362732V.034179689L.76220706 .021972657V0H.55371096V.021972657L.6220703 .034179689V.32714845C.6220703 .35416667 .6159668 .3749186 .60375979 .3894043 .59155276 .40388999 .57161459 .4111328 .5439453 .4111328 .53548178 .4111328 .52563479 .41048179 .5144043 .4091797 .5031738 .4078776 .49194337 .40641276 .4807129 .40478517 .46948243 .40315757 .45874024 .4012858 .44848634 .39916993 .43823243 .39705406 .4296875 .39534507 .42285157 .39404298 .4283854 .37646485 .43115235 .35709635 .43115235 .3359375V.034179689L.5 .021972657V0H.28222657V.021972657L.35009767 .034179689V.32714845C.35009767 .35416667 .34318034 .3749186 .3293457 .3894043 .31551109 .40388999 .29475913 .4111328 .26708985 .4111328 .25797526 .4111328 .24845378 .41064454 .23852539 .40966798 .228597 .4086914 .21883138 .4075521 .20922852 .40625 .19962566 .4049479 .19051107 .4034017 .18188477 .40161134 .17325847 .39982096 .16601563 .39827476 .16015625 .39697267V.034179689L.2290039 .021972657V0H.020996094V.021972657L.07910156 .034179689V.4248047L.020996094 .43701173V.45898438H.15527344L.15917969 .421875Z"/>
|
||||
<clipPath id="clip_3">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_4">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_5">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_6">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_7">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_8">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_9">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_10">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_11">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_12">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_13">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_14">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_15">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_16">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_17">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_18">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<clipPath id="clip_19">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
|
||||
</clipPath>
|
||||
<path id="font_2_17" d="M.46777345 .49635316C.46777345 .51559957 .46492515 .5323944 .45922853 .5467377 .4535319 .5610911 .44433595 .5730794 .43164063 .58270266 .4189453 .59232589 .4025065 .59950259 .38232423 .6042328 .36214195 .608963 .33740235 .6113281 .30810548 .6113281H.20703125V.36328126H.31396485C.34423829 .36328126 .36930339 .36646018 .38916017 .372818 .40901695 .3791758 .4247233 .3881429 .4362793 .39971925 .4478353 .41130577 .4559733 .4252523 .46069337 .44155885 .46541343 .4578654 .46777345 .47613017 .46777345 .49635316M.51708987 .18652344C.51708987 .20930989 .51342776 .2290039 .5061035 .24560547 .4987793 .26220704 .48738609 .2759603 .47192384 .28686524 .45646159 .29777018 .43652345 .3059082 .41210938 .3112793 .3876953 .3166504 .35839845 .31933595 .32421876 .31933595H.20703125V.043945314C.2220052 .04329427 .2376302 .04280599 .25390626 .04248047 .26790367 .041829427 .28336589 .041422529 .30029298 .041259767 .31722007 .041097005 .33398438 .041015626 .35058595 .041015626 .3811849 .041015626 .4070638 .044433595 .42822267 .05126953 .4493815 .05810547 .46655274 .06778971 .47973634 .080322269 .49291993 .09285482 .5024414 .108072917 .5083008 .12597656 .51416018 .1438802 .51708987 .1640625 .51708987 .18652344M.028808594 0V.025878907L.11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.328125C.37402345 .65527346 .41227214 .651769 .4428711 .64476016 .47347007 .6377513 .49804688 .6275635 .51660159 .6141968 .53515627 .6008301 .54833987 .58461 .55615237 .5655365 .56396487 .546463 .5678711 .5250244 .5678711 .5012207 .5678711 .48068238 .5645345 .46193443 .5578613 .4449768 .5511882 .4280192 .5419108 .4131012 .5300293 .40022279 .51814779 .38734437 .50390627 .3765869 .4873047 .36795045 .47070313 .35931397 .45263673 .35287477 .43310548 .3486328 .46142579 .34570313 .48706056 .3400065 .51000979 .33154298 .532959 .32307945 .5524089 .3120931 .5683594 .29858399 .5843099 .28507487 .5965983 .26904298 .6052246 .25048829 .6138509 .2319336 .61816409 .21126302 .61816409 .18847656 .61816409 .16015625 .6133626 .13419597 .60375979 .1105957 .5941569 .086995448 .57910159 .06681315 .55859377 .050048829 .53808596 .033284505 .51171877 .020263672 .4794922 .010986328 .44726563 .0017089844 .40836589-.0029296876 .36279298-.0029296876 .32633464-.0029296876 .29003907-.0024414063 .25390626-.0014648438 .21777344-.00048828127 .18440755 0 .1538086 0H.028808594Z"/>
|
||||
<path id="font_2_31" d="M.4091797 .25801087V.0390625L.5131836 .025878907V0H.2109375V.025878907L.3149414 .0390625V.25506593L.08496094 .61621096 .011230469 .6290741V.65527346H.28808595V.6290741L.20019531 .61621096 .3881836 .31419374 .56689456 .61621096 .48388673 .6290741V.65527346H.69677737V.6290741L.625 .61621096 .4091797 .25801087Z"/>
|
||||
<path id="font_2_20" d="M.028808594 .025878907 .11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.52001956V.49902345H.48779298L.47216798 .6045227C.4617513 .60581466 .44913737 .60694888 .43432618 .6079254 .41951499 .608902 .40445964 .60962936 .38916017 .6101074 .3738607 .6105957 .359375 .6109212 .34570313 .611084 .33203126 .61124679 .3214518 .6113281 .31396485 .6113281H.20703125V.35546876H.38378907L.39892579 .43359376H.43017579V.23242188H.39892579L.38378907 .31152345H.20703125V.043945314H.3359375C.35611979 .043945314 .37516276 .044189454 .3930664 .044677736 .41097007 .045166017 .42708335 .045735677 .44140626 .04638672 .45572917 .04703776 .46801759 .047851564 .47827149 .048828126 .4885254 .049804689 .49609376 .05078125 .50097659 .051757814L.5288086 .17285156H.56103518L.5517578 0H.028808594V.025878907Z"/>
|
||||
<path id="font_2_19" d="M.5800781 .3322754C.5800781 .38049317 .57340499 .42211405 .5600586 .45713807 .5467122 .49216209 .5275879 .52107748 .50268557 .5438843 .4777832 .5666911 .4477539 .5836334 .41259767 .5947113 .3774414 .6057892 .33821617 .6113281 .29492188 .6113281H.20703125V.045898439C.21679688 .045247396 .2273763 .044759115 .23876953 .044433595 .25016276 .044108076 .2618815 .043701173 .27392579 .04321289 .28597007 .04272461 .2980957 .04239909 .31030274 .042236329 .32250978 .042073568 .33447267 .041992189 .3461914 .041992189 .38753257 .041992189 .42293296 .04810079 .45239259 .060317995 .4818522 .072535198 .50602218 .090779628 .52490237 .11505127 .54378256 .13932292 .55769857 .1695404 .5666504 .20570374 .57560226 .24186707 .5800781 .28405763 .5800781 .3322754M.32617188 .65527346C.38053385 .65527346 .4296061 .6495717 .47338868 .63816836 .5171712 .62676498 .5545247 .608195 .5854492 .5824585 .6163737 .5567322 .6402181 .52334597 .6569824 .4822998 .67374679 .44125367 .6821289 .39092 .6821289 .33129884 .6821289 .2778727 .67529299 .23047383 .6616211 .18910218 .6479492 .14773052 .6272786 .11287435 .5996094 .08453369 .5719401 .056193036 .537028 .0346934 .49487306 .02003479 .4527181 .00537618 .40315757-.001953125 .3461914-.001953125 .32763673-.001953125 .30737306-.0018717448 .2854004-.0017089844 .26342774-.001546224 .24186199-.0013020834 .22070313-.0009765625 .19954427-.0006510417 .17944336-.00040690104 .16040039-.00024414063 .14135742-.00008138021 .12548828 0 .11279297 0H.028808594V.025878907L.11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.32617188Z"/>
|
||||
<path id="font_2_37" d="M.10986328 .41796876H.030761719V.44189454L.10986328 .4609375V.49316407C.10986328 .52766928 .11336263 .5580241 .12036133 .5842285 .12736003 .6104329 .13745117 .6324056 .15063477 .6501465 .16381836 .6678874 .17993164 .6813151 .19897461 .6904297 .21801758 .69954428 .23942058 .70410159 .2631836 .70410159 .27783204 .70410159 .29085288 .70320639 .3022461 .701416 .3136393 .6996257 .32389323 .6974284 .3330078 .6948242V.59472659H.30908204L.28710938 .65478518C.28190104 .65804037 .27620445 .6605632 .27001954 .6623535 .26383464 .66414389 .2561849 .66503909 .24707031 .66503909 .23567708 .66503909 .22639974 .6625163 .21923828 .6574707 .21207683 .6524251 .2063802 .6446126 .20214844 .6340332 .19791667 .6234538 .19498699 .61002609 .19335938 .59375 .19173177 .57747396 .19091797 .5579427 .19091797 .53515627V.45898438H.31298829V.41796876H.19091797V.038085939L.29003907 .021972657V0H.041992189V.021972657L.10986328 .038085939V.41796876Z"/>
|
||||
<path id="font_2_26" d="M.20703125 .28710938V.0390625L.30615235 .025878907V0H.03515625V.025878907L.11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.31152345C.35742188 .65527346 .39575196 .65119937 .42651368 .64305117 .4572754 .63490298 .4819336 .6232503 .5004883 .60809329 .51904299 .5929362 .53222659 .5746002 .54003909 .5530853 .54785159 .53157046 .5517578 .5072835 .5517578 .4802246 .5517578 .455129 .5480957 .4327189 .5407715 .41299439 .53344729 .39326988 .5236003 .37591044 .51123049 .36091615 .4988607 .34592185 .48453776 .3334554 .46826173 .32351686 .4519857 .31357829 .4350586 .30599977 .41748048 .30078126L.59472659 .0390625 .66552737 .025878907V0H.50878909L.32470704 .28710938H.20703125M.45458985 .47338868C.45458985 .4994812 .4514974 .5213318 .4453125 .5389404 .4391276 .5565491 .42944337 .57072958 .41625978 .58148196 .40307618 .59224447 .38614909 .5999095 .36547853 .6044769 .34480796 .6090444 .31982423 .6113281 .29052735 .6113281H.20703125V.3310547H.29345704C.32340495 .3310547 .3486328 .33374534 .36914063 .3391266 .38964845 .34450785 .40625 .35298667 .4189453 .364563 .43164063 .3761393 .44075523 .39089457 .44628907 .40882875 .4518229 .4267629 .45458985 .44828288 .45458985 .47338868Z"/>
|
||||
<path id="font_2_39" d="M.15917969 .49560548C.15917969 .4910482 .15909831 .4855143 .15893555 .4790039 .15877278 .4724935 .15861003 .46573893 .15844727 .45874024 .1582845 .45174156 .15795899 .44498698 .1574707 .43847657 .15698242 .43196617 .15641277 .42659507 .15576172 .42236329 .1648763 .42757163 .17594402 .43310548 .18896485 .43896485 .20198567 .44482423 .21557617 .4501953 .22973633 .45507813 .24389649 .45996095 .25805665 .46394859 .2722168 .46704103 .28637696 .47013346 .2993164 .4716797 .31103517 .4716797 .32861329 .4716797 .34472657 .46923319 .359375 .4643402 .37402345 .4594574 .38663737 .4516398 .3972168 .44088746 .40779624 .4301351 .41609703 .41612245 .42211915 .3988495 .42814128 .38157655 .43115235 .36071778 .43115235 .3362732V.034179689L.4868164 .021972657V0H.2890625V.021972657L.35009767 .034179689V.33007813C.35009767 .35709635 .34350587 .3778483 .33032228 .39233399 .31713868 .40681968 .29671226 .4140625 .26904298 .4140625 .25992839 .4140625 .25024415 .41358949 .23999024 .41264344 .22973633 .41170756 .2195638 .41060893 .20947266 .40934754 .19938152 .40808616 .1899414 .4065908 .18115235 .40486146 .17236328 .4031423 .16503906 .401652 .15917969 .40039063V.034179689L.2211914 .021972657V0H.020019532V.021972657L.078125 .034179689V.66015627L.009765625 .67204287V.69433596H.15917969V.49560548Z"/>
|
||||
<path id="font_2_23" d="M.42089845 0H.4038086L.1640625 .5629883V.0390625L.25195313 .025878907V0H.028808594V.025878907L.11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.22705078L.4399414 .15686035 .6723633 .65527346H.8598633V.6290741L.7758789 .61621096V.0390625L.8598633 .025878907V0H.5942383V.025878907L.6821289 .0390625V.5629883L.42089845 0Z"/>
|
||||
<path id="font_2_49" d="M.35302735 .12904358C.35302735 .10851542 .34985353 .08969625 .34350587 .07258606 .3371582 .055486043 .32714845 .040822347 .31347657 .02859497 .2998047 .016377768 .28214518 .0069274904 .26049806 .00024414063 .2388509-.0064290368 .21272786-.009765625 .1821289-.009765625 .16682942-.009765625 .1517741-.008870442 .13696289-.007080078 .122151698-.0052897136 .10839844-.0031738282 .095703128-.0007324219 .08300781 .0017089844 .0719401 .004231771 .0625 .0068359377 .053059896 .0094401049 .046223958 .011555989 .041992189 .013183594V.12597656H.063964847L.087890628 .062332155C.09798177 .053268434 .1110026 .045496625 .12695313 .039016725 .14290364 .032536825 .1616211 .029296875 .18310547 .029296875 .2133789 .029296875 .23673503 .035858156 .25317384 .048980714 .26961265 .06210327 .27783204 .08243815 .27783204 .10998535 .27783204 .12628174 .27441407 .13972473 .26757813 .15031433 .2607422 .16090393 .25179038 .16977947 .24072266 .17694092 .22965496 .18411255 .21704102 .1900584 .20288086 .19477844 .1887207 .19950867 .17423503 .20431519 .15942383 .209198 .14461263 .21409099 .13012696 .21962993 .1159668 .22581482 .10180664 .23200989 .08919271 .23999532 .078125 .24977112 .06705729 .2595469 .05810547 .27176414 .05126953 .28642274 .044433595 .30109153 .041015626 .31934104 .041015626 .34117127 .041015626 .36202494 .044759115 .38059999 .052246095 .39689637 .059733076 .41319276 .07023112 .42687989 .083740238 .43795777 .09724935 .44903565 .11336263 .45742289 .13208008 .4631195 .15079753 .4688263 .17138672 .4716797 .19384766 .4716797 .21598308 .4716797 .2376302 .47013346 .25878907 .46704103 .2799479 .46394859 .30029298 .46044923 .31982423 .45654298V.3564453H.296875L.2763672 .40966798C.26790367 .41715495 .25634767 .42285157 .24169922 .4267578 .22705078 .43066407 .21142578 .4326172 .19482422 .4326172 .16845703 .4326172 .14835613 .4260966 .13452149 .41305543 .12068685 .40001426 .11376953 .38241069 .11376953 .36024476 .11376953 .34525047 .1171875 .33294679 .12402344 .32333375 .13085938 .31373088 .13989258 .30558778 .15112305 .29890443 .16235352 .29222108 .1751302 .2864329 .18945313 .28153993 .20377605 .2766571 .21842449 .27160646 .23339844 .26638795 .24837239 .2611796 .26302085 .25523377 .27734376 .24855042 .29166667 .24187725 .30444337 .23332723 .31567384 .22290039 .3269043 .21247356 .3359375 .19976299 .34277345 .18476868 .34960938 .16978455 .35302735 .15120952 .35302735 .12904358Z"/>
|
||||
<path id="font_2_22" d="M.30810548 .6290741 .20703125 .61621096V.041992189H.3359375C.3766276 .041992189 .40983073 .04313151 .43554688 .045410158 .46126304 .047698976 .4790039 .04982503 .48876954 .05178833L.51904299 .18847656H.55078127L.5419922 0H.028808594V.025878907L.11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.30810548V.6290741Z"/>
|
||||
<path id="font_2_29" d="M.7109375 .65527346V.6290741L.63916018 .61621096 .37597657-.015625H.35107423L.08496094 .61621096 .011230469 .6290741V.65527346H.2758789V.6290741L.18798828 .61621096 .38623048 .13400269 .5839844 .61621096 .49804688 .6290741V.65527346H.7109375Z"/>
|
||||
<path id="font_2_34" d="M.41308595 .027832032C.4046224 .021647135 .39453126 .016194663 .3828125 .011474609 .37109376 .006754557 .3585612 .0028483074 .34521485-.00024414063 .3318685-.0033365887 .31795249-.0056966149 .3034668-.0073242189 .2889811-.008951823 .27490235-.009765625 .26123048-.009765625 .22151692-.009765625 .18758138-.004308065 .15942383 .0066070558 .13126628 .017522177 .10823568 .033406576 .09033203 .054260255 .07242838 .07511393 .059244794 .10061137 .05078125 .13075257 .042317708 .16089376 .038085939 .19502767 .038085939 .2331543 .038085939 .27486167 .04353841 .31078593 .05444336 .34092713 .065348308 .37106834 .080566409 .39583335 .100097659 .41522218 .119628909 .434611 .14282227 .4488678 .16967774 .45799256 .1965332 .4671173 .22591146 .4716797 .2578125 .4716797 .2841797 .4716797 .30973307 .47013346 .33447267 .46704103 .35921226 .46394859 .3816732 .46044923 .40185548 .45654298V.32861329H.375L.3540039 .40966798C.34195964 .4165039 .32739259 .42203776 .31030274 .42626954 .2932129 .4305013 .27539063 .4326172 .25683595 .4326172 .23567708 .4326172 .21704102 .42878724 .20092774 .42112733 .18481446 .4134674 .17114258 .40148927 .15991211 .38519288 .14868164 .36889649 .1402181 .34820048 .13452149 .32310487 .12882488 .29800926 .12597656 .26802574 .12597656 .2331543 .12597656 .20381673 .12841797 .17733257 .13330078 .15370178 .1381836 .130071 .14680989 .109944667 .15917969 .093322757 .17154949 .076700847 .18823242 .063827518 .20922852 .05470276 .23022461 .045578004 .25683595 .041015626 .2890625 .041015626 .30013023 .041015626 .31144206 .041422529 .32299806 .042236329 .33455406 .04305013 .34578453 .044189454 .35668946 .045654298 .3675944 .04711914 .3778483 .048828126 .38745118 .05078125 .39705406 .052734376 .40559898 .05485026 .41308595 .057128908V.027832032Z"/>
|
||||
<path id="font_2_9" d="M.4609375 .17848206C.4609375 .14881897 .45572917 .1223348 .4453125 .09902954 .43489585 .07572428 .41984049 .055999757 .40014649 .039855958 .38045249 .02372233 .35620118 .01141866 .32739259 .0029449464 .29858399-.005528768 .26578776-.009765625 .2290039-.009765625 .19677735-.009765625 .16552735-.0078074138 .1352539-.0038909913 .10498047 .000025431315 .07763672 .005086263 .053222658 .011291504L.047851564 .14941406H.080078128L.10205078 .057617189C.107910159 .05436198 .1155599 .05110677 .125 .047851564 .13444011 .044596357 .14461263 .041748048 .15551758 .03930664 .16642253 .036865236 .177653 .03491211 .18920899 .033447267 .20076497 .031982423 .21142578 .03125 .2211914 .03125 .25146485 .03125 .2762858 .03499349 .2956543 .04248047 .3150228 .04996745 .33032228 .060465497 .34155274 .07397461 .3527832 .08748373 .3605143 .10359701 .3647461 .12231445 .36897788 .1410319 .37109376 .16145833 .37109376 .18359375 .37109376 .20898438 .36751304 .23006185 .36035157 .24682617 .3531901 .26359049 .3433431 .2770996 .33081056 .28735353 .31827799 .29760743 .30362956 .3050944 .28686524 .30981446 .27010093 .31453453 .25211589 .3173828 .23291016 .31835938L.16308594 .32226563V.3623047L.23291016 .36668397C.269694 .36863709 .296875 .38000999 .31445313 .4008026 .33203126 .42159526 .3408203 .45311485 .3408203 .49536134 .3408203 .5164795 .33862306 .5349172 .33422853 .55067446 .32983399 .5664317 .32291667 .5795085 .31347657 .5899048 .30403648 .6003011 .29174806 .6080983 .27661134 .6132965 .2614746 .6184947 .2430013 .62109377 .2211914 .62109377 .21142578 .62109377 .20157878 .6203613 .19165039 .6188965 .181722 .61744186 .17220052 .61549887 .16308594 .6130676 .15397136 .61064657 .14550781 .6078949 .13769531 .6048126 .12988281 .60173037 .123046878 .5985718 .1171875 .5953369L.100097659 .515625H.067871097V.6411896C.07926432 .6441091 .090657558 .64686587 .10205078 .64945986 .11344401 .65205386 .12532552 .6542409 .13769531 .6560211 .15006511 .65781149 .16308594 .65927127 .17675781 .6604004 .19042969 .66153976 .20524089 .6621094 .2211914 .6621094 .2898763 .6621094 .34204103 .6489461 .37768556 .6226196 .41333009 .59629318 .43115235 .55582687 .43115235 .5012207 .43115235 .4807434 .4283854 .46164958 .42285157 .4439392 .41731773 .42622886 .4086914 .41054789 .39697267 .39689637 .3852539 .38324485 .37036134 .3717855 .35229493 .3625183 .33422853 .35325114 .31282554 .34683229 .28808595 .34326173 .34733073 .33641563 .39095054 .3193817 .4189453 .29216004 .4469401 .26494853 .4609375 .22705586 .4609375 .17848206Z"/>
|
||||
<path id="font_2_5" d="M.18408203 .044433595C.18408203 .036295576 .18253581 .028645834 .17944336 .021484375 .1763509 .014322917 .17220052 .008056641 .16699219 .0026855469 .16178386-.0026855469 .15551758-.006917318 .14819336-.010009766 .14086914-.013102214 .13313802-.0146484379 .125-.0146484379 .11653646-.0146484379 .10872396-.013102214 .1015625-.010009766 .09440104-.006917318 .08821615-.0026855469 .08300781 .0026855469 .07779948 .008056641 .073649089 .014322917 .07055664 .021484375 .067464198 .028645834 .06591797 .036295576 .06591797 .044433595 .06591797 .052571615 .067464198 .060302736 .07055664 .06762695 .073649089 .07495117 .07779948 .081217449 .08300781 .08642578 .08821615 .09163412 .09440104 .09578451 .1015625 .09887695 .10872396 .1019694 .11653646 .103515628 .125 .103515628 .13313802 .103515628 .14086914 .1019694 .14819336 .09887695 .15551758 .09578451 .16178386 .09163412 .16699219 .08642578 .17220052 .081217449 .1763509 .07495117 .17944336 .06762695 .18253581 .060302736 .18408203 .052571615 .18408203 .044433595Z"/>
|
||||
<path id="font_2_12" d="M.47021485 .2033844C.47021485 .16948955 .4659017 .13934326 .4572754 .11294556 .44864909 .08654785 .4358724 .064219158 .4189453 .045959474 .40201823 .027709961 .38110353 .013860066 .35620118 .00440979 .33129884-.005040487 .30257163-.009765625 .27001954-.009765625 .23421224-.009765625 .20222982-.002766927 .17407227 .011230469 .14591472 .025227866 .122151698 .046142579 .1027832 .07397461 .08341471 .10180664 .068603519 .13647461 .05834961 .17797852 .048095704 .21948242 .04296875 .26790367 .04296875 .3232422 .04296875 .38248698 .04972331 .433431 .06323242 .47607423 .07674154 .51871749 .094807948 .5538737 .11743164 .58154299 .14005535 .6092122 .16617839 .6295573 .19580078 .6425781 .22542317 .65559896 .2561849 .6621094 .28808595 .6621094 .3125 .6621094 .3372396 .66048178 .3623047 .65722659 .38736979 .6539714 .4099935 .64990237 .43017579 .64501956V.53222659H.39794923L.38085938 .5991211C.375 .6023763 .36824546 .60530599 .3605957 .60791018 .35294596 .61051437 .3449707 .61279299 .33666993 .6147461 .32836915 .6166992 .32006837 .6182454 .31176759 .61938479 .3034668 .6205241 .2955729 .62109377 .28808595 .62109377 .26497398 .62109377 .244222 .61557009 .22583008 .6045227 .20743816 .59347537 .19156902 .5767415 .17822266 .5543213 .1648763 .53190109 .15437825 .5037944 .14672852 .47000123 .13907878 .4362081 .13460286 .39640299 .13330078 .35058595 .15673828 .36295573 .18237305 .37304688 .21020508 .38085938 .23803711 .38867188 .265625 .39257813 .29296876 .39257813 .3203125 .39257813 .3449707 .38874818 .36694337 .38108827 .38891603 .37342835 .4075521 .36177574 .42285157 .34613038 .43815104 .33048503 .44986979 .3107656 .4580078 .28697206 .46614585 .2631887 .47021485 .23532613 .47021485 .2033844M.2680664 .029296875C.28889976 .029296875 .30647788 .032714845 .32080079 .03955078 .3351237 .04638672 .3466797 .056722005 .35546876 .07055664 .3642578 .08439127 .37052409 .10164388 .37426759 .12231445 .37801109 .14298503 .3798828 .16699219 .3798828 .19433594 .3798828 .24772136 .37150065 .28629557 .35473634 .3100586 .33797203 .33382163 .3113607 .34570313 .27490235 .34570313 .25276695 .34570313 .22957357 .34358726 .20532227 .33935548 .18107097 .3351237 .15690105 .32910157 .1328125 .32128907 .1328125 .27441407 .13533528 .23291016 .14038086 .19677735 .14542644 .16064453 .15348308 .13012696 .16455078 .10522461 .17561849 .080322269 .18961589 .06144206 .20654297 .048583986 .22347005 .03572591 .24397786 .029296875 .2680664 .029296875Z"/>
|
||||
<path id="font_2_16" d="M.22509766 .025878907V0H.009765625V.025878907L.083984378 .0390625 .3071289 .66015627H.39990235L.63183596 .0390625 .71484377 .025878907V0H.43798829V.025878907L.5258789 .0390625 .4609375 .22851563H.203125L.13720703 .0390625 .22509766 .025878907M.33007813 .58984377 .21777344 .27246095H.44580079L.33007813 .58984377Z"/>
|
||||
<path id="font_2_18" d="M.3779297-.009765625C.32454429-.009765625 .27693687-.0022786458 .23510742 .0126953129 .193278 .027669272 .15795899 .049316408 .12915039 .07763672 .1003418 .10595703 .07845052 .14054363 .06347656 .18139649 .048502607 .22224935 .041015626 .26839195 .041015626 .31982423 .041015626 .37841798 .048583986 .42919923 .0637207 .47216798 .07885742 .5151367 .10083008 .5506999 .12963867 .5788574 .15844727 .60701498 .19384766 .6279297 .23583985 .64160159 .27783204 .65527346 .32584635 .6621094 .3798828 .6621094 .40234376 .6621094 .4235026 .66137698 .44335938 .6599121 .46321617 .65844729 .48217774 .6565755 .50024417 .6542969 .51831057 .65201827 .53548178 .64941409 .5517578 .6464844 .5680339 .6435547 .5838216 .6404622 .5991211 .63720706L.6020508 .49414063H.5698242L.5551758 .57910159C.53238937 .59309896 .50594076 .60392257 .47583009 .61157229 .4457194 .619222 .41503907 .6230469 .38378907 .6230469 .34570313 .6230469 .31176759 .6178436 .28198243 .60743716 .25219728 .59703066 .22705078 .58003237 .20654297 .55644229 .18603516 .53286239 .17032878 .50180056 .15942383 .46325685 .14851888 .4247233 .1430664 .37731935 .1430664 .32104493 .1430664 .27160646 .1484375 .22859192 .15917969 .19200135 .16992188 .15541077 .18538411 .125 .2055664 .10076904 .2257487 .076538089 .2504069 .05840556 .27954103 .04637146 .30867515 .03433736 .34179688 .028320313 .37890626 .028320313 .39908854 .028320313 .41837565 .02961731 .43676759 .032211305 .45515953 .034805299 .47224937 .03837077 .4880371 .042907716 .5038249 .047454835 .51798507 .05272929 .5305176 .05873108 .5430501 .06473287 .55354818 .0709788 .5620117 .07746887L.5800781 .17480469H.6118164L.6088867 .020996094C.59521487 .017089844 .57958987 .013264974 .5620117 .009521484 .5444336 .0057779948 .5257161 .0024414063 .5058594-.00048828127 .4860026-.0034179688 .46516929-.0056966149 .44335938-.0073242189 .42154948-.008951823 .3997396-.009765625 .3779297-.009765625Z"/>
|
||||
<path id="font_2_4" d="M.037109376 .19824219V.2734375H.296875V.19824219H.037109376Z"/>
|
||||
<path id="font_2_46" d="M.07421875 .4248047 .021972657 .43701173V.45898438H.1508789L.15185547 .4326172C.1586914 .43847657 .16674805 .44376628 .17602539 .44848634 .18530274 .4532064 .1953125 .4572754 .20605469 .46069337 .21679688 .46411134 .22819011 .46679688 .24023438 .46875 .25227867 .47070313 .2644857 .4716797 .27685548 .4716797 .3055013 .4716797 .33121745 .46662904 .3540039 .4565277 .37679038 .4464264 .39615885 .4313558 .41210938 .41131593 .4280599 .39127604 .44018556 .36651103 .44848634 .33702088 .4567871 .30753074 .4609375 .27355958 .4609375 .23510742 .4609375 .19764202 .45670573 .1638387 .4482422 .13369751 .43977867 .10355631 .42708335 .077809657 .41015626 .05645752 .39322917 .03511556 .37198893 .01874288 .34643556 .0073394777 .32088218-.0040639245 .29101563-.009765625 .25683595-.009765625 .24023438-.009765625 .22273763-.008870442 .2043457-.007080078 .18595378-.0052897136 .16845703-.0026041668 .15185547 .0009765625 .15218099-.0029296876 .15258789-.0074005129 .15307617-.012435913 .15356446-.017471314 .15388997-.022664389 .15405274-.028015137 .1542155-.033376058 .15437825-.038324994 .15454102-.04286194 .15470378-.047409059 .15478516-.05114746 .15478516-.05407715V-.17773438L.23486328-.18962097V-.21289063H.016113282V-.18962097L.07421875-.17773438V.4248047M.37304688 .23509217C.37304688 .2682546 .37027995 .2965393 .3647461 .3199463 .35921226 .34335328 .3511556 .36245219 .34057618 .37724305 .32999674 .39204408 .31705729 .40285746 .3017578 .40968324 .28645835 .416509 .26920573 .41992188 .25 .41992188 .234375 .41992188 .21769206 .41853843 .19995117 .41577149 .18221028 .41300456 .16715496 .40901695 .15478516 .4038086V.037582399C.16845703 .034988405 .18359375 .032958986 .20019531 .03149414 .21679688 .030029297 .23339844 .029296875 .25 .029296875 .29296876 .029296875 .32421876 .047093709 .34375 .08268738 .36328126 .11829122 .37304688 .16909282 .37304688 .23509217Z"/>
|
||||
<path id="font_2_21" d="M.028808594 0V.025878907L.11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.29101563V.6290741L.20703125 .61621096V.359375H.5151367V.61621096L.43115235 .6290741V.65527346H.6928711V.6290741L.6088867 .61621096V.0390625L.6928711 .025878907V0H.43115235V.025878907L.5151367 .0390625V.3154297H.20703125V.0390625L.29101563 .025878907V0H.028808594Z"/>
|
||||
<path id="font_2_48" d="M.32421876 .4716797V.34765626H.30322267L.27490235 .4013672C.26578776 .4013672 .25602214 .40071617 .24560547 .39941407 .2351888 .39811198 .22477214 .39639793 .21435547 .39427186 .2039388 .39215599 .19392903 .38962809 .18432617 .38668824 .17472331 .38375855 .16634114 .3806661 .15917969 .3774109V.034179689L.23779297 .021972657V0H.020019532V.021972657L.078125 .034179689V.4248047L.020019532 .43701173V.45898438H.1538086L.15820313 .40234376C.16569011 .40852867 .17594402 .41560874 .18896485 .42358399 .20198567 .43155924 .21606446 .4391276 .23120117 .44628907 .24633789 .45345054 .2614746 .45947267 .27661134 .46435548 .29174806 .46923829 .30517579 .4716797 .31689454 .4716797H.32421876Z"/>
|
||||
<path id="font_2_52" d="M.4560547 .4267578 .27197267-.009765625H.23583985L.046875 .4248047 0 .43701173V.45898438H.21386719V.43701173L.14111328 .42382813 .27490235 .106933597 .40283204 .4248047 .33007813 .43701173V.45898438H.5V.43701173L.4560547 .4267578Z"/>
|
||||
<path id="font_2_30" d="M.8847656 .61621096 .67089846-.015625H.64501956L.47509767 .43652345 .30078126-.015625H.27490235L.05810547 .61621096 .0009765625 .6290741V.65527346H.25097657V.6290741L.15478516 .61621096 .31054688 .15429688 .4868164 .609375H.50878909L.67871096 .15429688 .82714846 .61621096 .72509768 .6290741V.65527346H.94189456V.6290741L.8847656 .61621096Z"/>
|
||||
<path id="font_2_10" d="M.3955078 .14453125V0H.31152345V.14453125H.01953125V.20996094L.33935548 .6582031H.3955078V.21484375H.484375V.14453125H.3955078M.31152345 .54345706H.30908204L.07470703 .21484375H.31152345V.54345706Z"/>
|
||||
<path id="font_20_10" d="M.44482423 .25801087V.048828126L.5488281 .03564453V0H.18652344V.03564453L.29052735 .048828126V.25506593L.091308597 .6064453 .017578125 .6192627V.65527346H.3544922V.6192627L.26660157 .6064453 .41601563 .33200074 .5605469 .6064453 .48046876 .6192627V.65527346H.703125V.6192627L.63378909 .6064453 .44482423 .25801087Z"/>
|
||||
<path id="font_20_12" d="M.34423829 .03955078C.33414714 .03434245 .32307945 .028645834 .31103517 .022460938 .29899089 .016276041 .28629557 .010579427 .27294923 .0053710939 .25960288 .00016276042 .24576824-.0041503908 .23144531-.0075683596 .21712239-.010986328 .20263672-.0126953129 .18798828-.0126953129 .1694336-.0126953129 .15234375-.01024882 .13671875-.005355835 .12109375-.00047302247 .107666019 .007344564 .09643555 .018096924 .08520508 .028859457 .0764974 .042872114 .0703125 .060134889 .0641276 .07740784 .061035158 .0982666 .061035158 .12271118V.41503907L.015136719 .4267578V.45898438H.20214844V.14170838C.20214844 .11433411 .20792644 .09298706 .21948242 .07766724 .2310384 .062347413 .2475586 .0546875 .26904298 .0546875 .29378257 .0546875 .31852214 .06022644 .34326173 .07130432V.41503907L.30126954 .4267578V.45898438H.484375V.043945314L.5292969 .032226564V0H.35107423L.34423829 .03955078Z"/>
|
||||
<path id="font_20_9" d="M.017089844 .03564453 .10107422 .048828126V.6064453L.017089844 .6192627V.65527346H.57470706V.4892578H.53027346L.51464846 .5947571C.50423178 .596049 .49161784 .5971832 .47680665 .5981598 .46199546 .59913638 .4469401 .5998637 .43164063 .6003418 .41634117 .6008301 .40185548 .6011556 .3881836 .60131838 .37451173 .60148116 .36393229 .6015625 .3564453 .6015625H.2548828V.36132813H.42626954L.44140626 .43359376H.48486329V.23242188H.44140626L.42626954 .30664063H.2548828V.053710939H.37841798C.39860026 .053710939 .41764323 .053955079 .43554688 .05444336 .45345054 .05493164 .4695638 .0555013 .48388673 .056152345 .49820964 .056803388 .51049807 .057617189 .52075198 .05859375 .53100588 .059570314 .5385742 .060546876 .54345706 .061523439L.57128909 .18261719H.61572268L.6064453 0H.017089844V.03564453Z"/>
|
||||
<path id="font_20_6" d="M.45703126 0H.041992189V.092285159C.06998698 .12223307 .09586588 .14900716 .119628909 .17260742 .14339192 .19620769 .16495769 .21818035 .18432617 .23852539 .20369466 .25887046 .22086589 .27823893 .23583985 .29663087 .2508138 .3150228 .26326499 .33406578 .27319337 .35375978 .28312174 .37345378 .2906901 .39453126 .29589845 .4169922 .30110679 .43945313 .30371095 .4650065 .30371095 .49365235 .30371095 .5135091 .30110679 .5308431 .29589845 .5456543 .2906901 .5604655 .2837728 .57283529 .27514649 .5827637 .26652018 .5926921 .2565104 .60009768 .24511719 .60498049 .23372396 .6098633 .22167969 .6123047 .20898438 .6123047 .18880208 .6123047 .1726888 .6101888 .16064453 .60595706 .14860027 .6017253 .13753255 .5953776 .1274414 .58691409L.10644531 .4921875H.063964847V.6411133C.09000651 .64697268 .115478519 .6519368 .14038086 .65600588 .1652832 .6600749 .19238281 .6621094 .22167969 .6621094 .29361979 .6621094 .3487142 .64729818 .3869629 .6176758 .42521159 .5880534 .44433595 .54589846 .44433595 .49121095 .44433595 .46516929 .4412435 .4416504 .4350586 .4206543 .4288737 .3996582 .41984049 .3798828 .40795899 .36132813 .39607749 .34277345 .38134767 .32470704 .36376954 .3071289 .3461914 .28955079 .32592774 .2709961 .30297853 .25146485 .2800293 .2319336 .25455729 .21077474 .2265625 .18798828 .1985677 .16520183 .16829427 .13932292 .13574219 .11035156H.45703126V0Z"/>
|
||||
<path id="font_20_3" d="M0 0Z"/>
|
||||
<path id="font_20_4" d="M.1772461 .24145508C.1772461 .20172119 .17862956 .16239421 .18139649 .12347412 .1841634 .08455404 .1899414 .047749837 .19873047 .013061523 .20751953-.021626791 .2199707-.05354309 .23608399-.08268738 .25219728-.111841838 .27376304-.13667806 .30078126-.15719605V-.21289063C.25716148-.18976848 .21931966-.16460674 .18725586-.1374054 .15519206-.11021423 .12849935-.07878622 .107177738-.043121339 .08585612-.0074564616 .06998698 .03357951 .059570314 .07998657 .049153646 .12640381 .043945314 .1803894 .043945314 .24194336 .043945314 .30317179 .049153646 .35682679 .059570314 .40290834 .06998698 .44900004 .08585612 .48979698 .107177738 .5252991 .12849935 .5608012 .15519206 .59206649 .18725586 .61909487 .21931966 .6461334 .25716148 .6712138 .30078126 .69433596V.63864138C.27376304 .61812338 .25219728 .59336856 .23608399 .56437686 .2199707 .5353953 .20751953 .50356039 .19873047 .46887208 .1899414 .43418376 .1841634 .39754234 .18139649 .35894776 .17862956 .32035319 .1772461 .28118897 .1772461 .24145508Z"/>
|
||||
<path id="font_20_8" d="M.42871095 .49342347C.42871095 .51201888 .4268392 .52808126 .4230957 .5416107 .4193522 .55515035 .4132487 .56640627 .40478517 .5753784 .39632163 .5843506 .3853353 .59095767 .37182618 .5951996 .35831706 .5994415 .34195964 .6015625 .3227539 .6015625H.2548828V.37304688H.32666017C.34651695 .37304688 .3630371 .37581889 .3762207 .38136292 .3894043 .38690696 .39990235 .39481608 .40771485 .40509034 .41552735 .41537477 .42097984 .42793785 .42407228 .44277955 .4271647 .45762126 .42871095 .47450257 .42871095 .49342347M.47802735 .19091797C.47802735 .23421224 .46720378 .2664388 .44555665 .28759767 .42390953 .3087565 .38850913 .31933595 .33935548 .31933595H.2548828V.053710939C.2705078 .053059896 .28548179 .052571615 .2998047 .052246095 .31184898 .05159505 .324056 .051188154 .33642579 .05102539 .34879557 .05086263 .3585612 .05078125 .36572267 .05078125 .38623048 .05078125 .40364585 .054036458 .41796876 .060546876 .43229167 .06705729 .44384767 .07633463 .45263673 .088378909 .46142579 .10042318 .46785484 .11507162 .47192384 .13232422 .47599284 .14957683 .47802735 .16910808 .47802735 .19091797M.016601563 0V.03564453L.10107422 .048828126V.6064453L.016601563 .6192627V.65527346H.33447267C.38297526 .65527346 .4235026 .6518504 .4560547 .6450043 .48860679 .63815817 .5147298 .62837728 .5344238 .6156616 .55411788 .602946 .56811526 .5874583 .576416 .5691986 .5847168 .5509389 .5888672 .5302327 .5888672 .5070801 .5888672 .48523966 .5852051 .46551515 .57788088 .4479065 .57055667 .43029786 .56070968 .41489158 .54833987 .40168763 .53597006 .38848368 .52164718 .37756349 .5053711 .368927 .48909507 .36029054 .47200523 .35401408 .45410157 .35009767 .48632813 .34684245 .5140788 .3408203 .5373535 .33203126 .56062826 .3232422 .579834 .31201173 .5949707 .29833985 .6101074 .28466798 .62125656 .26879884 .62841799 .25073243 .6355794 .23266602 .63916018 .21272786 .63916018 .19091797 .63916018 .16259766 .63411459 .13655599 .62402346 .11279297 .6139323 .089029949 .5979004 .068603519 .57592776 .051513673 .5539551 .034423829 .52579757 .021077475 .49145509 .011474609 .45711265 .0018717448 .4156901-.0029296876 .3671875-.0029296876 .328125-.0029296876 .28938804-.0024414063 .25097657-.0014648438 .21256511-.00048828127 .17708333 0 .14453125 0H.016601563Z"/>
|
||||
<path id="font_20_11" d="M.46191407 .23217774C.46191407 .19307454 .45792643 .15853374 .44995118 .1285553 .44197593 .098576869 .42936198 .07332357 .41210938 .05279541 .39485679 .032267255 .37263999 .016708374 .34545899 .0061187746 .31827799-.004470825 .28548179-.009765625 .24707031-.009765625 .2109375-.009765625 .1796875-.0045522057 .15332031 .005874634 .12695313 .016301474 .10522461 .031778974 .088134769 .05230713 .07104492 .07283529 .05843099 .09816996 .05029297 .12831116 .04215495 .15845235 .038085939 .19307454 .038085939 .23217774 .038085939 .2709554 .04215495 .30525209 .05029297 .33506776 .05843099 .36488343 .071126308 .38989259 .088378909 .4100952 .10563151 .43029786 .12768555 .44561259 .15454102 .45603944 .18139649 .46646629 .21354167 .4716797 .25097657 .4716797 .32356773 .4716797 .37687174 .45139567 .41088868 .41082765 .4449056 .3702596 .46191407 .31070964 .46191407 .23217774M.31884767 .2319336C.31884767 .26350913 .3178711 .29117839 .31591798 .3149414 .31396485 .33870445 .31038413 .3585612 .30517579 .37451173 .29996745 .39046226 .292806 .40234376 .2836914 .41015626 .2745768 .41796876 .2626953 .421875 .24804688 .421875 .23372396 .421875 .22216797 .41796876 .2133789 .41015626 .20458985 .40234376 .19783528 .39046226 .19311524 .37451173 .18839519 .3585612 .18522136 .33870445 .18359375 .3149414 .18196614 .29117839 .18115235 .26350913 .18115235 .2319336 .18115235 .20003255 .18196614 .17211914 .18359375 .14819336 .18522136 .12426758 .18839519 .104166667 .19311524 .087890628 .19783528 .071614589 .20458985 .05940755 .2133789 .05126953 .22216797 .04313151 .23372396 .0390625 .24804688 .0390625 .2626953 .0390625 .2745768 .04313151 .2836914 .05126953 .292806 .05940755 .29996745 .071614589 .30517579 .087890628 .31038413 .104166667 .31396485 .12426758 .31591798 .14819336 .3178711 .17211914 .31884767 .20003255 .31884767 .2319336Z"/>
|
||||
<path id="font_20_7" d="M.45166017 .49378968C.45166017 .45800273 .4428711 .4275055 .42529298 .40229798 .40771485 .37709046 .38297526 .35879518 .35107423 .3474121 .3683268 .3409017 .3841146 .33243308 .3984375 .32200624 .4127604 .31157939 .42496745 .2992808 .4350586 .28511048 .44514976 .27094016 .45296226 .25481669 .4584961 .23674011 .46402995 .21866353 .46679688 .19871013 .46679688 .17687989 .46679688 .114990238 .4489746 .06841024 .41333009 .037139894 .37768556 .0058695476 .32226563-.009765625 .24707031-.009765625 .17578125-.009765625 .12231445 .0056254069 .08666992 .03640747 .05102539 .06718954 .033203126 .11401367 .033203126 .17687989 .033203126 .1983846 .035888673 .21817525 .041259767 .23625183 .04663086 .2543284 .054280599 .27045188 .064208988 .2846222 .07413737 .2987925 .08610026 .3111725 .100097659 .32176209 .114095058 .33235169 .12988281 .3409017 .14746094 .3474121 .115885417 .3594462 .09147135 .37798564 .07421875 .4030304 .056966146 .42807517 .048339845 .45881654 .048339845 .49525453 .048339845 .5209503 .052652997 .5442861 .061279298 .56526187 .0699056 .5862376 .08268229 .604126 .099609378 .618927 .11653646 .633728 .13769531 .6451111 .16308594 .6530762 .18847656 .66105148 .21777344 .66503909 .25097657 .66503909 .28320313 .66503909 .31176759 .66105148 .33666993 .6530762 .36157228 .6451111 .38256837 .63364669 .3996582 .61868289 .41674806 .60371908 .4296875 .5857493 .43847657 .56477358 .44726563 .5437978 .45166017 .52013656 .45166017 .49378968M.328125 .17700196C.328125 .20007324 .32674156 .22046407 .3239746 .23817444 .32120768 .2558848 .3166504 .27083335 .31030274 .28302003 .30395509 .2952067 .2956543 .30446879 .2854004 .31080628 .27514649 .31714378 .26236979 .3203125 .24707031 .3203125 .23209636 .3203125 .21980794 .31714378 .21020508 .31080628 .20060222 .30446879 .19295247 .2952067 .18725586 .28302003 .18155925 .27083335 .17757161 .2558848 .17529297 .23817444 .17301433 .22046407 .171875 .20007324 .171875 .17700196 .171875 .15490723 .17301433 .1353302 .17529297 .118270877 .17757161 .10121155 .18155925 .08691406 .18725586 .07537842 .19295247 .06384277 .20060222 .05506897 .21020508 .049057008 .21980794 .043045045 .23209636 .040039064 .24707031 .040039064 .26236979 .040039064 .27514649 .043045045 .2854004 .049057008 .2956543 .05506897 .30395509 .06384277 .31030274 .07537842 .3166504 .08691406 .32120768 .10121155 .3239746 .118270877 .32674156 .1353302 .328125 .15490723 .328125 .17700196M.31298829 .49365235C.31298829 .51115927 .31201173 .5273692 .3100586 .5422821 .30810548 .557195 .3046061 .5700836 .29956056 .5809479 .29451499 .59181216 .2878418 .60024008 .27954103 .6062317 .27124024 .61223348 .2607422 .6152344 .24804688 .6152344 .23567708 .6152344 .22558594 .61223348 .21777344 .6062317 .20996094 .60024008 .20377605 .59181216 .19921875 .5809479 .19466146 .5700836 .19148763 .557195 .18969727 .5422821 .1879069 .5273692 .18701172 .51115927 .18701172 .49365235 .18701172 .47485353 .1879069 .45799256 .18969727 .44306947 .19148763 .42815653 .19466146 .41551209 .19921875 .4051361 .20377605 .3947703 .20996094 .38683067 .21777344 .38131715 .22558594 .37580363 .23567708 .37304688 .24804688 .37304688 .2607422 .37304688 .27124024 .37580363 .27954103 .38131715 .2878418 .38683067 .29451499 .3947703 .29956056 .4051361 .3046061 .41551209 .30810548 .42815653 .3100586 .44306947 .31201173 .45799256 .31298829 .47485353 .31298829 .49365235Z"/>
|
||||
<path id="font_20_5" d="M.032226564-.21289063V-.15719605C.05891927-.13667806 .08040365-.111841838 .09667969-.08268738 .11295573-.05354309 .12548828-.021626791 .13427735 .013061523 .1430664 .047749837 .1488444 .08455404 .15161133 .12347412 .15437825 .16239421 .15576172 .20172119 .15576172 .24145508 .15576172 .28118897 .15437825 .32035319 .15161133 .35894776 .1488444 .39754234 .1430664 .43418376 .13427735 .46887208 .12548828 .50356039 .11295573 .5353953 .09667969 .56437686 .08040365 .59336856 .05891927 .61812338 .032226564 .63864138V.69433596C.075520839 .6712138 .11328125 .6461334 .14550781 .61909487 .17773438 .59206649 .20450847 .5608012 .22583008 .5252991 .24715169 .48979698 .26302085 .44900004 .2734375 .40290834 .28385417 .35682679 .2890625 .30317179 .2890625 .24194336 .2890625 .1803894 .28385417 .12640381 .2734375 .07998657 .26302085 .03357951 .24715169-.0074564616 .22583008-.043121339 .20450847-.07878622 .17773438-.11021423 .14550781-.1374054 .11328125-.16460674 .075520839-.18976848 .032226564-.21289063Z"/>
|
||||
<path id="font_2_41" d="M.16796875 .2211914 .35595704 .42382813 .30810548 .43701173V.45898438H.47021485V.43701173L.41308595 .42578126 .28222657 .29203797 .4501953 .033203126 .5 .021972657V0H.31201173V.021972657L.3540039 .034179689 .22802735 .2318573 .16796875 .16601563V.034179689L.21679688 .021972657V0H.028808594V.021972657L.08691406 .034179689V.66015627L.019042969 .67204287V.69433596H.16796875V.2211914Z"/>
|
||||
<path id="font_2_15" d="M.032226564 .45507813C.032226564 .48860679 .037109376 .51831057 .046875 .54418948 .056640626 .57006838 .07063802 .5917155 .08886719 .60913088 .10709635 .6265462 .12923177 .6397298 .15527344 .64868167 .18131511 .6576335 .21061199 .6621094 .24316406 .6621094 .28027345 .6621094 .31233726 .65559896 .33935548 .6425781 .3663737 .6295573 .38875328 .60945639 .40649415 .5822754 .42423503 .5550944 .4374186 .5205078 .44604493 .47851563 .45467124 .43652345 .45898438 .38671876 .45898438 .32910157 .45898438 .28710938 .45581056 .2495931 .4494629 .21655274 .44311524 .18351238 .43424479 .15445964 .42285157 .12939453 .41145835 .10432943 .39786784 .08292643 .38208009 .06518555 .36629234 .04744466 .34895835 .033040365 .33007813 .021972657 .3111979 .010904948 .29109703 .0028483074 .2697754-.0021972657 .24845378-.0072428386 .2265625-.009765625 .20410156-.009765625 .17545574-.009765625 .14949544-.008382161 .1262207-.0056152346 .10294596-.0028483074 .08024088 .0013020834 .05810547 .0068359377V.12011719H.08984375L.106933597 .050186159C.11311849 .047276815 .120035808 .04468791 .12768555 .042419435 .13533528 .04015096 .14331055 .03812663 .15161133 .036346437 .15991211 .034566244 .16837566 .033269247 .17700196 .032455446 .18562825 .031651815 .19401042 .03125 .20214844 .03125 .22493489 .03125 .24617513 .036051435 .26586915 .045654298 .28556315 .05525716 .30273438 .07080078 .3173828 .092285159 .33203126 .11376953 .3438314 .14168294 .3527832 .17602539 .36173503 .21036785 .36702476 .25227867 .36865235 .3017578 .34716798 .28957115 .32389323 .27952577 .29882813 .2716217 .27376304 .26371766 .2467448 .25976563 .21777344 .25976563 .19042969 .25976563 .16536458 .26391603 .14257813 .2722168 .119791667 .28051759 .10017904 .29288737 .083740238 .30932618 .06730143 .32576499 .05460612 .3461914 .045654298 .37060548 .036702474 .39501954 .032226564 .4231771 .032226564 .45507813M.24414063 .6230469C.20214844 .6230469 .17130535 .6085002 .15161133 .57940676 .13191732 .5503235 .12207031 .50831606 .12207031 .4533844 .12207031 .4267324 .12475586 .4041443 .13012696 .38562013 .13549805 .36709596 .14331055 .35206095 .15356446 .34051515 .16381836 .3289795 .1763509 .32061259 .19116211 .31541444 .20597331 .31021629 .22298177 .3076172 .2421875 .3076172 .26367188 .3076172 .28515626 .30989585 .30664063 .31445313 .328125 .3190104 .34895835 .32535807 .36914063 .3334961 .36914063 .38126628 .36686198 .42326866 .3623047 .45950318 .3577474 .4957377 .35058595 .52596029 .3408203 .5501709 .3310547 .57438156 .31819663 .59258016 .3022461 .60476687 .28629557 .61695358 .2669271 .6230469 .24414063 .6230469Z"/>
|
||||
<path id="font_2_24" d="M.4189453 .4611969C.4189453 .48727928 .41609703 .5097707 .4104004 .52867129 .40470378 .547582 .39542643 .56315109 .38256837 .5753784 .3697103 .5876058 .3528646 .59665426 .33203126 .6025238 .3111979 .6083934 .28548179 .6113281 .2548828 .6113281H.20703125V.30078126H.2578125C.28841148 .30078126 .31404624 .30444847 .3347168 .31178285 .35538737 .31911723 .37198893 .32963053 .38452149 .34332276 .39705406 .35702516 .40592448 .37381999 .4111328 .39370729 .41634117 .41359458 .4189453 .4360911 .4189453 .4611969M.20703125 .25683595V.0390625L.31103517 .025878907V0H.03515625V.025878907L.11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.2758789C.32177735 .65527346 .36010743 .65030416 .39086915 .6403656 .42163087 .63042709 .44628907 .6167348 .46484376 .59928897 .48339845 .5818532 .49658204 .56140139 .50439456 .53793337 .51220706 .5144755 .5161133 .48921714 .5161133 .4621582 .5161133 .43543498 .5122884 .40968833 .5046387 .3849182 .49698893 .3601481 .48413087 .33831278 .46606446 .31941224 .44799806 .3005117 .42382813 .2853546 .3935547 .27394105 .36328126 .26253764 .3256836 .25683595 .28076173 .25683595H.20703125Z"/>
|
||||
<path id="font_2_25" d="M.1430664 .32739259C.1430664 .28186036 .14648438 .24071758 .15332031 .20396424 .16015625 .16721089 .171875 .13590496 .18847656 .11004639 .20507813 .08418783 .2273763 .06426493 .2553711 .05027771 .28336589 .036290487 .31852214 .029296875 .36083985 .029296875 .40315757 .029296875 .4383138 .036290487 .4663086 .05027771 .49430339 .06426493 .5166829 .08418783 .53344729 .11004639 .5502116 .13590496 .5620117 .16721089 .56884768 .20396424 .5756836 .24071758 .57910159 .28186036 .57910159 .32739259 .57910159 .37259928 .5756836 .41349793 .56884768 .4500885 .5620117 .48667909 .5502116 .5177409 .53344729 .5432739 .5166829 .56880697 .49430339 .5884857 .4663086 .6023102 .4383138 .61613467 .40315757 .6230469 .36083985 .6230469 .31852214 .6230469 .28336589 .61613467 .2553711 .6023102 .2273763 .5884857 .20507813 .56880697 .18847656 .5432739 .171875 .5177409 .16015625 .48667909 .15332031 .4500885 .14648438 .41349793 .1430664 .37259928 .1430664 .32739259M.041015626 .32714845C.041015626 .38509117 .048095704 .43513999 .06225586 .47729493 .076416019 .5194499 .09700521 .5541992 .12402344 .58154299 .15104167 .6088867 .18440755 .6291504 .2241211 .642334 .26383464 .6555176 .30940757 .6621094 .36083985 .6621094 .41064454 .6621094 .45532228 .6555176 .49487306 .642334 .5344238 .6291504 .5680339 .6088867 .5957031 .58154299 .6233724 .5541992 .64453127 .5194499 .6591797 .47729493 .6738281 .43513999 .68115237 .38509117 .68115237 .32714845 .68115237 .28548179 .67732748 .2479655 .66967776 .21459961 .662028 .18123372 .65079757 .1517741 .6359863 .1262207 .6211751 .10066732 .60302737 .07877604 .58154299 .060546876 .5600586 .042317708 .53548178 .02750651 .5078125 .016113282L.53222659-.014053345C.54622396-.03156026 .5592448-.046635946 .57128909-.059280397 .5833333-.07193502 .5946452-.0822347 .6052246-.09017944 .615804-.09812418 .62597659-.10395813 .6357422-.107681278 .6455078-.11141459 .65527346-.11328125 .66503909-.11328125 .6673177-.11328125 .67041018-.11319987 .6743164-.11303711 .67822268-.11287435 .6821289-.112711589 .68603518-.11254883 .6899414-.11238607 .6936849-.112223308 .6972656-.11206055 .7008464-.11189779 .7034505-.11165365 .7050781-.111328128V-.14385987C.7014974-.14550781 .6961263-.14731853 .68896487-.14929199 .6818034-.15126546 .67374679-.15323384 .6647949-.15519715 .6558431-.15717061 .6464844-.15881348 .63671877-.16012573 .6269531-.16144817 .6176758-.16210938 .6088867-.16210938 .59033206-.16210938 .5736491-.15974935 .5588379-.1550293 .5440267-.15030925 .529541-.14282227 .51538088-.13256836 .5012207-.12231445 .48673503-.1089681 .47192384-.0925293 .45711265-.07609049 .4404297-.056315107 .421875-.033203126L.40185548-.0078125C.3959961-.008463542 .38972984-.008951823 .38305665-.009277344 .37638346-.009602864 .36897788-.009765625 .36083985-.009765625 .31103517-.009765625 .2664388-.003092448 .22705078 .010253906 .18766277 .02360026 .15413411 .044108076 .12646485 .071777347 .09879557 .09944662 .07763672 .13444011 .06298828 .17675781 .048339845 .21907552 .041015626 .26920573 .041015626 .32714845Z"/>
|
||||
<path id="font_2_14" d="M.44189454 .4951172C.44189454 .4593099 .43318687 .42895509 .41577149 .40405274 .3983561 .3791504 .37483726 .3601888 .34521485 .34716798 .36279298 .34065757 .3787435 .332194 .3930664 .32177735 .4073893 .3113607 .41967774 .29907228 .42993165 .2849121 .44018556 .27075196 .44807945 .25463868 .45361329 .23657227 .45914714 .21850586 .46191407 .1985677 .46191407 .17675781 .46191407 .11490885 .4444987 .068359378 .40966798 .037109376 .37483726 .005859375 .32063804-.009765625 .24707031-.009765625 .17740886-.009765625 .12516277 .0056152346 .09033203 .036376954 .0555013 .06713867 .038085939 .11393229 .038085939 .17675781 .038085939 .22005208 .048502607 .25594078 .06933594 .28442384 .09016927 .3129069 .11832682 .33382163 .1538086 .34716798 .12548828 .3601888 .10245768 .379069 .0847168 .4038086 .066975917 .4285482 .05810547 .45898438 .05810547 .4951172 .05810547 .5208333 .062093099 .54418948 .07006836 .56518557 .07804362 .58618167 .09000651 .60408529 .10595703 .6188965 .121907558 .6337077 .14192708 .6451009 .16601563 .6530762 .19010417 .66105148 .21842449 .66503909 .25097657 .66503909 .28222657 .66503909 .30973307 .6611328 .3334961 .6533203 .35725913 .6455078 .37719728 .63427737 .39331056 .6196289 .40942384 .60498049 .42154948 .5871582 .4296875 .5661621 .43782554 .545166 .44189454 .5214844 .44189454 .4951172M.37402345 .17700196C.37402345 .20072429 .37190757 .22176616 .36767579 .24012757 .363444 .25848899 .35636393 .27400718 .34643556 .28668214 .33650718 .2993571 .32340495 .3089447 .3071289 .31544496 .29085288 .3219452 .27083335 .3251953 .24707031 .3251953 .22298177 .3251953 .203125 .3219452 .1875 .31544496 .171875 .3089447 .1595052 .2993571 .15039063 .28668214 .14127605 .27400718 .13492839 .25848899 .13134766 .24012757 .12776692 .22176616 .12597656 .20072429 .12597656 .17700196 .12597656 .1529541 .12776692 .13174947 .13134766 .11338806 .13492839 .09502665 .14127605 .07967123 .15039063 .06732178 .1595052 .054972333 .171875 .045547487 .1875 .03904724 .203125 .032546998 .22298177 .029296875 .24707031 .029296875 .27083335 .029296875 .29085288 .032546998 .3071289 .03904724 .32340495 .045547487 .33650718 .054972333 .34643556 .06732178 .35636393 .07967123 .363444 .09502665 .36767579 .11338806 .37190757 .13174947 .37402345 .1529541 .37402345 .17700196M.3540039 .4951172C.3540039 .51432296 .35229493 .53190109 .34887696 .54785159 .34545899 .56380209 .339681 .57755538 .33154298 .5891113 .32340495 .6006673 .3125 .6097005 .29882813 .61621096 .28515626 .6227214 .26822917 .62597659 .24804688 .62597659 .22786458 .62597659 .21118164 .6227214 .19799805 .61621096 .18481446 .6097005 .17439778 .6006673 .16674805 .5891113 .15909831 .57755538 .15372722 .56380209 .15063477 .54785159 .14754232 .53190109 .1459961 .51432296 .1459961 .4951172 .1459961 .47558595 .14754232 .4580078 .15063477 .4423828 .15372722 .4267578 .15909831 .41341148 .16674805 .40234376 .17439778 .39127604 .18481446 .3828125 .19799805 .37695313 .21118164 .37109376 .22786458 .36816407 .24804688 .36816407 .26822917 .36816407 .28515626 .37109376 .29882813 .37695313 .3125 .3828125 .32340495 .39127604 .33154298 .40234376 .339681 .41341148 .34545899 .4267578 .34887696 .4423828 .35229493 .4580078 .3540039 .47558595 .3540039 .4951172Z"/>
|
||||
<path id="font_2_33" d="M.37402345 .2421875C.37402345 .27539063 .37101237 .30330406 .36499024 .32592774 .3589681 .34855143 .3503418 .36686198 .33911134 .38085938 .32788087 .39485679 .31437174 .40486656 .29858399 .41088868 .28279624 .4169108 .26529948 .41992188 .24609375 .41992188 .23828125 .41992188 .2298991 .41959635 .22094727 .4189453 .21199544 .41829429 .203125 .41723634 .19433594 .41577149 .18554688 .41430665 .17708333 .41259767 .16894531 .41064454 .1608073 .4086914 .1538086 .40641276 .14794922 .4038086V.040039064C.1616211 .037434896 .1772461 .03548177 .19482422 .034179689 .21240235 .032877607 .22949219 .032226564 .24609375 .032226564 .29101563 .032226564 .32356773 .049804689 .34375 .08496094 .36393229 .12011719 .37402345 .17252605 .37402345 .2421875M.06689453 .66015627 0 .67204287V.69433596H.14794922V.52996829C.14794922 .5237732 .14786785 .51667788 .14770508 .50868228 .14754232 .50069686 .14737956 .49238078 .1472168 .48373414 .14705403 .47509767 .14672852 .4664561 .14624024 .45780946 .14575196 .4491628 .14534505 .44092814 .14501953 .43310548 .15966797 .4446411 .17749024 .45395408 .19848633 .4610443 .21948242 .46813456 .24267578 .4716797 .2680664 .4716797 .3305664 .4716797 .37849937 .45269776 .41186524 .4147339 .4452311 .37677003 .46191407 .31934104 .46191407 .2424469 .46191407 .20366924 .45768229 .16872151 .44921876 .13760376 .44075523 .106486 .42773438 .08000692 .41015626 .058166505 .39257813 .03633626 .37036134 .01955668 .34350587 .007827759 .3166504-.0039011639 .28483073-.009765625 .24804688-.009765625 .23242188-.009765625 .21655274-.008870442 .20043946-.007080078 .18432617-.0052897136 .16845703-.0029296876 .15283203 0 .13720703 .0029296876 .12207031 .0064290368 .107421878 .010498047 .09277344 .014567058 .07926432 .019042969 .06689453 .023925782V.66015627Z"/>
|
||||
</defs>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M0 0H396V170.87078H0Z" fill="#ffffff"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74V132.17078H39.6Z" fill="#ffffff"/>
|
||||
<g clip-path="url(#clip_1)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".45" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#e9eef1" d="M50.184675 50.361539H314.80149"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".45" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#e9eef1" d="M50.184675 74.42308H314.80149"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".45" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#e9eef1" d="M50.184675 98.48461H314.80149"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".45" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#e9eef1" d="M50.184675 122.54615H314.80149"/>
|
||||
</g>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M50.184675 26.3V24"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M50.184675 26.3V24"/>
|
||||
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,48.184675,154.91765)"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M116.33888 26.3V24"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M116.33888 26.3V24"/>
|
||||
<use data-text="2" xlink:href="#font_2_8" transform="matrix(8,0,0,-8,112.338878,154.91765)"/>
|
||||
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8,0,0,-8,116.338878,154.91765)"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M182.49309 26.3V24"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M182.49309 26.3V24"/>
|
||||
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8,0,0,-8,178.49309,154.91765)"/>
|
||||
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,182.49309,154.91765)"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M248.6473 26.3V24"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M248.6473 26.3V24"/>
|
||||
<use data-text="7" xlink:href="#font_2_13" transform="matrix(8,0,0,-8,244.6473,154.91765)"/>
|
||||
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8,0,0,-8,248.6473,154.91765)"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M314.80149 26.3V24"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M314.80149 26.3V24"/>
|
||||
<use data-text="1" xlink:href="#font_2_7" transform="matrix(8,0,0,-8,308.80149,154.91765)"/>
|
||||
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,312.80149,154.91765)"/>
|
||||
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,316.80149,154.91765)"/>
|
||||
<use data-text="S" xlink:href="#font_2_27" transform="matrix(9.5,0,0,-9.5,141.71688,166.55828)"/>
|
||||
<use data-text="o" xlink:href="#font_2_45" transform="matrix(9.5,0,0,-9.5,146.99887,166.55828)"/>
|
||||
<use data-text="n" xlink:href="#font_2_44" transform="matrix(9.5,0,0,-9.5,151.74887,166.55828)"/>
|
||||
<use data-text="g" xlink:href="#font_2_38" transform="matrix(9.5,0,0,-9.5,156.49887,166.55828)"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(9.5,0,0,-9.5,161.24887,166.55828)"/>
|
||||
<use data-text="q" xlink:href="#font_2_47" transform="matrix(9.5,0,0,-9.5,163.62387,166.55828)"/>
|
||||
<use data-text="u" xlink:href="#font_2_51" transform="matrix(9.5,0,0,-9.5,168.37387,166.55828)"/>
|
||||
<use data-text="a" xlink:href="#font_2_32" transform="matrix(9.5,0,0,-9.5,173.12387,166.55828)"/>
|
||||
<use data-text="l" xlink:href="#font_2_42" transform="matrix(9.5,0,0,-9.5,177.34188,166.55828)"/>
|
||||
<use data-text="i" xlink:href="#font_2_40" transform="matrix(9.5,0,0,-9.5,179.98288,166.55828)"/>
|
||||
<use data-text="t" xlink:href="#font_2_50" transform="matrix(9.5,0,0,-9.5,182.62387,166.55828)"/>
|
||||
<use data-text="y" xlink:href="#font_2_54" transform="matrix(9.5,0,0,-9.5,185.26486,166.55828)"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(9.5,0,0,-9.5,190.01486,166.55828)"/>
|
||||
<use data-text="i" xlink:href="#font_2_40" transform="matrix(9.5,0,0,-9.5,192.38986,166.55828)"/>
|
||||
<use data-text="n" xlink:href="#font_2_44" transform="matrix(9.5,0,0,-9.5,195.03087,166.55828)"/>
|
||||
<use data-text="d" xlink:href="#font_2_35" transform="matrix(9.5,0,0,-9.5,199.78087,166.55828)"/>
|
||||
<use data-text="e" xlink:href="#font_2_36" transform="matrix(9.5,0,0,-9.5,204.53087,166.55828)"/>
|
||||
<use data-text="x" xlink:href="#font_2_53" transform="matrix(9.5,0,0,-9.5,208.74887,166.55828)"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(9.5,0,0,-9.5,213.49887,166.55828)"/>
|
||||
<use data-text="↑" xlink:href="#font_2_55" transform="matrix(9.5,0,0,-9.5,215.87387,166.55828)"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M50.184675 26.3H47.884675"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M50.184675 26.3H47.884675"/>
|
||||
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,41.384675,147.34421)"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M50.184675 50.361539H47.884675"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M50.184675 50.361539H47.884675"/>
|
||||
<use data-text="2" xlink:href="#font_2_8" transform="matrix(8,0,0,-8,37.384675,123.28267)"/>
|
||||
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8,0,0,-8,41.384675,123.28267)"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M50.184675 74.42308H47.884675"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M50.184675 74.42308H47.884675"/>
|
||||
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8,0,0,-8,37.384675,99.22113)"/>
|
||||
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,41.384675,99.22113)"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M50.184675 98.48461H47.884675"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M50.184675 98.48461H47.884675"/>
|
||||
<use data-text="7" xlink:href="#font_2_13" transform="matrix(8,0,0,-8,37.384675,75.1596)"/>
|
||||
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8,0,0,-8,41.384675,75.1596)"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M50.184675 122.54615H47.884675"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M50.184675 122.54615H47.884675"/>
|
||||
<use data-text="1" xlink:href="#font_2_7" transform="matrix(8,0,0,-8,33.384675,51.09806)"/>
|
||||
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,37.384675,51.09806)"/>
|
||||
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,41.384675,51.09806)"/>
|
||||
<use data-text="T" xlink:href="#font_2_28" transform="matrix(0,-9.5,-9.5,-0,26.228423,138.88681)"/>
|
||||
<use data-text="e" xlink:href="#font_2_36" transform="matrix(0,-9.5,-9.5,-0,26.228423,133.73856)"/>
|
||||
<use data-text="x" xlink:href="#font_2_53" transform="matrix(0,-9.5,-9.5,-0,26.228423,129.52056)"/>
|
||||
<use data-text="t" xlink:href="#font_2_50" transform="matrix(0,-9.5,-9.5,-0,26.228423,124.77056)"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(0,-9.5,-9.5,-0,26.228423,122.12956)"/>
|
||||
<use data-text="a" xlink:href="#font_2_32" transform="matrix(0,-9.5,-9.5,-0,26.228423,119.75456)"/>
|
||||
<use data-text="l" xlink:href="#font_2_42" transform="matrix(0,-9.5,-9.5,-0,26.228423,115.53656)"/>
|
||||
<use data-text="i" xlink:href="#font_2_40" transform="matrix(0,-9.5,-9.5,-0,26.228423,112.89555)"/>
|
||||
<use data-text="g" xlink:href="#font_2_38" transform="matrix(0,-9.5,-9.5,-0,26.228423,110.254558)"/>
|
||||
<use data-text="n" xlink:href="#font_2_44" transform="matrix(0,-9.5,-9.5,-0,26.228423,105.504558)"/>
|
||||
<use data-text="m" xlink:href="#font_2_43" transform="matrix(0,-9.5,-9.5,-0,26.228423,100.754558)"/>
|
||||
<use data-text="e" xlink:href="#font_2_36" transform="matrix(0,-9.5,-9.5,-0,26.228423,93.363559)"/>
|
||||
<use data-text="n" xlink:href="#font_2_44" transform="matrix(0,-9.5,-9.5,-0,26.228423,89.14555)"/>
|
||||
<use data-text="t" xlink:href="#font_2_50" transform="matrix(0,-9.5,-9.5,-0,26.228423,84.39555)"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(0,-9.5,-9.5,-0,26.228423,81.754558)"/>
|
||||
<use data-text="i" xlink:href="#font_2_40" transform="matrix(0,-9.5,-9.5,-0,26.228423,79.379558)"/>
|
||||
<use data-text="n" xlink:href="#font_2_44" transform="matrix(0,-9.5,-9.5,-0,26.228423,76.738559)"/>
|
||||
<use data-text="d" xlink:href="#font_2_35" transform="matrix(0,-9.5,-9.5,-0,26.228423,71.988559)"/>
|
||||
<use data-text="e" xlink:href="#font_2_36" transform="matrix(0,-9.5,-9.5,-0,26.228423,67.238559)"/>
|
||||
<use data-text="x" xlink:href="#font_2_53" transform="matrix(0,-9.5,-9.5,-0,26.228423,63.020555)"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(0,-9.5,-9.5,-0,26.228423,58.270555)"/>
|
||||
<use data-text="↑" xlink:href="#font_2_55" transform="matrix(0,-9.5,-9.5,-0,26.228423,55.895555)"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".8" stroke-linecap="square" stroke-miterlimit="10" stroke-linejoin="miter" fill="none" stroke="#000000" d="M50.184675 26.3V122.54615"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".8" stroke-linecap="square" stroke-miterlimit="10" stroke-linejoin="miter" fill="none" stroke="#000000" d="M50.184675 26.3H314.80149"/>
|
||||
<g clip-path="url(#clip_3)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M281.9906 109.89433C282.79344 109.89433 283.56349 110.213298 284.13117 110.78098 284.69886 111.34866 285.01783 112.11871 285.01783 112.92154 285.01783 113.724369 284.69886 114.494419 284.13117 115.062099 283.56349 115.62978 282.79344 115.948749 281.9906 115.948749 281.18778 115.948749 280.41773 115.62978 279.85005 115.062099 279.28236 114.494419 278.9634 113.724369 278.9634 112.92154 278.9634 112.11871 279.28236 111.34866 279.85005 110.78098 280.41773 110.213298 281.18778 109.89433 281.9906 109.89433Z" fill="#87b3cb"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width="1" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#293844" d="M281.9906 109.89433C282.79344 109.89433 283.56349 110.213298 284.13117 110.78098 284.69886 111.34866 285.01783 112.11871 285.01783 112.92154 285.01783 113.724369 284.69886 114.494419 284.13117 115.062099 283.56349 115.62978 282.79344 115.948749 281.9906 115.948749 281.18778 115.948749 280.41773 115.62978 279.85005 115.062099 279.28236 114.494419 278.9634 113.724369 278.9634 112.92154 278.9634 112.11871 279.28236 111.34866 279.85005 110.78098 280.41773 110.213298 281.18778 109.89433 281.9906 109.89433Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_4)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M272.79505 97.80651C273.65718 97.80651 274.4841 98.14904 275.09373 98.75865 275.70335 99.36826 276.04588 100.19519 276.04588 101.05731 276.04588 101.91943 275.70335 102.74637 275.09373 103.35598 274.4841 103.96559 273.65718 104.30811 272.79505 104.30811 271.93293 104.30811 271.10603 103.96559 270.4964 103.35598 269.88679 102.74637 269.54426 101.91943 269.54426 101.05731 269.54426 100.19519 269.88679 99.36826 270.4964 98.75865 271.10603 98.14904 271.93293 97.80651 272.79505 97.80651Z" fill="#87b3cb"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#456e87" d="M272.79505 97.80651C273.65718 97.80651 274.4841 98.14904 275.09373 98.75865 275.70335 99.36826 276.04588 100.19519 276.04588 101.05731 276.04588 101.91943 275.70335 102.74637 275.09373 103.35598 274.4841 103.96559 273.65718 104.30811 272.79505 104.30811 271.93293 104.30811 271.10603 103.96559 270.4964 103.35598 269.88679 102.74637 269.54426 101.91943 269.54426 101.05731 269.54426 100.19519 269.88679 99.36826 270.4964 98.75865 271.10603 98.14904 271.93293 97.80651 272.79505 97.80651Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_5)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M266.0198 101.86222C266.84117 101.86222 267.62898 102.188549 268.20973 102.76932 268.7905 103.35009 269.11683 104.1379 269.11683 104.959239 269.11683 105.78057 268.7905 106.56838 268.20973 107.149158 267.62898 107.72993 266.84117 108.05625 266.0198 108.05625 265.1985 108.05625 264.41069 107.72993 263.8299 107.149158 263.2491 106.56838 262.9228 105.78057 262.9228 104.959239 262.9228 104.1379 263.2491 103.35009 263.8299 102.76932 264.41069 102.188549 265.1985 101.86222 266.0198 101.86222Z" fill="#87b3cb"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#456e87" d="M266.0198 101.86222C266.84117 101.86222 267.62898 102.188549 268.20973 102.76932 268.7905 103.35009 269.11683 104.1379 269.11683 104.959239 269.11683 105.78057 268.7905 106.56838 268.20973 107.149158 267.62898 107.72993 266.84117 108.05625 266.0198 108.05625 265.1985 108.05625 264.41069 107.72993 263.8299 107.149158 263.2491 106.56838 262.9228 105.78057 262.9228 104.959239 262.9228 104.1379 263.2491 103.35009 263.8299 102.76932 264.41069 102.188549 265.1985 101.86222 266.0198 101.86222Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_6)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M261.87083 105.26901C262.6437 105.26901 263.38505 105.57608 263.93156 106.12259 264.47807 106.6691 264.78514 107.41042 264.78514 108.183307 264.78514 108.956188 264.47807 109.69752 263.93156 110.244029 263.38505 110.790538 262.6437 111.0976 261.87083 111.0976 261.09794 111.0976 260.3566 110.790538 259.8101 110.244029 259.26359 109.69752 258.9565 108.956188 258.9565 108.183307 258.9565 107.41042 259.26359 106.6691 259.8101 106.12259 260.3566 105.57608 261.09794 105.26901 261.87083 105.26901Z" fill="#87b3cb"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#456e87" d="M261.87083 105.26901C262.6437 105.26901 263.38505 105.57608 263.93156 106.12259 264.47807 106.6691 264.78514 107.41042 264.78514 108.183307 264.78514 108.956188 264.47807 109.69752 263.93156 110.244029 263.38505 110.790538 262.6437 111.0976 261.87083 111.0976 261.09794 111.0976 260.3566 110.790538 259.8101 110.244029 259.26359 109.69752 258.9565 108.956188 258.9565 108.183307 258.9565 107.41042 259.26359 106.6691 259.8101 106.12259 260.3566 105.57608 261.09794 105.26901 261.87083 105.26901Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_7)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M249.76737 105.46063C250.5765 105.46063 251.35262 105.782108 251.92476 106.354259 252.49692 106.92641 252.81839 107.702518 252.81839 108.51166 252.81839 109.3208 252.49692 110.09691 251.92476 110.66906 251.35262 111.24121 250.5765 111.56268 249.76737 111.56268 248.95822 111.56268 248.18212 111.24121 247.60996 110.66906 247.03781 110.09691 246.71634 109.3208 246.71634 108.51166 246.71634 107.702518 247.03781 106.92641 247.60996 106.354259 248.18212 105.782108 248.95822 105.46063 249.76737 105.46063Z" fill="#87b3cb"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#456e87" d="M249.76737 105.46063C250.5765 105.46063 251.35262 105.782108 251.92476 106.354259 252.49692 106.92641 252.81839 107.702518 252.81839 108.51166 252.81839 109.3208 252.49692 110.09691 251.92476 110.66906 251.35262 111.24121 250.5765 111.56268 249.76737 111.56268 248.95822 111.56268 248.18212 111.24121 247.60996 110.66906 247.03781 110.09691 246.71634 109.3208 246.71634 108.51166 246.71634 107.702518 247.03781 106.92641 247.60996 106.354259 248.18212 105.782108 248.95822 105.46063 249.76737 105.46063Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_8)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M238.20763 91.03978C239.01134 91.03978 239.78224 91.35909 240.35056 91.92741 240.91887 92.49572 241.23819 93.266628 241.23819 94.070339 241.23819 94.874057 240.91887 95.64496 240.35056 96.213268 239.78224 96.78158 239.01134 97.1009 238.20763 97.1009 237.40392 97.1009 236.63301 96.78158 236.0647 96.213268 235.49639 95.64496 235.17707 94.874057 235.17707 94.070339 235.17707 93.266628 235.49639 92.49572 236.0647 91.92741 236.63301 91.35909 237.40392 91.03978 238.20763 91.03978Z" fill="#87b3cb"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#456e87" d="M238.20763 91.03978C239.01134 91.03978 239.78224 91.35909 240.35056 91.92741 240.91887 92.49572 241.23819 93.266628 241.23819 94.070339 241.23819 94.874057 240.91887 95.64496 240.35056 96.213268 239.78224 96.78158 239.01134 97.1009 238.20763 97.1009 237.40392 97.1009 236.63301 96.78158 236.0647 96.213268 235.49639 95.64496 235.17707 94.874057 235.17707 94.070339 235.17707 93.266628 235.49639 92.49572 236.0647 91.92741 236.63301 91.35909 237.40392 91.03978 238.20763 91.03978Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_9)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M288.3398 98.6635C289.0268 98.6635 289.68574 98.93643 290.1715 99.4222 290.65727 99.907978 290.9302 100.56691 290.9302 101.25389 290.9302 101.94087 290.65727 102.59981 290.1715 103.08558 289.68574 103.57135 289.0268 103.84429 288.3398 103.84429 287.65284 103.84429 286.9939 103.57135 286.50813 103.08558 286.02238 102.59981 285.74943 101.94087 285.74943 101.25389 285.74943 100.56691 286.02238 99.907978 286.50813 99.4222 286.9939 98.93643 287.65284 98.6635 288.3398 98.6635Z" fill="#87b3cb"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width="1" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#293844" d="M288.3398 98.6635C289.0268 98.6635 289.68574 98.93643 290.1715 99.4222 290.65727 99.907978 290.9302 100.56691 290.9302 101.25389 290.9302 101.94087 290.65727 102.59981 290.1715 103.08558 289.68574 103.57135 289.0268 103.84429 288.3398 103.84429 287.65284 103.84429 286.9939 103.57135 286.50813 103.08558 286.02238 102.59981 285.74943 101.94087 285.74943 101.25389 285.74943 100.56691 286.02238 99.907978 286.50813 99.4222 286.9939 98.93643 287.65284 98.6635 288.3398 98.6635Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_10)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M115.17731 54.870515C115.71685 54.870515 116.23437 55.084878 116.61588 55.466394 116.9974 55.847906 117.21176 56.36542 117.21176 56.904966 117.21176 57.444509 116.9974 57.962026 116.61588 58.34354 116.23437 58.725057 115.71685 58.93942 115.17731 58.93942 114.637767 58.93942 114.12025 58.725057 113.73873 58.34354 113.357219 57.962026 113.14285 57.444509 113.14285 56.904966 113.14285 56.36542 113.357219 55.847906 113.73873 55.466394 114.12025 55.084878 114.637767 54.870515 115.17731 54.870515Z" fill="#ccd6df"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M115.17731 54.870515C115.71685 54.870515 116.23437 55.084878 116.61588 55.466394 116.9974 55.847906 117.21176 56.36542 117.21176 56.904966 117.21176 57.444509 116.9974 57.962026 116.61588 58.34354 116.23437 58.725057 115.71685 58.93942 115.17731 58.93942 114.637767 58.93942 114.12025 58.725057 113.73873 58.34354 113.357219 57.962026 113.14285 57.444509 113.14285 56.904966 113.14285 56.36542 113.357219 55.847906 113.73873 55.466394 114.12025 55.084878 114.637767 54.870515 115.17731 54.870515Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_11)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M76.646358 32.941564C77.43747 32.941564 78.19629 33.255876 78.75569 33.815279 79.315097 34.37468 79.62941 35.1335 79.62941 35.924615 79.62941 36.71573 79.315097 37.47455 78.75569 38.03395 78.19629 38.593355 77.43747 38.90767 76.646358 38.90767 75.85524 38.90767 75.09642 38.593355 74.53702 38.03395 73.977619 37.47455 73.6633 36.71573 73.6633 35.924615 73.6633 35.1335 73.977619 34.37468 74.53702 33.815279 75.09642 33.255876 75.85524 32.941564 76.646358 32.941564Z" fill="#ccd6df"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M76.646358 32.941564C77.43747 32.941564 78.19629 33.255876 78.75569 33.815279 79.315097 34.37468 79.62941 35.1335 79.62941 35.924615 79.62941 36.71573 79.315097 37.47455 78.75569 38.03395 78.19629 38.593355 77.43747 38.90767 76.646358 38.90767 75.85524 38.90767 75.09642 38.593355 74.53702 38.03395 73.977619 37.47455 73.6633 36.71573 73.6633 35.924615 73.6633 35.1335 73.977619 34.37468 74.53702 33.815279 75.09642 33.255876 75.85524 32.941564 76.646358 32.941564Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_12)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M234.55899 63.05629C235.51313 63.05629 236.4283 63.435369 237.10298 64.11004 237.77765 64.78471 238.15673 65.6999 238.15673 66.65403 238.15673 67.608158 237.77765 68.52334 237.10298 69.19801 236.4283 69.87269 235.51313 70.25176 234.55899 70.25176 233.60486 70.25176 232.68968 69.87269 232.015 69.19801 231.34033 68.52334 230.96126 67.608158 230.96126 66.65403 230.96126 65.6999 231.34033 64.78471 232.015 64.11004 232.68968 63.435369 233.60486 63.05629 234.55899 63.05629Z" fill="#ccd6df"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M234.55899 63.05629C235.51313 63.05629 236.4283 63.435369 237.10298 64.11004 237.77765 64.78471 238.15673 65.6999 238.15673 66.65403 238.15673 67.608158 237.77765 68.52334 237.10298 69.19801 236.4283 69.87269 235.51313 70.25176 234.55899 70.25176 233.60486 70.25176 232.68968 69.87269 232.015 69.19801 231.34033 68.52334 230.96126 67.608158 230.96126 66.65403 230.96126 65.6999 231.34033 64.78471 232.015 64.11004 232.68968 63.435369 233.60486 63.05629 234.55899 63.05629Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_13)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M208.17214 94.90285C208.88362 94.90285 209.56606 95.185527 210.06914 95.688617 210.57224 96.1917 210.8549 96.87414 210.8549 97.58561 210.8549 98.29709 210.57224 98.97952 210.06914 99.48261 209.56606 99.9857 208.88362 100.26838 208.17214 100.26838 207.46067 100.26838 206.77823 99.9857 206.27513 99.48261 205.77205 98.97952 205.48938 98.29709 205.48938 97.58561 205.48938 96.87414 205.77205 96.1917 206.27513 95.688617 206.77823 95.185527 207.46067 94.90285 208.17214 94.90285Z" fill="#ccd6df"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M208.17214 94.90285C208.88362 94.90285 209.56606 95.185527 210.06914 95.688617 210.57224 96.1917 210.8549 96.87414 210.8549 97.58561 210.8549 98.29709 210.57224 98.97952 210.06914 99.48261 209.56606 99.9857 208.88362 100.26838 208.17214 100.26838 207.46067 100.26838 206.77823 99.9857 206.27513 99.48261 205.77205 98.97952 205.48938 98.29709 205.48938 97.58561 205.48938 96.87414 205.77205 96.1917 206.27513 95.688617 206.77823 95.185527 207.46067 94.90285 208.17214 94.90285Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_14)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M256.95979 59.883405C257.84819 59.883405 258.70033 60.236368 259.32853 60.864564 259.9567 61.492757 260.30967 62.34489 260.30967 63.23329 260.30967 64.12169 259.9567 64.97382 259.32853 65.60202 258.70033 66.23021 257.84819 66.583179 256.95979 66.583179 256.07139 66.583179 255.21926 66.23021 254.59107 65.60202 253.96286 64.97382 253.6099 64.12169 253.6099 63.23329 253.6099 62.34489 253.96286 61.492757 254.59107 60.864564 255.21926 60.236368 256.07139 59.883405 256.95979 59.883405Z" fill="#ccd6df"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M256.95979 59.883405C257.84819 59.883405 258.70033 60.236368 259.32853 60.864564 259.9567 61.492757 260.30967 62.34489 260.30967 63.23329 260.30967 64.12169 259.9567 64.97382 259.32853 65.60202 258.70033 66.23021 257.84819 66.583179 256.95979 66.583179 256.07139 66.583179 255.21926 66.23021 254.59107 65.60202 253.96286 64.97382 253.6099 64.12169 253.6099 63.23329 253.6099 62.34489 253.96286 61.492757 254.59107 60.864564 255.21926 60.236368 256.07139 59.883405 256.95979 59.883405Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_15)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M149.00801 75.03411C149.65602 75.03411 150.27758 75.291568 150.73578 75.74978 151.194 76.20799 151.45145 76.829547 151.45145 77.477558 151.45145 78.125568 151.194 78.74712 150.73578 79.20533 150.27758 79.66354 149.65602 79.921 149.00801 79.921 148.36 79.921 147.73844 79.66354 147.28023 79.20533 146.82202 78.74712 146.56456 78.125568 146.56456 77.477558 146.56456 76.829547 146.82202 76.20799 147.28023 75.74978 147.73844 75.291568 148.36 75.03411 149.00801 75.03411Z" fill="#ccd6df"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M149.00801 75.03411C149.65602 75.03411 150.27758 75.291568 150.73578 75.74978 151.194 76.20799 151.45145 76.829547 151.45145 77.477558 151.45145 78.125568 151.194 78.74712 150.73578 79.20533 150.27758 79.66354 149.65602 79.921 149.00801 79.921 148.36 79.921 147.73844 79.66354 147.28023 79.20533 146.82202 78.74712 146.56456 78.125568 146.56456 77.477558 146.56456 76.829547 146.82202 76.20799 147.28023 75.74978 147.73844 75.291568 148.36 75.03411 149.00801 75.03411Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_16)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M206.54537 83.58646C207.25676 83.58646 207.93914 83.8691 208.44217 84.37214 208.9452 84.875179 209.22785 85.55754 209.22785 86.26894 209.22785 86.980358 208.9452 87.66271 208.44217 88.16576 207.93914 88.66879 207.25676 88.95144 206.54537 88.95144 205.83396 88.95144 205.1516 88.66879 204.64855 88.16576 204.14551 87.66271 203.86287 86.980358 203.86287 86.26894 203.86287 85.55754 204.14551 84.875179 204.64855 84.37214 205.1516 83.8691 205.83396 83.58646 206.54537 83.58646Z" fill="#ccd6df"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M206.54537 83.58646C207.25676 83.58646 207.93914 83.8691 208.44217 84.37214 208.9452 84.875179 209.22785 85.55754 209.22785 86.26894 209.22785 86.980358 208.9452 87.66271 208.44217 88.16576 207.93914 88.66879 207.25676 88.95144 206.54537 88.95144 205.83396 88.95144 205.1516 88.66879 204.64855 88.16576 204.14551 87.66271 203.86287 86.980358 203.86287 86.26894 203.86287 85.55754 204.14551 84.875179 204.64855 84.37214 205.1516 83.8691 205.83396 83.58646 206.54537 83.58646Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_17)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M236.06554 84.95782C236.94675 84.95782 237.79198 85.30792 238.41509 85.93104 239.0382 86.554149 239.3883 87.39938 239.3883 88.280597 239.3883 89.161808 239.0382 90.00704 238.41509 90.63015 237.79198 91.253269 236.94675 91.60337 236.06554 91.60337 235.18433 91.60337 234.33908 91.253269 233.71598 90.63015 233.09287 90.00704 232.74275 89.161808 232.74275 88.280597 232.74275 87.39938 233.09287 86.554149 233.71598 85.93104 234.33908 85.30792 235.18433 84.95782 236.06554 84.95782Z" fill="#ccd6df"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M236.06554 84.95782C236.94675 84.95782 237.79198 85.30792 238.41509 85.93104 239.0382 86.554149 239.3883 87.39938 239.3883 88.280597 239.3883 89.161808 239.0382 90.00704 238.41509 90.63015 237.79198 91.253269 236.94675 91.60337 236.06554 91.60337 235.18433 91.60337 234.33908 91.253269 233.71598 90.63015 233.09287 90.00704 232.74275 89.161808 232.74275 88.280597 232.74275 87.39938 233.09287 86.554149 233.71598 85.93104 234.33908 85.30792 235.18433 84.95782 236.06554 84.95782Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_18)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M269.333 104.44925C270.199 104.44925 271.02967 104.79331 271.64204 105.40568 272.2544 106.018039 272.59846 106.848697 272.59846 107.7147 272.59846 108.5807 272.2544 109.41136 271.64204 110.02372 271.02967 110.636089 270.199 110.98015 269.333 110.98015 268.467 110.98015 267.63636 110.636089 267.024 110.02372 266.41163 109.41136 266.06758 108.5807 266.06758 107.7147 266.06758 106.848697 266.41163 106.018039 267.024 105.40568 267.63636 104.79331 268.467 104.44925 269.333 104.44925Z" fill="#e58b65"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#b65332" d="M269.333 104.44925C270.199 104.44925 271.02967 104.79331 271.64204 105.40568 272.2544 106.018039 272.59846 106.848697 272.59846 107.7147 272.59846 108.5807 272.2544 109.41136 271.64204 110.02372 271.02967 110.636089 270.199 110.98015 269.333 110.98015 268.467 110.98015 267.63636 110.636089 267.024 110.02372 266.41163 109.41136 266.06758 108.5807 266.06758 107.7147 266.06758 106.848697 266.41163 106.018039 267.024 105.40568 267.63636 104.79331 268.467 104.44925 269.333 104.44925Z"/>
|
||||
</g>
|
||||
<g clip-path="url(#clip_19)">
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M283.95637 103.747409C284.83018 103.747409 285.6683 104.094577 286.2862 104.71245 286.90406 105.33032 287.25123 106.16846 287.25123 107.04227 287.25123 107.91608 286.90406 108.75421 286.2862 109.372089 285.6683 109.98996 284.83018 110.33713 283.95637 110.33713 283.08256 110.33713 282.24443 109.98996 281.62657 109.372089 281.00868 108.75421 280.6615 107.91608 280.6615 107.04227 280.6615 106.16846 281.00868 105.33032 281.62657 104.71245 282.24443 104.094577 283.08256 103.747409 283.95637 103.747409Z" fill="#e58b65"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width="1" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#293844" d="M283.95637 103.747409C284.83018 103.747409 285.6683 104.094577 286.2862 104.71245 286.90406 105.33032 287.25123 106.16846 287.25123 107.04227 287.25123 107.91608 286.90406 108.75421 286.2862 109.372089 285.6683 109.98996 284.83018 110.33713 283.95637 110.33713 283.08256 110.33713 282.24443 109.98996 281.62657 109.372089 281.00868 108.75421 280.6615 107.91608 280.6615 107.04227 280.6615 106.16846 281.00868 105.33032 281.62657 104.71245 282.24443 104.094577 283.08256 103.747409 283.95637 103.747409Z"/>
|
||||
</g>
|
||||
<use data-text="S" xlink:href="#font_2_27" transform="matrix(8.2,0,0,-8.2,81.646358,128.85242)" fill="#4e5b65"/>
|
||||
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,86.20555,128.85242)" fill="#4e5b65"/>
|
||||
<use data-text="n" xlink:href="#font_2_44" transform="matrix(8.2,0,0,-8.2,90.30556,128.85242)" fill="#4e5b65"/>
|
||||
<use data-text="g" xlink:href="#font_2_38" transform="matrix(8.2,0,0,-8.2,94.405559,128.85242)" fill="#4e5b65"/>
|
||||
<use data-text="B" xlink:href="#font_2_17" transform="matrix(8.2,0,0,-8.2,98.505558,128.85242)" fill="#4e5b65"/>
|
||||
<use data-text="l" xlink:href="#font_2_42" transform="matrix(8.2,0,0,-8.2,103.97496,128.85242)" fill="#4e5b65"/>
|
||||
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,106.254558,128.85242)" fill="#4e5b65"/>
|
||||
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,110.35455,128.85242)" fill="#4e5b65"/>
|
||||
<use data-text="m" xlink:href="#font_2_43" transform="matrix(8.2,0,0,-8.2,114.45456,128.85242)" fill="#4e5b65"/>
|
||||
<use data-text="Y" xlink:href="#font_2_31" transform="matrix(8.2,0,0,-8.2,105.98981,106.88768)" fill="#4e5b65"/>
|
||||
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,111.00396,106.88768)" fill="#4e5b65"/>
|
||||
<use data-text="E" xlink:href="#font_2_20" transform="matrix(8.2,0,0,-8.2,115.10396,106.88768)" fill="#4e5b65"/>
|
||||
<use data-text="1" xlink:href="#font_2_7" transform="matrix(8.2,0,0,-8.2,120.11416,106.88768)" fill="#4e5b65"/>
|
||||
<use data-text="D" xlink:href="#font_2_19" transform="matrix(8.2,0,0,-8.2,98.60175,86.346347)" fill="#4e5b65"/>
|
||||
<use data-text="i" xlink:href="#font_2_40" transform="matrix(8.2,0,0,-8.2,104.522159,86.346347)" fill="#4e5b65"/>
|
||||
<use data-text="f" xlink:href="#font_2_37" transform="matrix(8.2,0,0,-8.2,106.80176,86.346347)" fill="#4e5b65"/>
|
||||
<use data-text="f" xlink:href="#font_2_37" transform="matrix(8.2,0,0,-8.2,109.39173,86.346347)" fill="#4e5b65"/>
|
||||
<use data-text="R" xlink:href="#font_2_26" transform="matrix(8.2,0,0,-8.2,112.12233,86.346347)" fill="#4e5b65"/>
|
||||
<use data-text="h" xlink:href="#font_2_39" transform="matrix(8.2,0,0,-8.2,117.59173,86.346347)" fill="#4e5b65"/>
|
||||
<use data-text="y" xlink:href="#font_2_54" transform="matrix(8.2,0,0,-8.2,121.69173,86.346347)" fill="#4e5b65"/>
|
||||
<use data-text="t" xlink:href="#font_2_50" transform="matrix(8.2,0,0,-8.2,125.79173,86.346347)" fill="#4e5b65"/>
|
||||
<use data-text="h" xlink:href="#font_2_39" transform="matrix(8.2,0,0,-8.2,128.07134,86.346347)" fill="#4e5b65"/>
|
||||
<use data-text="m" xlink:href="#font_2_43" transform="matrix(8.2,0,0,-8.2,132.17133,86.346347)" fill="#4e5b65"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,138.55094,86.346347)" fill="#4e5b65"/>
|
||||
<use data-text="2" xlink:href="#font_2_8" transform="matrix(8.2,0,0,-8.2,140.60092,86.346347)" fill="#4e5b65"/>
|
||||
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,183.20162,92.523708)" fill="#4e5b65"/>
|
||||
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,190.49141,92.523708)" fill="#4e5b65"/>
|
||||
<use data-text="s" xlink:href="#font_2_49" transform="matrix(8.2,0,0,-8.2,194.59142,92.523708)" fill="#4e5b65"/>
|
||||
<use data-text="e" xlink:href="#font_2_36" transform="matrix(8.2,0,0,-8.2,197.78122,92.523708)" fill="#4e5b65"/>
|
||||
<use data-text="L" xlink:href="#font_2_22" transform="matrix(8.2,0,0,-8.2,205.71524,112.138629)" fill="#4e5b65"/>
|
||||
<use data-text="e" xlink:href="#font_2_36" transform="matrix(8.2,0,0,-8.2,210.72544,112.138629)" fill="#4e5b65"/>
|
||||
<use data-text="V" xlink:href="#font_2_29" transform="matrix(8.2,0,0,-8.2,214.36624,112.138629)" fill="#4e5b65"/>
|
||||
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,219.22414,112.138629)" fill="#4e5b65"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,223.32415,112.138629)" fill="#4e5b65"/>
|
||||
<use data-text="2" xlink:href="#font_2_8" transform="matrix(8.2,0,0,-8.2,225.37415,112.138629)" fill="#4e5b65"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#4e5b65" d="M249.29737 75.38554H247.29346L238.77109 85.17332"/>
|
||||
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,251.29346,97.40711)" fill="#4e5b65"/>
|
||||
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,258.58326,97.40711)" fill="#4e5b65"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,265.87306,97.40711)" fill="#4e5b65"/>
|
||||
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,267.92308,97.40711)" fill="#4e5b65"/>
|
||||
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,275.21287,97.40711)" fill="#4e5b65"/>
|
||||
<use data-text="s" xlink:href="#font_2_49" transform="matrix(8.2,0,0,-8.2,279.31288,97.40711)" fill="#4e5b65"/>
|
||||
<use data-text="i" xlink:href="#font_2_40" transform="matrix(8.2,0,0,-8.2,282.50267,97.40711)" fill="#4e5b65"/>
|
||||
<use data-text="c" xlink:href="#font_2_34" transform="matrix(8.2,0,0,-8.2,284.78227,97.40711)" fill="#4e5b65"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,288.42308,97.40711)" fill="#4e5b65"/>
|
||||
<use data-text="3" xlink:href="#font_2_9" transform="matrix(8.2,0,0,-8.2,290.47306,97.40711)" fill="#4e5b65"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#456e87" d="M249.29737 85.010158H247.29346L240.9205 91.36511"/>
|
||||
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,251.29346,87.78249)" fill="#456e87"/>
|
||||
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,258.58326,87.78249)" fill="#456e87"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,265.87306,87.78249)" fill="#456e87"/>
|
||||
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,267.92308,87.78249)" fill="#456e87"/>
|
||||
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,275.21287,87.78249)" fill="#456e87"/>
|
||||
<use data-text="s" xlink:href="#font_2_49" transform="matrix(8.2,0,0,-8.2,279.31288,87.78249)" fill="#456e87"/>
|
||||
<use data-text="i" xlink:href="#font_2_40" transform="matrix(8.2,0,0,-8.2,282.50267,87.78249)" fill="#456e87"/>
|
||||
<use data-text="c" xlink:href="#font_2_34" transform="matrix(8.2,0,0,-8.2,284.78227,87.78249)" fill="#456e87"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,288.42308,87.78249)" fill="#456e87"/>
|
||||
<use data-text="2" xlink:href="#font_2_8" transform="matrix(8.2,0,0,-8.2,290.47306,87.78249)" fill="#456e87"/>
|
||||
<use data-text="." xlink:href="#font_2_5" transform="matrix(8.2,0,0,-8.2,294.57307,87.78249)" fill="#456e87"/>
|
||||
<use data-text="6" xlink:href="#font_2_12" transform="matrix(8.2,0,0,-8.2,296.62306,87.78249)" fill="#456e87"/>
|
||||
<use data-text="A" xlink:href="#font_2_16" transform="matrix(8.2,0,0,-8.2,154.87526,74.20704)" fill="#4e5b65"/>
|
||||
<use data-text="C" xlink:href="#font_2_18" transform="matrix(8.2,0,0,-8.2,160.79566,74.20704)" fill="#4e5b65"/>
|
||||
<use data-text="E" xlink:href="#font_2_20" transform="matrix(8.2,0,0,-8.2,166.26506,74.20704)" fill="#4e5b65"/>
|
||||
<use data-text="-" xlink:href="#font_2_4" transform="matrix(8.2,0,0,-8.2,171.27527,74.20704)" fill="#4e5b65"/>
|
||||
<use data-text="S" xlink:href="#font_2_27" transform="matrix(8.2,0,0,-8.2,174.00586,74.20704)" fill="#4e5b65"/>
|
||||
<use data-text="t" xlink:href="#font_2_50" transform="matrix(8.2,0,0,-8.2,178.56507,74.20704)" fill="#4e5b65"/>
|
||||
<use data-text="e" xlink:href="#font_2_36" transform="matrix(8.2,0,0,-8.2,180.84467,74.20704)" fill="#4e5b65"/>
|
||||
<use data-text="p" xlink:href="#font_2_46" transform="matrix(8.2,0,0,-8.2,184.48546,74.20704)" fill="#4e5b65"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,188.58547,74.20704)" fill="#4e5b65"/>
|
||||
<use data-text="1" xlink:href="#font_2_7" transform="matrix(8.2,0,0,-8.2,190.63547,74.20704)" fill="#4e5b65"/>
|
||||
<use data-text="." xlink:href="#font_2_5" transform="matrix(8.2,0,0,-8.2,194.73546,74.20704)" fill="#4e5b65"/>
|
||||
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8.2,0,0,-8.2,196.78546,74.20704)" fill="#4e5b65"/>
|
||||
<use data-text="H" xlink:href="#font_2_21" transform="matrix(8.2,0,0,-8.2,261.95979,115.55936)" fill="#4e5b65"/>
|
||||
<use data-text="e" xlink:href="#font_2_36" transform="matrix(8.2,0,0,-8.2,267.8802,115.55936)" fill="#4e5b65"/>
|
||||
<use data-text="a" xlink:href="#font_2_32" transform="matrix(8.2,0,0,-8.2,271.52098,115.55936)" fill="#4e5b65"/>
|
||||
<use data-text="r" xlink:href="#font_2_48" transform="matrix(8.2,0,0,-8.2,275.16178,115.55936)" fill="#4e5b65"/>
|
||||
<use data-text="t" xlink:href="#font_2_50" transform="matrix(8.2,0,0,-8.2,277.89237,115.55936)" fill="#4e5b65"/>
|
||||
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,280.17198,115.55936)" fill="#4e5b65"/>
|
||||
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,287.4618,115.55936)" fill="#4e5b65"/>
|
||||
<use data-text="L" xlink:href="#font_2_22" transform="matrix(8.2,0,0,-8.2,291.56178,115.55936)" fill="#4e5b65"/>
|
||||
<use data-text="a" xlink:href="#font_2_32" transform="matrix(8.2,0,0,-8.2,296.572,115.55936)" fill="#4e5b65"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#456e87" d="M304.86689 131.20832H302.86299L284.8672 115.44177"/>
|
||||
<use data-text="S" xlink:href="#font_2_27" transform="matrix(8.2,0,0,-8.2,306.86299,41.584337)" fill="#456e87"/>
|
||||
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,311.42219,41.584337)" fill="#456e87"/>
|
||||
<use data-text="n" xlink:href="#font_2_44" transform="matrix(8.2,0,0,-8.2,315.5222,41.584337)" fill="#456e87"/>
|
||||
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,319.6222,41.584337)" fill="#456e87"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,323.72218,41.584337)" fill="#456e87"/>
|
||||
<use data-text="v" xlink:href="#font_2_52" transform="matrix(8.2,0,0,-8.2,325.7722,41.584337)" fill="#456e87"/>
|
||||
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8.2,0,0,-8.2,329.8722,41.584337)" fill="#456e87"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#456e87" d="M247.21591 103.959239H249.21982L266.0198 96.959239V101.064708"/>
|
||||
<use data-text="S" xlink:href="#font_2_27" transform="matrix(8.2,0,0,-8.2,211.75107,68.83341)" fill="#456e87"/>
|
||||
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,216.31027,68.83341)" fill="#456e87"/>
|
||||
<use data-text="n" xlink:href="#font_2_44" transform="matrix(8.2,0,0,-8.2,220.41027,68.83341)" fill="#456e87"/>
|
||||
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,224.51027,68.83341)" fill="#456e87"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,228.61028,68.83341)" fill="#456e87"/>
|
||||
<use data-text="v" xlink:href="#font_2_52" transform="matrix(8.2,0,0,-8.2,230.66027,68.83341)" fill="#456e87"/>
|
||||
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8.2,0,0,-8.2,234.76027,68.83341)" fill="#456e87"/>
|
||||
<use data-text="." xlink:href="#font_2_5" transform="matrix(8.2,0,0,-8.2,238.86028,68.83341)" fill="#456e87"/>
|
||||
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8.2,0,0,-8.2,240.91027,68.83341)" fill="#456e87"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#456e87" d="M255.94353 144.68277H257.93965L261.473 111.876918"/>
|
||||
<use data-text="S" xlink:href="#font_2_27" transform="matrix(8.2,0,0,-8.2,226.65837,28.109879)" fill="#456e87"/>
|
||||
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,231.21758,28.109879)" fill="#456e87"/>
|
||||
<use data-text="n" xlink:href="#font_2_44" transform="matrix(8.2,0,0,-8.2,235.31757,28.109879)" fill="#456e87"/>
|
||||
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,239.41757,28.109879)" fill="#456e87"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,243.51758,28.109879)" fill="#456e87"/>
|
||||
<use data-text="v" xlink:href="#font_2_52" transform="matrix(8.2,0,0,-8.2,245.56757,28.109879)" fill="#456e87"/>
|
||||
<use data-text="6" xlink:href="#font_2_12" transform="matrix(8.2,0,0,-8.2,249.66757,28.109879)" fill="#456e87"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#456e87" d="M226.82787 131.20832H228.83177L247.15808 111.34042"/>
|
||||
<use data-text="S" xlink:href="#font_2_27" transform="matrix(8.2,0,0,-8.2,179.48802,41.584337)" fill="#456e87"/>
|
||||
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,184.04723,41.584337)" fill="#456e87"/>
|
||||
<use data-text="n" xlink:href="#font_2_44" transform="matrix(8.2,0,0,-8.2,188.14722,41.584337)" fill="#456e87"/>
|
||||
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,192.24723,41.584337)" fill="#456e87"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,196.34723,41.584337)" fill="#456e87"/>
|
||||
<use data-text="v" xlink:href="#font_2_52" transform="matrix(8.2,0,0,-8.2,198.39722,41.584337)" fill="#456e87"/>
|
||||
<use data-text="6" xlink:href="#font_2_12" transform="matrix(8.2,0,0,-8.2,202.49723,41.584337)" fill="#456e87"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,206.59723,41.584337)" fill="#456e87"/>
|
||||
<use data-text="W" xlink:href="#font_2_30" transform="matrix(8.2,0,0,-8.2,208.50659,41.584337)" fill="#456e87"/>
|
||||
<use data-text="i" xlink:href="#font_2_40" transform="matrix(8.2,0,0,-8.2,215.91928,41.584337)" fill="#456e87"/>
|
||||
<use data-text="l" xlink:href="#font_2_42" transform="matrix(8.2,0,0,-8.2,218.19889,41.584337)" fill="#456e87"/>
|
||||
<use data-text="d" xlink:href="#font_2_35" transform="matrix(8.2,0,0,-8.2,220.47849,41.584337)" fill="#456e87"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#456e87" d="M304.86689 90.78492H302.86299L276.6306 99.74693"/>
|
||||
<use data-text="S" xlink:href="#font_2_27" transform="matrix(8.2,0,0,-8.2,306.86299,82.00773)" fill="#456e87"/>
|
||||
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,311.42219,82.00773)" fill="#456e87"/>
|
||||
<use data-text="n" xlink:href="#font_2_44" transform="matrix(8.2,0,0,-8.2,315.5222,82.00773)" fill="#456e87"/>
|
||||
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,319.6222,82.00773)" fill="#456e87"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,323.72218,82.00773)" fill="#456e87"/>
|
||||
<use data-text="v" xlink:href="#font_2_52" transform="matrix(8.2,0,0,-8.2,325.7722,82.00773)" fill="#456e87"/>
|
||||
<use data-text="4" xlink:href="#font_2_10" transform="matrix(8.2,0,0,-8.2,329.8722,82.00773)" fill="#456e87"/>
|
||||
<use data-text="." xlink:href="#font_2_5" transform="matrix(8.2,0,0,-8.2,333.97218,82.00773)" fill="#456e87"/>
|
||||
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8.2,0,0,-8.2,336.0222,82.00773)" fill="#456e87"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#b65332" d="M279.46369 144.68277H277.45979L270.20503 111.681369"/>
|
||||
<use data-text="Y" xlink:href="#font_20_10" transform="matrix(9.3,0,0,-9.3,281.45979,28.352066)" fill="#b65332"/>
|
||||
<use data-text="u" xlink:href="#font_20_12" transform="matrix(9.3,0,0,-9.3,287.33064,28.352066)" fill="#b65332"/>
|
||||
<use data-text="E" xlink:href="#font_20_9" transform="matrix(9.3,0,0,-9.3,292.50144,28.352066)" fill="#b65332"/>
|
||||
<use data-text="2" xlink:href="#font_20_6" transform="matrix(9.3,0,0,-9.3,298.70454,28.352066)" fill="#b65332"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#b65332" d="M304.86689 117.73385H302.86299L287.52214 109.058689"/>
|
||||
<use data-text="Y" xlink:href="#font_20_10" transform="matrix(9.3,0,0,-9.3,306.86299,55.300989)" fill="#b65332"/>
|
||||
<use data-text="u" xlink:href="#font_20_12" transform="matrix(9.3,0,0,-9.3,312.73384,55.300989)" fill="#b65332"/>
|
||||
<use data-text="E" xlink:href="#font_20_9" transform="matrix(9.3,0,0,-9.3,317.90464,55.300989)" fill="#b65332"/>
|
||||
<use data-text="2" xlink:href="#font_20_6" transform="matrix(9.3,0,0,-9.3,324.10774,55.300989)" fill="#b65332"/>
|
||||
<use data-text=" " xlink:href="#font_20_3" transform="matrix(9.3,0,0,-9.3,328.75773,55.300989)" fill="#b65332"/>
|
||||
<use data-text="(" xlink:href="#font_20_4" transform="matrix(9.3,0,0,-9.3,331.08274,55.300989)" fill="#b65332"/>
|
||||
<use data-text="B" xlink:href="#font_20_8" transform="matrix(9.3,0,0,-9.3,334.17964,55.300989)" fill="#b65332"/>
|
||||
<use data-text="o" xlink:href="#font_20_11" transform="matrix(9.3,0,0,-9.3,340.38273,55.300989)" fill="#b65332"/>
|
||||
<use data-text="8" xlink:href="#font_20_7" transform="matrix(9.3,0,0,-9.3,345.0327,55.300989)" fill="#b65332"/>
|
||||
<use data-text=")" xlink:href="#font_20_5" transform="matrix(9.3,0,0,-9.3,349.68275,55.300989)" fill="#b65332"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#456e87" d="M304.86689 104.259387H302.86299L291.66215 101.94143"/>
|
||||
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,306.86299,68.533267)" fill="#456e87"/>
|
||||
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,314.15278,68.533267)" fill="#456e87"/>
|
||||
<use data-text="r" xlink:href="#font_2_48" transform="matrix(8.2,0,0,-8.2,318.25279,68.533267)" fill="#456e87"/>
|
||||
<use data-text="e" xlink:href="#font_2_36" transform="matrix(8.2,0,0,-8.2,320.98338,68.533267)" fill="#456e87"/>
|
||||
<use data-text="k" xlink:href="#font_2_41" transform="matrix(8.2,0,0,-8.2,324.62419,68.533267)" fill="#456e87"/>
|
||||
<use data-text="a" xlink:href="#font_2_32" transform="matrix(8.2,0,0,-8.2,328.72419,68.533267)" fill="#456e87"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,332.365,68.533267)" fill="#456e87"/>
|
||||
<use data-text="9" xlink:href="#font_2_15" transform="matrix(8.2,0,0,-8.2,334.41499,68.533267)" fill="#456e87"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M277.2 160.63213C277.77293 160.63213 278.32243 160.85974 278.72755 161.26485 279.13264 161.66996 279.36027 162.21947 279.36027 162.79238 279.36027 163.36528 279.13264 163.9148 278.72755 164.3199 278.32243 164.725 277.77293 164.95262 277.2 164.95262 276.6271 164.95262 276.07759 164.725 275.6725 164.3199 275.26737 163.9148 275.03977 163.36528 275.03977 162.79238 275.03977 162.21947 275.26737 161.66996 275.6725 161.26485 276.07759 160.85974 276.6271 160.63213 277.2 160.63213Z" fill="#dde4e9"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".45" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M277.2 160.63213C277.77293 160.63213 278.32243 160.85974 278.72755 161.26485 279.13264 161.66996 279.36027 162.21947 279.36027 162.79238 279.36027 163.36528 279.13264 163.9148 278.72755 164.3199 278.32243 164.725 277.77293 164.95262 277.2 164.95262 276.6271 164.95262 276.07759 164.725 275.6725 164.3199 275.26737 163.9148 275.03977 163.36528 275.03977 162.79238 275.03977 162.21947 275.26737 161.66996 275.6725 161.26485 276.07759 160.85974 276.6271 160.63213 277.2 160.63213Z"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M314.82 159.96395C315.5701 159.96395 316.28959 160.26197 316.82 160.79238 317.3504 161.32277 317.64845 162.04227 317.64845 162.79238 317.64845 163.54248 317.3504 164.26197 316.82 164.79238 316.28959 165.32277 315.5701 165.62079 314.82 165.62079 314.0699 165.62079 313.3504 165.32277 312.82 164.79238 312.28959 164.26197 311.99159 163.54248 311.99159 162.79238 311.99159 162.04227 312.28959 161.32277 312.82 160.79238 313.3504 160.26197 314.0699 159.96395 314.82 159.96395Z" fill="#dde4e9"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".45" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M314.82 159.96395C315.5701 159.96395 316.28959 160.26197 316.82 160.79238 317.3504 161.32277 317.64845 162.04227 317.64845 162.79238 317.64845 163.54248 317.3504 164.26197 316.82 164.79238 316.28959 165.32277 315.5701 165.62079 314.82 165.62079 314.0699 165.62079 313.3504 165.32277 312.82 164.79238 312.28959 164.26197 311.99159 163.54248 311.99159 162.79238 311.99159 162.04227 312.28959 161.32277 312.82 160.79238 313.3504 160.26197 314.0699 159.96395 314.82 159.96395Z"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M352.44 159.42588C353.3328 159.42588 354.18919 159.78058 354.82048 160.4119 355.45179 161.0432 355.8065 161.89957 355.8065 162.79238 355.8065 163.68518 355.45179 164.54154 354.82048 165.17285 354.18919 165.80416 353.3328 166.15888 352.44 166.15888 351.54719 166.15888 350.69084 165.80416 350.0595 165.17285 349.42823 164.54154 349.0735 163.68518 349.0735 162.79238 349.0735 161.89957 349.42823 161.0432 350.0595 160.4119 350.69084 159.78058 351.54719 159.42588 352.44 159.42588Z" fill="#dde4e9"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".45" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M352.44 159.42588C353.3328 159.42588 354.18919 159.78058 354.82048 160.4119 355.45179 161.0432 355.8065 161.89957 355.8065 162.79238 355.8065 163.68518 355.45179 164.54154 354.82048 165.17285 354.18919 165.80416 353.3328 166.15888 352.44 166.15888 351.54719 166.15888 350.69084 165.80416 350.0595 165.17285 349.42823 164.54154 349.0735 163.68518 349.0735 162.79238 349.0735 161.89957 349.42823 161.0432 350.0595 160.4119 350.69084 159.78058 351.54719 159.42588 352.44 159.42588Z"/>
|
||||
<use data-text="A" xlink:href="#font_2_16" transform="matrix(7.4,0,0,-7.4,223.74,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="u" xlink:href="#font_2_51" transform="matrix(7.4,0,0,-7.4,229.08281,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="d" xlink:href="#font_2_35" transform="matrix(7.4,0,0,-7.4,232.7828,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="i" xlink:href="#font_2_40" transform="matrix(7.4,0,0,-7.4,236.4828,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="o" xlink:href="#font_2_45" transform="matrix(7.4,0,0,-7.4,238.54001,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="B" xlink:href="#font_2_17" transform="matrix(7.4,0,0,-7.4,242.24,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="o" xlink:href="#font_2_45" transform="matrix(7.4,0,0,-7.4,247.17581,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="x" xlink:href="#font_2_53" transform="matrix(7.4,0,0,-7.4,250.87581,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text=" " xlink:href="#font_2_3" transform="matrix(7.4,0,0,-7.4,254.5758,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="P" xlink:href="#font_2_24" transform="matrix(7.4,0,0,-7.4,256.4258,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="Q" xlink:href="#font_2_25" transform="matrix(7.4,0,0,-7.4,260.54023,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="7" xlink:href="#font_2_13" transform="matrix(7.4,0,0,-7.4,284.328,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="." xlink:href="#font_2_5" transform="matrix(7.4,0,0,-7.4,288.028,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="9" xlink:href="#font_2_15" transform="matrix(7.4,0,0,-7.4,289.878,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="8" xlink:href="#font_2_14" transform="matrix(7.4,0,0,-7.4,321.948,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="." xlink:href="#font_2_5" transform="matrix(7.4,0,0,-7.4,325.648,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="1" xlink:href="#font_2_7" transform="matrix(7.4,0,0,-7.4,327.498,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="8" xlink:href="#font_2_14" transform="matrix(7.4,0,0,-7.4,359.568,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="." xlink:href="#font_2_5" transform="matrix(7.4,0,0,-7.4,363.268,9.7659)" fill="#4e5b65"/>
|
||||
<use data-text="3" xlink:href="#font_2_9" transform="matrix(7.4,0,0,-7.4,365.11799,9.7659)" fill="#4e5b65"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M43.3 163.02677C43.96301 163.02677 44.59895 163.29019 45.06777 163.759 45.536584 164.22782 45.8 164.86376 45.8 165.52677 45.8 166.18978 45.536584 166.82572 45.06777 167.29454 44.59895 167.76335 43.96301 168.02677 43.3 168.02677 42.636995 168.02677 42.00105 167.76335 41.532236 167.29454 41.063417 166.82572 40.8 166.18978 40.8 165.52677 40.8 164.86376 41.063417 164.22782 41.532236 163.759 42.00105 163.29019 42.636995 163.02677 43.3 163.02677Z" fill="#ccd6df"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M43.3 163.02677C43.96301 163.02677 44.59895 163.29019 45.06777 163.759 45.536584 164.22782 45.8 164.86376 45.8 165.52677 45.8 166.18978 45.536584 166.82572 45.06777 167.29454 44.59895 167.76335 43.96301 168.02677 43.3 168.02677 42.636995 168.02677 42.00105 167.76335 41.532236 167.29454 41.063417 166.82572 40.8 166.18978 40.8 165.52677 40.8 164.86376 41.063417 164.22782 41.532236 163.759 42.00105 163.29019 42.636995 163.02677 43.3 163.02677Z"/>
|
||||
<use data-text="P" xlink:href="#font_2_24" transform="matrix(7.4,0,0,-7.4,50.33,7.9340059)" fill="#293844"/>
|
||||
<use data-text="u" xlink:href="#font_2_51" transform="matrix(7.4,0,0,-7.4,54.4444,7.9340059)" fill="#293844"/>
|
||||
<use data-text="b" xlink:href="#font_2_33" transform="matrix(7.4,0,0,-7.4,58.1444,7.9340059)" fill="#293844"/>
|
||||
<use data-text="l" xlink:href="#font_2_42" transform="matrix(7.4,0,0,-7.4,61.844404,7.9340059)" fill="#293844"/>
|
||||
<use data-text="i" xlink:href="#font_2_40" transform="matrix(7.4,0,0,-7.4,63.901605,7.9340059)" fill="#293844"/>
|
||||
<use data-text="c" xlink:href="#font_2_34" transform="matrix(7.4,0,0,-7.4,65.9588,7.9340059)" fill="#293844"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M80.65938 163.02677C81.32238 163.02677 81.95833 163.29019 82.42714 163.759 82.89596 164.22782 83.15938 164.86376 83.15938 165.52677 83.15938 166.18978 82.89596 166.82572 82.42714 167.29454 81.95833 167.76335 81.32238 168.02677 80.65938 168.02677 79.99637 168.02677 79.36043 167.76335 78.89161 167.29454 78.42279 166.82572 78.15938 166.18978 78.15938 165.52677 78.15938 164.86376 78.42279 164.22782 78.89161 163.759 79.36043 163.29019 79.99637 163.02677 80.65938 163.02677Z" fill="#87b3cb"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#456e87" d="M80.65938 163.02677C81.32238 163.02677 81.95833 163.29019 82.42714 163.759 82.89596 164.22782 83.15938 164.86376 83.15938 165.52677 83.15938 166.18978 82.89596 166.82572 82.42714 167.29454 81.95833 167.76335 81.32238 168.02677 80.65938 168.02677 79.99637 168.02677 79.36043 167.76335 78.89161 167.29454 78.42279 166.82572 78.15938 166.18978 78.15938 165.52677 78.15938 164.86376 78.42279 164.22782 78.89161 163.759 79.36043 163.29019 79.99637 163.02677 80.65938 163.02677Z"/>
|
||||
<use data-text="P" xlink:href="#font_2_24" transform="matrix(7.4,0,0,-7.4,87.68938,7.9340059)" fill="#293844"/>
|
||||
<use data-text="r" xlink:href="#font_2_48" transform="matrix(7.4,0,0,-7.4,91.80378,7.9340059)" fill="#293844"/>
|
||||
<use data-text="o" xlink:href="#font_2_45" transform="matrix(7.4,0,0,-7.4,94.267978,7.9340059)" fill="#293844"/>
|
||||
<use data-text="p" xlink:href="#font_2_46" transform="matrix(7.4,0,0,-7.4,97.96798,7.9340059)" fill="#293844"/>
|
||||
<use data-text="r" xlink:href="#font_2_48" transform="matrix(7.4,0,0,-7.4,101.66798,7.9340059)" fill="#293844"/>
|
||||
<use data-text="i" xlink:href="#font_2_40" transform="matrix(7.4,0,0,-7.4,104.13218,7.9340059)" fill="#293844"/>
|
||||
<use data-text="e" xlink:href="#font_2_36" transform="matrix(7.4,0,0,-7.4,106.18938,7.9340059)" fill="#293844"/>
|
||||
<use data-text="t" xlink:href="#font_2_50" transform="matrix(7.4,0,0,-7.4,109.474979,7.9340059)" fill="#293844"/>
|
||||
<use data-text="a" xlink:href="#font_2_32" transform="matrix(7.4,0,0,-7.4,111.53218,7.9340059)" fill="#293844"/>
|
||||
<use data-text="r" xlink:href="#font_2_48" transform="matrix(7.4,0,0,-7.4,114.81778,7.9340059)" fill="#293844"/>
|
||||
<use data-text="y" xlink:href="#font_2_54" transform="matrix(7.4,0,0,-7.4,117.28198,7.9340059)" fill="#293844"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M132.34688 162.67678C133.1027 162.67678 133.82769 162.97707 134.36212 163.51152 134.89658 164.04596 135.19687 164.77094 135.19687 165.52677 135.19687 166.2826 134.89658 167.00757 134.36212 167.54203 133.82769 168.07648 133.1027 168.37677 132.34688 168.37677 131.59105 168.37677 130.86608 168.07648 130.33162 167.54203 129.79717 167.00757 129.49687 166.2826 129.49687 165.52677 129.49687 164.77094 129.79717 164.04596 130.33162 163.51152 130.86608 162.97707 131.59105 162.67678 132.34688 162.67678Z" fill="#ffffff"/>
|
||||
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width="1" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#293844" d="M132.34688 162.67678C133.1027 162.67678 133.82769 162.97707 134.36212 163.51152 134.89658 164.04596 135.19687 164.77094 135.19687 165.52677 135.19687 166.2826 134.89658 167.00757 134.36212 167.54203 133.82769 168.07648 133.1027 168.37677 132.34688 168.37677 131.59105 168.37677 130.86608 168.07648 130.33162 167.54203 129.79717 167.00757 129.49687 166.2826 129.49687 165.52677 129.49687 164.77094 129.79717 164.04596 130.33162 163.51152 130.86608 162.97707 131.59105 162.67678 132.34688 162.67678Z"/>
|
||||
<use data-text="P" xlink:href="#font_2_24" transform="matrix(7.4,0,0,-7.4,139.37688,7.9340059)" fill="#293844"/>
|
||||
<use data-text="a" xlink:href="#font_2_32" transform="matrix(7.4,0,0,-7.4,143.49127,7.9340059)" fill="#293844"/>
|
||||
<use data-text="r" xlink:href="#font_2_48" transform="matrix(7.4,0,0,-7.4,146.77687,7.9340059)" fill="#293844"/>
|
||||
<use data-text="e" xlink:href="#font_2_36" transform="matrix(7.4,0,0,-7.4,149.24108,7.9340059)" fill="#293844"/>
|
||||
<use data-text="t" xlink:href="#font_2_50" transform="matrix(7.4,0,0,-7.4,152.52667,7.9340059)" fill="#293844"/>
|
||||
<use data-text="o" xlink:href="#font_2_45" transform="matrix(7.4,0,0,-7.4,154.58388,7.9340059)" fill="#293844"/>
|
||||
<use data-text="-" xlink:href="#font_2_4" transform="matrix(7.4,0,0,-7.4,158.28388,7.9340059)" fill="#293844"/>
|
||||
<use data-text="o" xlink:href="#font_2_45" transform="matrix(7.4,0,0,-7.4,160.74808,7.9340059)" fill="#293844"/>
|
||||
<use data-text="p" xlink:href="#font_2_46" transform="matrix(7.4,0,0,-7.4,164.44808,7.9340059)" fill="#293844"/>
|
||||
<use data-text="t" xlink:href="#font_2_50" transform="matrix(7.4,0,0,-7.4,168.14807,7.9340059)" fill="#293844"/>
|
||||
<use data-text="i" xlink:href="#font_2_40" transform="matrix(7.4,0,0,-7.4,170.20528,7.9340059)" fill="#293844"/>
|
||||
<use data-text="m" xlink:href="#font_2_43" transform="matrix(7.4,0,0,-7.4,172.26248,7.9340059)" fill="#293844"/>
|
||||
<use data-text="a" xlink:href="#font_2_32" transform="matrix(7.4,0,0,-7.4,178.01969,7.9340059)" fill="#293844"/>
|
||||
<use data-text="l" xlink:href="#font_2_42" transform="matrix(7.4,0,0,-7.4,181.30529,7.9340059)" fill="#293844"/>
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 141 KiB |
|
After Width: | Height: | Size: 170 KiB |
|
After Width: | Height: | Size: 170 KiB |
|
Before Width: | Height: | Size: 14 KiB After Width: | Height: | Size: 14 KiB |
|
Before Width: | Height: | Size: 11 KiB |
|
Before Width: | Height: | Size: 14 KiB |
@@ -0,0 +1,18 @@
|
||||
dataset,system,n,songbench_global_avg,songbench_musicality,songeval_global_avg,songeval_musicality,audiobox_pq,mulan,allmusiccaps,control_overall,per
|
||||
WSB,YuE2,192,6.731599404761909,5.907463020833336,4.262452916666666,4.2151218749999995,8.25979095697403,0.5068482221880307,0.40536119728009606,4.681896986601725,0.08443630690279864
|
||||
WSB,YuE2 (best-of-8),192,6.963229315476184,6.266602604166664,4.296048333333334,4.250610416666664,8.271365446348986,0.5050805467180908,0.398045457637636,4.70087253751977,0.09786539961704843
|
||||
WSB,ACE-Step 1.5,192,6.011793824404761,5.158814583333332,3.846484166666666,3.805107291666667,8.051833018660545,0.43723272854307044,0.38685302749703016,4.580894501177339,0.07457884998484957
|
||||
WSB,DiffRhythm 2,192,5.242845982142859,4.477539583333331,3.5244677083333333,3.471093229166668,7.978225273390611,0.37817035724098486,0.32549789230688475,4.087044602340498,0.18413265339015286
|
||||
WSB,HeartMuLa,192,6.248347544642855,5.496267708333332,4.551859166666667,4.5329015625,8.293305036922296,0.3823466453177389,0.2786190857962841,3.490655976705673,0.10705881594471056
|
||||
WSB,LeVo 2,192,6.324671800595234,5.4589927083333345,4.023369479166667,3.981881250000001,8.396623315910498,0.3542381529114209,0.2679880544101252,3.9457878565807953,0.26116136186343974
|
||||
WSB,MiniMax Music 2.6,192,6.3221788690476215,5.443674479166666,4.098453333333334,4.055989583333334,8.171058195332686,0.42510899043797207,0.3669951342102043,4.568809841873251,0.24545647955335212
|
||||
WSB,MiniMax Music 3,192,6.282988392857145,5.348229166666669,4.099352187500001,4.0430729166666675,8.28245102117459,0.39279851930526394,0.36085424468425725,4.43616720498247,0.06268414232388418
|
||||
WSB,Mureka 9,192,6.937748809523812,6.048820833333333,4.411131979166667,4.361940104166667,8.022609589000544,0.43942587031051517,0.41017753163275,4.636807564214055,0.11690337887061523
|
||||
WSB,Muse,192,6.034903943452382,5.169220312499998,3.788667708333335,3.736199999999999,8.051745663086573,0.3936510290513979,0.3465821466803997,4.403758449771435,0.33418648580693827
|
||||
WSB,SongBloom,192,4.235043750000001,3.4492635416666655,3.2051130208333345,3.204842187499999,8.153916046023369,0.2697471845106823,0.19264957021005102,3.0286910546158463,0.19193351857701904
|
||||
WSB,Suno v4.5,192,6.699528199404766,5.831729166666669,4.366563958333332,4.319757291666664,8.254062737027803,0.5021767822715143,0.38734816436772235,4.41490242329515,0.05798428222369787
|
||||
WSB,Suno v5,192,6.872074330357141,5.9918354166666665,4.357936874999999,4.305118229166664,8.169838989774385,0.5428044088184834,0.43525151138116297,4.5907186541713685,0.080993611026942
|
||||
WSB,Suno v5.5,192,6.714983407738093,5.808658854166672,4.2151607291666675,4.149727604166669,8.195489337046942,0.5088621161606474,0.3917035845806822,4.591446755445953,0.05958313723720433
|
||||
WSB,Suno v6,192,6.556171130952383,5.6557864583333375,4.308628645833337,4.263533854166667,8.129587392012278,0.4916386870124067,0.4304573973834825,4.625818883206533,0.07580377922906821
|
||||
WSB,Suno v6 Wild,192,6.419507440476191,5.564441145833332,4.219912500000002,4.171640104166664,8.178526304662228,0.49989257991546765,0.43164352465343353,4.589767688464477,0.07454661500473854
|
||||
WSB,YuE 1,192,4.916463244047622,4.084726562499998,3.214967395833334,3.1524088541666675,7.868339642882347,0.262323199029197,0.2881557388876293,3.7300764989993715,0.3637846445448743
|
||||
|
@@ -0,0 +1,316 @@
|
||||
{
|
||||
"snapshot": "2026-09-12",
|
||||
"scope": "15 comparator systems and 2 YuE2 settings on WildSongBench; recorded generation and selection protocols, not matched compute.",
|
||||
"datasets": [
|
||||
{
|
||||
"id": "WSB",
|
||||
"label": "WildSongBench",
|
||||
"prompts": 192
|
||||
}
|
||||
],
|
||||
"metrics": [
|
||||
{
|
||||
"key": "songbench_global_avg",
|
||||
"label": "SongBench",
|
||||
"description": "Seven-dimension global average",
|
||||
"maximum": 10,
|
||||
"lowerIsBetter": false
|
||||
},
|
||||
{
|
||||
"key": "songbench_musicality",
|
||||
"label": "SongBench Musicality",
|
||||
"description": "SongBench musicality dimension",
|
||||
"maximum": 10,
|
||||
"lowerIsBetter": false
|
||||
},
|
||||
{
|
||||
"key": "songeval_global_avg",
|
||||
"label": "SongEval",
|
||||
"description": "Five-dimension global average",
|
||||
"maximum": 5,
|
||||
"lowerIsBetter": false
|
||||
},
|
||||
{
|
||||
"key": "songeval_musicality",
|
||||
"label": "SongEval Musicality",
|
||||
"description": "SongEval musicality dimension",
|
||||
"maximum": 5,
|
||||
"lowerIsBetter": false
|
||||
},
|
||||
{
|
||||
"key": "audiobox_pq",
|
||||
"label": "AudioBox PQ",
|
||||
"description": "Production quality",
|
||||
"maximum": 10,
|
||||
"lowerIsBetter": false
|
||||
},
|
||||
{
|
||||
"key": "mulan",
|
||||
"label": "MuLan",
|
||||
"description": "Style–audio similarity",
|
||||
"maximum": 1,
|
||||
"lowerIsBetter": false
|
||||
},
|
||||
{
|
||||
"key": "allmusiccaps",
|
||||
"label": "AllMusicCaps",
|
||||
"description": "Style–audio similarity",
|
||||
"maximum": 1,
|
||||
"lowerIsBetter": false
|
||||
},
|
||||
{
|
||||
"key": "control_overall",
|
||||
"label": "Q3O Control",
|
||||
"description": "Prompt-weighted overall control",
|
||||
"maximum": 5,
|
||||
"lowerIsBetter": false
|
||||
},
|
||||
{
|
||||
"key": "per",
|
||||
"label": "PER",
|
||||
"description": "Phoneme error rate",
|
||||
"maximum": 1,
|
||||
"lowerIsBetter": true
|
||||
}
|
||||
],
|
||||
"rows": [
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "YuE2",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 6.731599404761909,
|
||||
"songbench_musicality": 5.907463020833336,
|
||||
"songeval_global_avg": 4.262452916666666,
|
||||
"songeval_musicality": 4.2151218749999995,
|
||||
"audiobox_pq": 8.25979095697403,
|
||||
"mulan": 0.5068482221880307,
|
||||
"allmusiccaps": 0.40536119728009606,
|
||||
"control_overall": 4.681896986601725,
|
||||
"per": 0.08443630690279864
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "YuE2 (best-of-8)",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 6.963229315476184,
|
||||
"songbench_musicality": 6.266602604166664,
|
||||
"songeval_global_avg": 4.296048333333334,
|
||||
"songeval_musicality": 4.250610416666664,
|
||||
"audiobox_pq": 8.271365446348986,
|
||||
"mulan": 0.5050805467180908,
|
||||
"allmusiccaps": 0.398045457637636,
|
||||
"control_overall": 4.70087253751977,
|
||||
"per": 0.09786539961704843
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "ACE-Step 1.5",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 6.011793824404761,
|
||||
"songbench_musicality": 5.158814583333332,
|
||||
"songeval_global_avg": 3.846484166666666,
|
||||
"songeval_musicality": 3.805107291666667,
|
||||
"audiobox_pq": 8.051833018660545,
|
||||
"mulan": 0.43723272854307044,
|
||||
"allmusiccaps": 0.38685302749703016,
|
||||
"control_overall": 4.580894501177339,
|
||||
"per": 0.07457884998484957
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "DiffRhythm 2",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 5.242845982142859,
|
||||
"songbench_musicality": 4.477539583333331,
|
||||
"songeval_global_avg": 3.5244677083333333,
|
||||
"songeval_musicality": 3.471093229166668,
|
||||
"audiobox_pq": 7.978225273390611,
|
||||
"mulan": 0.37817035724098486,
|
||||
"allmusiccaps": 0.32549789230688475,
|
||||
"control_overall": 4.087044602340498,
|
||||
"per": 0.18413265339015286
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "HeartMuLa",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 6.248347544642855,
|
||||
"songbench_musicality": 5.496267708333332,
|
||||
"songeval_global_avg": 4.551859166666667,
|
||||
"songeval_musicality": 4.5329015625,
|
||||
"audiobox_pq": 8.293305036922296,
|
||||
"mulan": 0.3823466453177389,
|
||||
"allmusiccaps": 0.2786190857962841,
|
||||
"control_overall": 3.490655976705673,
|
||||
"per": 0.10705881594471056
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "LeVo 2",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 6.324671800595234,
|
||||
"songbench_musicality": 5.4589927083333345,
|
||||
"songeval_global_avg": 4.023369479166667,
|
||||
"songeval_musicality": 3.981881250000001,
|
||||
"audiobox_pq": 8.396623315910498,
|
||||
"mulan": 0.3542381529114209,
|
||||
"allmusiccaps": 0.2679880544101252,
|
||||
"control_overall": 3.9457878565807953,
|
||||
"per": 0.26116136186343974
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "MiniMax Music 2.6",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 6.3221788690476215,
|
||||
"songbench_musicality": 5.443674479166666,
|
||||
"songeval_global_avg": 4.098453333333334,
|
||||
"songeval_musicality": 4.055989583333334,
|
||||
"audiobox_pq": 8.171058195332686,
|
||||
"mulan": 0.42510899043797207,
|
||||
"allmusiccaps": 0.3669951342102043,
|
||||
"control_overall": 4.568809841873251,
|
||||
"per": 0.24545647955335212
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "MiniMax Music 3",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 6.282988392857145,
|
||||
"songbench_musicality": 5.348229166666669,
|
||||
"songeval_global_avg": 4.099352187500001,
|
||||
"songeval_musicality": 4.0430729166666675,
|
||||
"audiobox_pq": 8.28245102117459,
|
||||
"mulan": 0.39279851930526394,
|
||||
"allmusiccaps": 0.36085424468425725,
|
||||
"control_overall": 4.43616720498247,
|
||||
"per": 0.06268414232388418
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "Mureka 9",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 6.937748809523812,
|
||||
"songbench_musicality": 6.048820833333333,
|
||||
"songeval_global_avg": 4.411131979166667,
|
||||
"songeval_musicality": 4.361940104166667,
|
||||
"audiobox_pq": 8.022609589000544,
|
||||
"mulan": 0.43942587031051517,
|
||||
"allmusiccaps": 0.41017753163275,
|
||||
"control_overall": 4.636807564214055,
|
||||
"per": 0.11690337887061523
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "Muse",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 6.034903943452382,
|
||||
"songbench_musicality": 5.169220312499998,
|
||||
"songeval_global_avg": 3.788667708333335,
|
||||
"songeval_musicality": 3.736199999999999,
|
||||
"audiobox_pq": 8.051745663086573,
|
||||
"mulan": 0.3936510290513979,
|
||||
"allmusiccaps": 0.3465821466803997,
|
||||
"control_overall": 4.403758449771435,
|
||||
"per": 0.33418648580693827
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "SongBloom",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 4.235043750000001,
|
||||
"songbench_musicality": 3.4492635416666655,
|
||||
"songeval_global_avg": 3.2051130208333345,
|
||||
"songeval_musicality": 3.204842187499999,
|
||||
"audiobox_pq": 8.153916046023369,
|
||||
"mulan": 0.2697471845106823,
|
||||
"allmusiccaps": 0.19264957021005102,
|
||||
"control_overall": 3.0286910546158463,
|
||||
"per": 0.19193351857701904
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "Suno v4.5",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 6.699528199404766,
|
||||
"songbench_musicality": 5.831729166666669,
|
||||
"songeval_global_avg": 4.366563958333332,
|
||||
"songeval_musicality": 4.319757291666664,
|
||||
"audiobox_pq": 8.254062737027803,
|
||||
"mulan": 0.5021767822715143,
|
||||
"allmusiccaps": 0.38734816436772235,
|
||||
"control_overall": 4.41490242329515,
|
||||
"per": 0.05798428222369787
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "Suno v5",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 6.872074330357141,
|
||||
"songbench_musicality": 5.9918354166666665,
|
||||
"songeval_global_avg": 4.357936874999999,
|
||||
"songeval_musicality": 4.305118229166664,
|
||||
"audiobox_pq": 8.169838989774385,
|
||||
"mulan": 0.5428044088184834,
|
||||
"allmusiccaps": 0.43525151138116297,
|
||||
"control_overall": 4.5907186541713685,
|
||||
"per": 0.080993611026942
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "Suno v5.5",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 6.714983407738093,
|
||||
"songbench_musicality": 5.808658854166672,
|
||||
"songeval_global_avg": 4.2151607291666675,
|
||||
"songeval_musicality": 4.149727604166669,
|
||||
"audiobox_pq": 8.195489337046942,
|
||||
"mulan": 0.5088621161606474,
|
||||
"allmusiccaps": 0.3917035845806822,
|
||||
"control_overall": 4.591446755445953,
|
||||
"per": 0.05958313723720433
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "Suno v6",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 6.556171130952383,
|
||||
"songbench_musicality": 5.6557864583333375,
|
||||
"songeval_global_avg": 4.308628645833337,
|
||||
"songeval_musicality": 4.263533854166667,
|
||||
"audiobox_pq": 8.129587392012278,
|
||||
"mulan": 0.4916386870124067,
|
||||
"allmusiccaps": 0.4304573973834825,
|
||||
"control_overall": 4.625818883206533,
|
||||
"per": 0.07580377922906821
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "Suno v6 Wild",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 6.419507440476191,
|
||||
"songbench_musicality": 5.564441145833332,
|
||||
"songeval_global_avg": 4.219912500000002,
|
||||
"songeval_musicality": 4.171640104166664,
|
||||
"audiobox_pq": 8.178526304662228,
|
||||
"mulan": 0.49989257991546765,
|
||||
"allmusiccaps": 0.43164352465343353,
|
||||
"control_overall": 4.589767688464477,
|
||||
"per": 0.07454661500473854
|
||||
},
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"system": "YuE 1",
|
||||
"n": 192,
|
||||
"songbench_global_avg": 4.916463244047622,
|
||||
"songbench_musicality": 4.084726562499998,
|
||||
"songeval_global_avg": 3.214967395833334,
|
||||
"songeval_musicality": 3.1524088541666675,
|
||||
"audiobox_pq": 7.868339642882347,
|
||||
"mulan": 0.262323199029197,
|
||||
"allmusiccaps": 0.2881557388876293,
|
||||
"control_overall": 3.7300764989993715,
|
||||
"per": 0.3637846445448743
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
# Benchmark results and evaluation scope
|
||||
|
||||
The reported results describe generation with symbolic planning under the recorded automatic-evaluation protocols. They support **frontier quality: YuE2 is competitive with Suno v5/v6**. They do not establish universal metric superiority or human preference.
|
||||
|
||||
The [interactive demo results](https://map-yue2.github.io/#model-overview), [full WSB results CSV](benchmark-results.csv), and [WildSongBench dataset](https://huggingface.co/datasets/m-a-p/WildSongBench) provide the public result and benchmark resources.
|
||||
|
||||
## WildSongBench
|
||||
|
||||
The September 12, 2026 comparison covers **192 prompts and 17 settings**: eight public comparison systems, seven proprietary systems, YuE2, and YuE2 (best-of-8). All displayed aggregate metrics cover the same 192 selected outputs per setting.
|
||||
|
||||
| Setting | SongBench Avg ↑ | SB Musicality ↑ | SongEval Avg ↑ | SE Musicality ↑ | AudioBox PQ ↑ | MuLan ↑ | AllMusicCaps ↑ | Q3O ↑ | PER ↓ |
|
||||
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
||||
| **YuE2 (best-of-8)** | 6.9632 | 6.2666 | 4.2960 | 4.2506 | 8.2714 | 0.5051 | 0.3980 | 4.7009 | 9.79% |
|
||||
| Mureka 9 | 6.9377 | 6.0488 | 4.4111 | 4.3619 | 8.0226 | 0.4394 | 0.4102 | 4.6368 | 11.69% |
|
||||
| Suno v5 | 6.8721 | 5.9918 | 4.3579 | 4.3051 | 8.1698 | 0.5428 | 0.4353 | 4.5907 | 8.10% |
|
||||
| **YuE2** | 6.7316 | 5.9075 | 4.2625 | 4.2151 | 8.2598 | 0.5068 | 0.4054 | 4.6819 | 8.44% |
|
||||
| Suno v5.5 | 6.7150 | 5.8087 | 4.2152 | 4.1497 | 8.1955 | 0.5089 | 0.3917 | 4.5914 | 5.96% |
|
||||
| Suno v4.5 | 6.6995 | 5.8317 | 4.3666 | 4.3198 | 8.2541 | 0.5022 | 0.3873 | 4.4149 | 5.80% |
|
||||
| Suno v6 | 6.5562 | 5.6558 | 4.3086 | 4.2635 | 8.1296 | 0.4916 | 0.4305 | 4.6258 | 7.58% |
|
||||
| Suno v6 Wild | 6.4195 | 5.5644 | 4.2199 | 4.1716 | 8.1785 | 0.4999 | 0.4316 | 4.5898 | 7.45% |
|
||||
| LeVo 2 | 6.3247 | 5.4590 | 4.0234 | 3.9819 | 8.3966 | 0.3542 | 0.2680 | 3.9458 | 26.12% |
|
||||
| MiniMax Music 2.6 | 6.3222 | 5.4437 | 4.0985 | 4.0560 | 8.1711 | 0.4251 | 0.3670 | 4.5688 | 24.55% |
|
||||
| MiniMax Music 3 | 6.2830 | 5.3482 | 4.0994 | 4.0431 | 8.2825 | 0.3928 | 0.3609 | 4.4362 | 6.27% |
|
||||
| HeartMuLa | 6.2483 | 5.4963 | 4.5519 | 4.5329 | 8.2933 | 0.3823 | 0.2786 | 3.4907 | 10.71% |
|
||||
| Muse | 6.0349 | 5.1692 | 3.7887 | 3.7362 | 8.0517 | 0.3937 | 0.3466 | 4.4038 | 33.42% |
|
||||
| ACE-Step 1.5 | 6.0118 | 5.1588 | 3.8465 | 3.8051 | 8.0518 | 0.4372 | 0.3869 | 4.5809 | 7.46% |
|
||||
| DiffRhythm 2 | 5.2428 | 4.4775 | 3.5245 | 3.4711 | 7.9782 | 0.3782 | 0.3255 | 4.0870 | 18.41% |
|
||||
| YuE 1 | 4.9165 | 4.0847 | 3.2150 | 3.1524 | 7.8683 | 0.2623 | 0.2882 | 3.7301 | 36.38% |
|
||||
| SongBloom | 4.2350 | 3.4493 | 3.2051 | 3.2048 | 8.1539 | 0.2697 | 0.1926 | 3.0287 | 19.19% |
|
||||
|
||||
SongBench Avg is the mean of seven dimensions. SongEval Avg is the mean of five dimensions from a separate automatic quality evaluator. SB Musicality and SE Musicality report the respective musicality dimensions. AudioBox PQ measures production quality; MuLan and AllMusicCaps assess text/audio alignment; prompt control uses Q3O on a 0–5 scale. PER is phoneme error rate, for which lower is better.
|
||||
|
||||
**Candidate selection.** Standard YuE2 selects the lower-PER candidate from two generations. Best-of-8 selects by SongBench Musicality, then prompt control, then PER. Each candidate's PER uses the lowest PER from four ASR passes. This selection uses a SongBench dimension and is not equivalent to one unselected pipeline call. Prompt-control weights differ on 10 of 192 prompts between the two selected YuE2 settings.
|
||||
|
||||
Public baselines, Suno v6, and Suno v6 Wild use two candidates and four ASR passes per candidate, followed by lower-PER selection; earlier proprietary systems retain their delivered-candidate protocols. MiniMax Music 3 uses its official caption rewriter and is classified as a public model because its weights are released and the evaluated campaign uses them locally. SongBloom uses a fixed audio prompt. These are documented system comparisons, not matched-compute experiments.
|
||||
|
||||
Both YuE2 settings use melody-and-chord planning and the verified evaluation decoder distributed as **[YuE2-Vae-legacy](https://huggingface.co/m-a-p/YuE2-Vae-legacy)**. Default listening uses **[YuE2-Vae](https://huggingface.co/m-a-p/YuE2-Vae)**. Keep their audio separate; use the same cached acoustic latents for decoder comparisons. The release names determine these roles, not the everyday meaning of “legacy.”
|
||||
|
||||
**What the lead means.** YuE2 (best-of-8) has the highest observed SongBench Avg, 6.9632; the highest proprietary mean is Mureka 9 at 6.9377, while Suno v5 scores 6.8721, Suno v6 scores 6.5562, and Suno v6 Wild scores 6.4195. The small gaps are descriptive, not claims of statistical significance. Different systems lead other metrics: Suno v6 has higher SongEval scores than both YuE2 settings, and both v6 variants have lower PER and higher AllMusicCaps scores. The unqualified YuE2 setting scores 6.7316 and is already within the proprietary quality range.
|
||||
|
||||
The overview figure combines SongBench and SongEval into a normalized song-quality index and MuLan, AllMusicCaps, and prompt control into a normalized text-alignment index. These axes are comparison indices, not percentages of correct output. Bubble area indicates AudioBox PQ; outlined points are Pareto-optimal on the two plotted axes. The 15 external baselines define the z-score reference; quality weights SongBench Avg and SongEval Avg at 2:1, while alignment weights its three metrics equally. Each axis maps the observed extrema across all 17 settings to 10 and 90. [Figure data and normalization](frontier-composites.json) · [Full-precision CSV](benchmark-results.csv) · [JSON](benchmark-results.json).
|
||||
|
||||
## Zero-shot cover generation
|
||||
|
||||
The cover evaluation uses **948 SHS100K works**, two requested styles and two seeds per work: **3,792 outputs per method**, without candidate selection. All YuE2 conditions use the general song-generation checkpoint and the benchmark decoder. The generator received no original–cover paired supervision or cover-specific fine-tuning; exclusion of the evaluated works from generator training was confirmed by the authors. This claim does not assert training-data exclusion for external analyzers, evaluators, or comparison models.
|
||||
|
||||
| Method | CLEWS mAP ↑ | CLEWS Hit@1 ↑ | Discogs-VINet mAP ↑ | MuLan ↑ | SongBench Musicality ↑ |
|
||||
|---|---:|---:|---:|---:|---:|
|
||||
| SongEcho | 0.419 | 48.4% | 0.122 | 0.366 | 3.286 |
|
||||
| ACE-Step 1.5 | 0.024 | 2.4% | 0.006 | 0.166 | 3.689 |
|
||||
| YuE2 (full score) | 0.647 | 71.3% | 0.288 | 0.382 | 5.104 |
|
||||
| YuE2 (without chords) | 0.598 | 67.3% | 0.179 | 0.417 | 5.490 |
|
||||
| YuE2 (without score) | 0.006 | 0.3% | 0.004 | 0.474 | 5.691 |
|
||||
|
||||
CLEWS and Discogs-VINet assess preserved work identity against a source-excluded retrieval gallery of 10,545 recordings per query. MuLan assesses alignment with the requested target style; SongBench measures musicality. Every metric displayed here covers all 3,792 outputs per method. Incomplete Q3O results are omitted.
|
||||
|
||||
The full score best preserves work identity in this comparison. Relaxing the supplied score improves target-style alignment and quality in the current cover configuration. These are distinct outcomes: a fixed transcription from a source performance can constrain adaptation to a contrasting style. This does not show that generating a fresh symbolic plan from the current prompt reduces quality. Product guidance recommends melody-only covers to leave accompaniment freer to adapt.
|
||||
|
||||
## Editing evidence
|
||||
|
||||
A separate paired editing study uses ten original works, two seeds, and 380 full-song recordings. Local changed-note melody attainment increases from 0.0083 to 0.9375; changed-duration harmony attainment increases from 0 to 0.8313. Their units differ and should not be combined into one score. Content outside melody/harmony edits remains close to unedited regeneration on the automatic measurements.
|
||||
|
||||
This is evidence for selective control through the score, not identical waveform preservation. The measurements cover ten works and were developed on that cohort. The public agentic demo illustrates a multi-step workflow; it is not a separate statistically controlled human-preference study.
|
||||
@@ -0,0 +1,73 @@
|
||||
# Cover a recording
|
||||
|
||||
A cover starts with a readable composition: transcribe the source recording, review the melody, then ask YuE2 to realize it in a new style. The general YuE2 checkpoint supports this workflow without cover-specific fine-tuning.
|
||||
|
||||
## 1. Set up SheetSage2 separately
|
||||
|
||||
SheetSage2 and YuE2 use different dependency versions. Keep separate environments and exchange ABC and audio files. Run transcription and generation sequentially so they can share one GPU.
|
||||
|
||||
SheetSage2 uses Python 3.10 or 3.11 and requires FFmpeg 6.1 and its shared libraries. Follow the platform setup in its [model card](https://huggingface.co/m-a-p/SheetSage2), then run from the YuE repository root:
|
||||
|
||||
```bash
|
||||
python3.11 -m venv .venv-sheetsage2
|
||||
.venv-sheetsage2/bin/python -m pip install huggingface-hub==0.36.0
|
||||
.venv-sheetsage2/bin/huggingface-cli download m-a-p/SheetSage2 \
|
||||
--local-dir models/SheetSage2
|
||||
.venv-sheetsage2/bin/python -m pip install \
|
||||
torch==2.8.0 torchaudio==2.8.0 \
|
||||
--index-url https://download.pytorch.org/whl/cu126
|
||||
.venv-sheetsage2/bin/python -m pip install -r models/SheetSage2/requirements.txt
|
||||
```
|
||||
|
||||
Loading SheetSage2 automatically loads the MERT-v2-FullSong encoder selected by its configuration. A separate MERT2 feature-extraction step is unnecessary.
|
||||
|
||||
## 2. Transcribe and review the melody
|
||||
|
||||
```bash
|
||||
.venv-sheetsage2/bin/python models/SheetSage2/infer.py source.wav \
|
||||
--output cover-score --melody-only
|
||||
```
|
||||
|
||||
`cover-score/score.abc` retains vocal and instrumental melodies while omitting chord symbols. Check the command's exit status and transcription warnings, then listen to the source while reviewing notes, meter, and section order. Transcription errors can carry into the cover.
|
||||
|
||||
The same operation is available through Transformers:
|
||||
|
||||
```python
|
||||
from transformers import AutoModel
|
||||
|
||||
model = AutoModel.from_pretrained(
|
||||
"models/SheetSage2", trust_remote_code=True,
|
||||
).eval().to("cuda")
|
||||
result = model.transcribe(
|
||||
"source.wav", output_dir="cover-score", melody_only=True,
|
||||
)
|
||||
if not result.get("abc") or result.get("abc_error"):
|
||||
raise RuntimeError("Transcription did not produce a usable melody score")
|
||||
print(result.get("warnings", []))
|
||||
```
|
||||
|
||||
For repeatable runs, select reviewed model revisions and retain the downloaded configuration. The model's Python implementation is executed by `trust_remote_code=True`.
|
||||
|
||||
## 3. Supply lyrics and a target style
|
||||
|
||||
Prepare `cover-request.json` with `style`, `lyrics`, `cot`, and `seed`, following [the original example](../examples/song.json). Set `cot` to `melody`. Use source lyrics you have available, or transcribe and correct the words; align section tags and lyric order with the score. When translating lyrics, match phrasing and syllable counts to the melody.
|
||||
|
||||
Switch back to the YuE2 environment:
|
||||
|
||||
```bash
|
||||
.venv/bin/python examples/generate.py --request cover-request.json \
|
||||
--abc-file cover-score/score.abc --cot melody --output outputs/cover
|
||||
```
|
||||
|
||||
For a complete score with fixed harmony, use `cot="full"`. Melody-only is recommended for changing styles because it gives the accompaniment more freedom. The [benchmark comparison](benchmarks.md#zero-shot-cover-generation) reports both identity retention and target-style/quality trade-offs.
|
||||
|
||||
## Try the included original melody
|
||||
|
||||
This example requires only YuE2:
|
||||
|
||||
```bash
|
||||
.venv/bin/python examples/generate.py --request examples/song.json \
|
||||
--abc-file examples/melody.abc --cot melody --output outputs/original-melody
|
||||
```
|
||||
|
||||
It tests the score-conditioned interface with original material; it is not a transcription or a benchmark result. The [public cover demos](https://map-yue2.github.io/#cover) illustrate the complete workflow.
|
||||
@@ -0,0 +1,57 @@
|
||||
# Edit with a score and an agent
|
||||
|
||||
YuE2 makes the intended composition available as ABC notation. An agent can change specific notes, chord symbols, tempo, or section order, then use the same generator to render the revised composition. This is the editable interface meant by **white-box music generation**.
|
||||
|
||||
## A complete harmony-edit example
|
||||
|
||||
The original fixtures keep the exact melody, lyric order, meter, tempo, style, and seed fixed. `score-jazz.abc` changes the chord symbols in `score.abc` to seventh chords.
|
||||
|
||||
```bash
|
||||
python skills/yue2-music/scripts/abc_tools.py inspect examples/score.abc
|
||||
python skills/yue2-music/scripts/abc_tools.py inspect examples/score-jazz.abc
|
||||
python skills/yue2-music/scripts/abc_tools.py compare \
|
||||
examples/score.abc examples/score-jazz.abc
|
||||
python examples/generate.py --abc-file examples/score.abc \
|
||||
--cot full --output outputs/harmony-before
|
||||
python examples/generate.py --abc-file examples/score-jazz.abc \
|
||||
--cot full --output outputs/harmony-after
|
||||
```
|
||||
|
||||
The symbolic comparison should report `match: true`: notes, timing, meter, and tempo are unchanged. That check deliberately allows different harmony. Listen to both outputs to judge whether the revised harmony is realized and whether you prefer it.
|
||||
|
||||
## Start from a generated composition
|
||||
|
||||
Generate a song with `save_artifacts()` to retain both its audio and score, or export only the plan:
|
||||
|
||||
```bash
|
||||
python skills/yue2-music/scripts/run_yue2.py plan \
|
||||
--request examples/song.json --output outputs/plan
|
||||
cp outputs/plan/score.abc edited.abc
|
||||
```
|
||||
|
||||
Give the agent the ABC, lyrics, style, and a bounded brief. For example:
|
||||
|
||||
> Reharmonize this song with jazz-influenced chords. Preserve every vocal and instrumental melody note, note duration, bar boundary, section order, and tempo. Keep the lyrics and style unchanged for this comparison. Save a new ABC file and explain the chord changes.
|
||||
|
||||
Then inspect the edited score and verify the requested invariants:
|
||||
|
||||
```bash
|
||||
python skills/yue2-music/scripts/abc_tools.py inspect edited.abc
|
||||
python skills/yue2-music/scripts/abc_tools.py compare outputs/plan/score.abc edited.abc
|
||||
python examples/generate.py --request examples/song.json \
|
||||
--abc-file edited.abc --cot full --output outputs/edited
|
||||
```
|
||||
|
||||
The helper checks a limited native ABC dialect. A rejected score may need notation conversion or correction; do not treat an unsupported construct as a demonstrated musical error. See the [ABC reference](../skills/yue2-music/references/abc-editing.md).
|
||||
|
||||
## Edit lyrics, style, tempo, or form
|
||||
|
||||
Copy the request JSON and update only the intended fields. Align revised lyrics with the melody's phrasing and syllable counts. For section deletion, duplication, or reordering, update both the score and corresponding lyric sections. To change tempo while preserving notes, use the comparison helper's `--allow-tempo-change` option; listen to verify the actual response.
|
||||
|
||||
Editing renders a new complete recording. Matching the unchanged score does not guarantee identical singing, timbre, or waveform outside the edit. Preserve the original and use fresh output directories for every version. Check truncation and compare complete outputs rather than choosing examples solely by the desired result.
|
||||
|
||||
## Use the packaged skill
|
||||
|
||||
Load [yue2-music](../skills/yue2-music/SKILL.md) in an agent supporting `SKILL.md` packages. Its helpers cover generation, transcription, ABC checks, cached decoding, and listening comparisons. The skill guides the agent between existing APIs; no separate agentic-generation checkpoint is required.
|
||||
|
||||
[Explore the 9-step, 14-version editing demo](https://map-yue2.github.io/#agentic-music-editing), where each version includes its audio, score, prompts, lyrics, and conversation.
|
||||
@@ -0,0 +1,639 @@
|
||||
{
|
||||
"dataset": "WSB",
|
||||
"n_prompts": 192,
|
||||
"source_sha256": "579ffb6b9b069f6d3cdd32a76788998ec633a12426a0ebb7c9f83e6fb0d5b1fc",
|
||||
"reference_systems": [
|
||||
"Suno v5",
|
||||
"Suno v4.5",
|
||||
"Suno v5.5",
|
||||
"Suno v6",
|
||||
"Suno v6 Wild",
|
||||
"MiniMax Music 2.6",
|
||||
"Mureka 9",
|
||||
"YuE 1",
|
||||
"SongBloom",
|
||||
"LeVo 2",
|
||||
"ACE-Step 1.5",
|
||||
"HeartMuLa",
|
||||
"DiffRhythm 2",
|
||||
"Muse",
|
||||
"MiniMax Music 3"
|
||||
],
|
||||
"normalization": "z = (system mean - reference mean) / reference population std",
|
||||
"reference_statistics": {
|
||||
"songbench_global_avg": {
|
||||
"mean": 6.121283377976191,
|
||||
"population_std": 0.7357457614725659
|
||||
},
|
||||
"songeval_global_avg": {
|
||||
"mean": 4.015471256944445,
|
||||
"population_std": 0.40701956006795303
|
||||
},
|
||||
"songbench_musicality": {
|
||||
"mean": 5.265866701388889,
|
||||
"population_std": 0.7053243399537521
|
||||
},
|
||||
"songeval_musicality": {
|
||||
"mean": 3.9703476041666663,
|
||||
"population_std": 0.4051162346821886
|
||||
},
|
||||
"mulan": {
|
||||
"mean": 0.4186944833891175,
|
||||
"population_std": 0.08094623772340859
|
||||
},
|
||||
"allmusiccaps": {
|
||||
"mean": 0.3533851072454733,
|
||||
"population_std": 0.06820180734302435
|
||||
},
|
||||
"control_overall": {
|
||||
"mean": 4.248089863722948,
|
||||
"population_std": 0.4756164261685128
|
||||
}
|
||||
},
|
||||
"quality_weights": {
|
||||
"songbench": 0.6666666666666666,
|
||||
"songeval": 0.33333333333333337
|
||||
},
|
||||
"alignment_weights": {
|
||||
"mulan": 0.3333333333333333,
|
||||
"allmusiccaps": 0.3333333333333333,
|
||||
"control_overall": 0.3333333333333333
|
||||
},
|
||||
"values": {
|
||||
"YuE2": {
|
||||
"quality_global_avg": 0.7552819777715396,
|
||||
"quality_musicality": 0.8078339775660601,
|
||||
"alignment": 0.9210758825216073,
|
||||
"audiobox_pq": 8.25979095697403
|
||||
},
|
||||
"YuE2 (best-of-8)": {
|
||||
"quality_global_avg": 0.9926775311951548,
|
||||
"quality_musicality": 1.1764900036325572,
|
||||
"alignment": 0.8913402282808791,
|
||||
"audiobox_pq": 8.271365446348986
|
||||
},
|
||||
"ACE-Step 1.5": {
|
||||
"quality_global_avg": -0.2376035047550664,
|
||||
"quality_musicality": -0.23714600301672967,
|
||||
"alignment": 0.47315715541685155,
|
||||
"audiobox_pq": 8.051833018660545
|
||||
},
|
||||
"DiffRhythm 2": {
|
||||
"quality_global_avg": -1.1980739812149017,
|
||||
"quality_musicality": -1.1559112521857333,
|
||||
"alignment": -0.41604199452568785,
|
||||
"audiobox_pq": 7.978225273390611
|
||||
},
|
||||
"HeartMuLa": {
|
||||
"quality_global_avg": 0.5544151508643054,
|
||||
"quality_musicality": 0.6806476331364909,
|
||||
"alignment": -1.0459382238546173,
|
||||
"audiobox_pq": 8.293305036922296
|
||||
},
|
||||
"LeVo 2": {
|
||||
"quality_global_avg": 0.19076064388585245,
|
||||
"quality_musicality": 0.19203107542441739,
|
||||
"alignment": -0.8946697129809009,
|
||||
"audiobox_pq": 8.396623315910498
|
||||
},
|
||||
"MiniMax Music 2.6": {
|
||||
"quality_global_avg": 0.24999255629041012,
|
||||
"quality_musicality": 0.2385294260172236,
|
||||
"alignment": 0.317708040770442,
|
||||
"audiobox_pq": 8.171058195332686
|
||||
},
|
||||
"MiniMax Music 3": {
|
||||
"quality_global_avg": 0.21521779684943682,
|
||||
"quality_musicality": 0.13768736380561403,
|
||||
"alignment": 0.06167958802908772,
|
||||
"audiobox_pq": 8.28245102117459
|
||||
},
|
||||
"Mureka 9": {
|
||||
"quality_global_avg": 1.063838458020569,
|
||||
"quality_musicality": 1.0622475778213865,
|
||||
"alignment": 0.6353722959962361,
|
||||
"audiobox_pq": 8.022609589000544
|
||||
},
|
||||
"Muse": {
|
||||
"quality_global_avg": -0.2640126435903063,
|
||||
"quality_musicality": -0.28400814299371513,
|
||||
"alignment": -0.027277572405822206,
|
||||
"audiobox_pq": 8.051745663086573
|
||||
},
|
||||
"SongBloom": {
|
||||
"quality_global_avg": -2.3727929454241776,
|
||||
"quality_musicality": -2.3469029518895956,
|
||||
"alignment": -2.25355622224482,
|
||||
"audiobox_pq": 8.153916046023369
|
||||
},
|
||||
"Suno v4.5": {
|
||||
"quality_global_avg": 0.8114848653568392,
|
||||
"quality_musicality": 0.822345947368047,
|
||||
"alignment": 0.6266794053102317,
|
||||
"audiobox_pq": 8.254062737027803
|
||||
},
|
||||
"Suno v5": {
|
||||
"quality_global_avg": 0.9607654088358881,
|
||||
"quality_musicality": 0.9616318815497118,
|
||||
"alignment": 1.1513277335650094,
|
||||
"audiobox_pq": 8.169838989774385
|
||||
},
|
||||
"Suno v5.5": {
|
||||
"quality_global_avg": 0.7014955760788759,
|
||||
"quality_musicality": 0.6606381030854666,
|
||||
"alignment": 0.7992264544775121,
|
||||
"audiobox_pq": 8.195489337046942
|
||||
},
|
||||
"Suno v6": {
|
||||
"quality_global_avg": 0.6341407893639396,
|
||||
"quality_musicality": 0.6097852138267854,
|
||||
"alignment": 0.9417981546473502,
|
||||
"audiobox_pq": 8.129587392012278
|
||||
},
|
||||
"Suno v6 Wild": {
|
||||
"quality_global_avg": 0.43765333454931554,
|
||||
"quality_musicality": 0.44783537273294216,
|
||||
"alignment": 0.9563182017109547,
|
||||
"audiobox_pq": 8.178526304662228
|
||||
},
|
||||
"YuE 1": {
|
||||
"quality_global_avg": -1.7472815051109758,
|
||||
"quality_musicality": -1.7894112446823107,
|
||||
"alignment": -1.3257833039118254,
|
||||
"audiobox_pq": 7.868339642882347
|
||||
}
|
||||
},
|
||||
"sensitivity": [
|
||||
{
|
||||
"variant": "global_avg",
|
||||
"songbench_weight": 0.5,
|
||||
"best_public": "HeartMuLa",
|
||||
"yue2_minus_best_public": -0.02710930368588904,
|
||||
"rank_yue2": 6,
|
||||
"rank_bo8": 3,
|
||||
"ranking": [
|
||||
"Mureka 9",
|
||||
"Suno v5",
|
||||
"YuE2 (best-of-8)",
|
||||
"Suno v4.5",
|
||||
"HeartMuLa",
|
||||
"YuE2",
|
||||
"Suno v6",
|
||||
"Suno v5.5",
|
||||
"Suno v6 Wild",
|
||||
"MiniMax Music 2.6",
|
||||
"MiniMax Music 3",
|
||||
"LeVo 2",
|
||||
"ACE-Step 1.5",
|
||||
"Muse",
|
||||
"DiffRhythm 2",
|
||||
"YuE 1",
|
||||
"SongBloom"
|
||||
]
|
||||
},
|
||||
{
|
||||
"variant": "global_avg",
|
||||
"songbench_weight": 0.6,
|
||||
"best_public": "HeartMuLa",
|
||||
"yue2_minus_best_public": 0.10967637466998481,
|
||||
"rank_yue2": 5,
|
||||
"rank_bo8": 2,
|
||||
"ranking": [
|
||||
"Mureka 9",
|
||||
"YuE2 (best-of-8)",
|
||||
"Suno v5",
|
||||
"Suno v4.5",
|
||||
"YuE2",
|
||||
"Suno v5.5",
|
||||
"Suno v6",
|
||||
"HeartMuLa",
|
||||
"Suno v6 Wild",
|
||||
"MiniMax Music 2.6",
|
||||
"MiniMax Music 3",
|
||||
"LeVo 2",
|
||||
"ACE-Step 1.5",
|
||||
"Muse",
|
||||
"DiffRhythm 2",
|
||||
"YuE 1",
|
||||
"SongBloom"
|
||||
]
|
||||
},
|
||||
{
|
||||
"variant": "global_avg",
|
||||
"songbench_weight": 0.6666666666666666,
|
||||
"best_public": "HeartMuLa",
|
||||
"yue2_minus_best_public": 0.20086682690723423,
|
||||
"rank_yue2": 5,
|
||||
"rank_bo8": 2,
|
||||
"ranking": [
|
||||
"Mureka 9",
|
||||
"YuE2 (best-of-8)",
|
||||
"Suno v5",
|
||||
"Suno v4.5",
|
||||
"YuE2",
|
||||
"Suno v5.5",
|
||||
"Suno v6",
|
||||
"HeartMuLa",
|
||||
"Suno v6 Wild",
|
||||
"MiniMax Music 2.6",
|
||||
"MiniMax Music 3",
|
||||
"LeVo 2",
|
||||
"ACE-Step 1.5",
|
||||
"Muse",
|
||||
"DiffRhythm 2",
|
||||
"YuE 1",
|
||||
"SongBloom"
|
||||
]
|
||||
},
|
||||
{
|
||||
"variant": "global_avg",
|
||||
"songbench_weight": 0.7,
|
||||
"best_public": "HeartMuLa",
|
||||
"yue2_minus_best_public": 0.24646205302585877,
|
||||
"rank_yue2": 5,
|
||||
"rank_bo8": 2,
|
||||
"ranking": [
|
||||
"Mureka 9",
|
||||
"YuE2 (best-of-8)",
|
||||
"Suno v5",
|
||||
"Suno v4.5",
|
||||
"YuE2",
|
||||
"Suno v5.5",
|
||||
"Suno v6",
|
||||
"HeartMuLa",
|
||||
"Suno v6 Wild",
|
||||
"MiniMax Music 2.6",
|
||||
"MiniMax Music 3",
|
||||
"LeVo 2",
|
||||
"ACE-Step 1.5",
|
||||
"Muse",
|
||||
"DiffRhythm 2",
|
||||
"YuE 1",
|
||||
"SongBloom"
|
||||
]
|
||||
},
|
||||
{
|
||||
"variant": "global_avg",
|
||||
"songbench_weight": 0.75,
|
||||
"best_public": "HeartMuLa",
|
||||
"yue2_minus_best_public": 0.3148548922037958,
|
||||
"rank_yue2": 5,
|
||||
"rank_bo8": 2,
|
||||
"ranking": [
|
||||
"Mureka 9",
|
||||
"YuE2 (best-of-8)",
|
||||
"Suno v5",
|
||||
"Suno v4.5",
|
||||
"YuE2",
|
||||
"Suno v5.5",
|
||||
"Suno v6",
|
||||
"HeartMuLa",
|
||||
"Suno v6 Wild",
|
||||
"MiniMax Music 2.6",
|
||||
"MiniMax Music 3",
|
||||
"LeVo 2",
|
||||
"ACE-Step 1.5",
|
||||
"Muse",
|
||||
"DiffRhythm 2",
|
||||
"YuE 1",
|
||||
"SongBloom"
|
||||
]
|
||||
},
|
||||
{
|
||||
"variant": "global_avg",
|
||||
"songbench_weight": 0.8,
|
||||
"best_public": "HeartMuLa",
|
||||
"yue2_minus_best_public": 0.3832477313817329,
|
||||
"rank_yue2": 5,
|
||||
"rank_bo8": 2,
|
||||
"ranking": [
|
||||
"Mureka 9",
|
||||
"YuE2 (best-of-8)",
|
||||
"Suno v5",
|
||||
"Suno v4.5",
|
||||
"YuE2",
|
||||
"Suno v5.5",
|
||||
"Suno v6",
|
||||
"Suno v6 Wild",
|
||||
"HeartMuLa",
|
||||
"MiniMax Music 2.6",
|
||||
"LeVo 2",
|
||||
"MiniMax Music 3",
|
||||
"ACE-Step 1.5",
|
||||
"Muse",
|
||||
"DiffRhythm 2",
|
||||
"YuE 1",
|
||||
"SongBloom"
|
||||
]
|
||||
},
|
||||
{
|
||||
"variant": "musicality",
|
||||
"songbench_weight": 0.5,
|
||||
"best_public": "HeartMuLa",
|
||||
"yue2_minus_best_public": -0.10071426090411806,
|
||||
"rank_yue2": 6,
|
||||
"rank_bo8": 1,
|
||||
"ranking": [
|
||||
"YuE2 (best-of-8)",
|
||||
"Mureka 9",
|
||||
"Suno v5",
|
||||
"HeartMuLa",
|
||||
"Suno v4.5",
|
||||
"YuE2",
|
||||
"Suno v6",
|
||||
"Suno v5.5",
|
||||
"Suno v6 Wild",
|
||||
"MiniMax Music 2.6",
|
||||
"LeVo 2",
|
||||
"MiniMax Music 3",
|
||||
"ACE-Step 1.5",
|
||||
"Muse",
|
||||
"DiffRhythm 2",
|
||||
"YuE 1",
|
||||
"SongBloom"
|
||||
]
|
||||
},
|
||||
{
|
||||
"variant": "musicality",
|
||||
"songbench_weight": 0.6,
|
||||
"best_public": "HeartMuLa",
|
||||
"yue2_minus_best_public": 0.036026102296094376,
|
||||
"rank_yue2": 5,
|
||||
"rank_bo8": 1,
|
||||
"ranking": [
|
||||
"YuE2 (best-of-8)",
|
||||
"Mureka 9",
|
||||
"Suno v5",
|
||||
"Suno v4.5",
|
||||
"YuE2",
|
||||
"HeartMuLa",
|
||||
"Suno v5.5",
|
||||
"Suno v6",
|
||||
"Suno v6 Wild",
|
||||
"MiniMax Music 2.6",
|
||||
"LeVo 2",
|
||||
"MiniMax Music 3",
|
||||
"ACE-Step 1.5",
|
||||
"Muse",
|
||||
"DiffRhythm 2",
|
||||
"YuE 1",
|
||||
"SongBloom"
|
||||
]
|
||||
},
|
||||
{
|
||||
"variant": "musicality",
|
||||
"songbench_weight": 0.6666666666666666,
|
||||
"best_public": "HeartMuLa",
|
||||
"yue2_minus_best_public": 0.12718634442956922,
|
||||
"rank_yue2": 5,
|
||||
"rank_bo8": 1,
|
||||
"ranking": [
|
||||
"YuE2 (best-of-8)",
|
||||
"Mureka 9",
|
||||
"Suno v5",
|
||||
"Suno v4.5",
|
||||
"YuE2",
|
||||
"HeartMuLa",
|
||||
"Suno v5.5",
|
||||
"Suno v6",
|
||||
"Suno v6 Wild",
|
||||
"MiniMax Music 2.6",
|
||||
"LeVo 2",
|
||||
"MiniMax Music 3",
|
||||
"ACE-Step 1.5",
|
||||
"Muse",
|
||||
"DiffRhythm 2",
|
||||
"YuE 1",
|
||||
"SongBloom"
|
||||
]
|
||||
},
|
||||
{
|
||||
"variant": "musicality",
|
||||
"songbench_weight": 0.7,
|
||||
"best_public": "HeartMuLa",
|
||||
"yue2_minus_best_public": 0.1727664654963067,
|
||||
"rank_yue2": 5,
|
||||
"rank_bo8": 1,
|
||||
"ranking": [
|
||||
"YuE2 (best-of-8)",
|
||||
"Mureka 9",
|
||||
"Suno v5",
|
||||
"Suno v4.5",
|
||||
"YuE2",
|
||||
"Suno v5.5",
|
||||
"HeartMuLa",
|
||||
"Suno v6",
|
||||
"Suno v6 Wild",
|
||||
"MiniMax Music 2.6",
|
||||
"LeVo 2",
|
||||
"MiniMax Music 3",
|
||||
"ACE-Step 1.5",
|
||||
"Muse",
|
||||
"DiffRhythm 2",
|
||||
"YuE 1",
|
||||
"SongBloom"
|
||||
]
|
||||
},
|
||||
{
|
||||
"variant": "musicality",
|
||||
"songbench_weight": 0.75,
|
||||
"best_public": "HeartMuLa",
|
||||
"yue2_minus_best_public": 0.24113664709641314,
|
||||
"rank_yue2": 4,
|
||||
"rank_bo8": 1,
|
||||
"ranking": [
|
||||
"YuE2 (best-of-8)",
|
||||
"Mureka 9",
|
||||
"Suno v5",
|
||||
"YuE2",
|
||||
"Suno v4.5",
|
||||
"Suno v5.5",
|
||||
"Suno v6",
|
||||
"HeartMuLa",
|
||||
"Suno v6 Wild",
|
||||
"MiniMax Music 2.6",
|
||||
"LeVo 2",
|
||||
"MiniMax Music 3",
|
||||
"ACE-Step 1.5",
|
||||
"Muse",
|
||||
"DiffRhythm 2",
|
||||
"YuE 1",
|
||||
"SongBloom"
|
||||
]
|
||||
},
|
||||
{
|
||||
"variant": "musicality",
|
||||
"songbench_weight": 0.8,
|
||||
"best_public": "HeartMuLa",
|
||||
"yue2_minus_best_public": 0.30950682869651935,
|
||||
"rank_yue2": 4,
|
||||
"rank_bo8": 1,
|
||||
"ranking": [
|
||||
"YuE2 (best-of-8)",
|
||||
"Mureka 9",
|
||||
"Suno v5",
|
||||
"YuE2",
|
||||
"Suno v4.5",
|
||||
"Suno v5.5",
|
||||
"Suno v6",
|
||||
"HeartMuLa",
|
||||
"Suno v6 Wild",
|
||||
"MiniMax Music 2.6",
|
||||
"LeVo 2",
|
||||
"MiniMax Music 3",
|
||||
"ACE-Step 1.5",
|
||||
"Muse",
|
||||
"DiffRhythm 2",
|
||||
"YuE 1",
|
||||
"SongBloom"
|
||||
]
|
||||
}
|
||||
],
|
||||
"display_normalization": {
|
||||
"formula": "index = 10 + 80 * (composite - min) / (max - min)",
|
||||
"display_range": [
|
||||
10,
|
||||
90
|
||||
],
|
||||
"axis_range": [
|
||||
0,
|
||||
100
|
||||
],
|
||||
"endpoint_systems": [
|
||||
"YuE2",
|
||||
"YuE2 (best-of-8)",
|
||||
"ACE-Step 1.5",
|
||||
"DiffRhythm 2",
|
||||
"HeartMuLa",
|
||||
"LeVo 2",
|
||||
"MiniMax Music 2.6",
|
||||
"MiniMax Music 3",
|
||||
"Mureka 9",
|
||||
"Muse",
|
||||
"SongBloom",
|
||||
"Suno v4.5",
|
||||
"Suno v5",
|
||||
"Suno v5.5",
|
||||
"Suno v6",
|
||||
"Suno v6 Wild",
|
||||
"YuE 1"
|
||||
],
|
||||
"bounds": {
|
||||
"quality_global_avg": {
|
||||
"min": -2.3727929454241776,
|
||||
"max": 1.063838458020569
|
||||
},
|
||||
"quality_musicality": {
|
||||
"min": -2.3469029518895956,
|
||||
"max": 1.1764900036325572
|
||||
},
|
||||
"alignment": {
|
||||
"min": -2.25355622224482,
|
||||
"max": 1.1513277335650094
|
||||
}
|
||||
},
|
||||
"scope": "Observed extrema across the frozen 17-setting WSB comparison."
|
||||
},
|
||||
"display_values": {
|
||||
"YuE2": {
|
||||
"quality_global_avg": 82.81723422675485,
|
||||
"quality_musicality": 81.6295223219151,
|
||||
"alignment": 84.59008050713697
|
||||
},
|
||||
"YuE2 (best-of-8)": {
|
||||
"quality_global_avg": 88.34347258180588,
|
||||
"quality_musicality": 90.0,
|
||||
"alignment": 83.89142164823544
|
||||
},
|
||||
"ACE-Step 1.5": {
|
||||
"quality_global_avg": 59.70424092683039,
|
||||
"quality_musicality": 57.90284763591369,
|
||||
"alignment": 74.06593383035025
|
||||
},
|
||||
"DiffRhythm 2": {
|
||||
"quality_global_avg": 37.34582389095981,
|
||||
"quality_musicality": 37.04192724997628,
|
||||
"alignment": 53.17361182506653
|
||||
},
|
||||
"HeartMuLa": {
|
||||
"quality_global_avg": 78.1413338271742,
|
||||
"quality_musicality": 78.7417071724812,
|
||||
"alignment": 38.373783402036175
|
||||
},
|
||||
"LeVo 2": {
|
||||
"quality_global_avg": 69.67596261246808,
|
||||
"quality_musicality": 67.6474792193652,
|
||||
"alignment": 41.9279370903721
|
||||
},
|
||||
"MiniMax Music 2.6": {
|
||||
"quality_global_avg": 71.0547991637534,
|
||||
"quality_musicality": 68.70324225641005,
|
||||
"alignment": 70.41355409197678
|
||||
},
|
||||
"MiniMax Music 3": {
|
||||
"quality_global_avg": 70.24529112268468,
|
||||
"quality_musicality": 66.41358422542463,
|
||||
"alignment": 64.3979962976035
|
||||
},
|
||||
"Mureka 9": {
|
||||
"quality_global_avg": 90.0,
|
||||
"quality_musicality": 87.40608152985898,
|
||||
"alignment": 77.87728582201137
|
||||
},
|
||||
"Muse": {
|
||||
"quality_global_avg": 59.08947290000577,
|
||||
"quality_musicality": 56.83882462017735,
|
||||
"alignment": 62.30788899081859
|
||||
},
|
||||
"SongBloom": {
|
||||
"quality_global_avg": 10.0,
|
||||
"quality_musicality": 10.0,
|
||||
"alignment": 10.0
|
||||
},
|
||||
"Suno v4.5": {
|
||||
"quality_global_avg": 84.12555929249136,
|
||||
"quality_musicality": 81.95902220989649,
|
||||
"alignment": 77.67304060722402
|
||||
},
|
||||
"Suno v5": {
|
||||
"quality_global_avg": 87.60060275113905,
|
||||
"quality_musicality": 85.12156322510432,
|
||||
"alignment": 90.0
|
||||
},
|
||||
"Suno v5.5": {
|
||||
"quality_global_avg": 81.56516159216855,
|
||||
"quality_musicality": 78.28738305243859,
|
||||
"alignment": 81.72714762306776
|
||||
},
|
||||
"Suno v6": {
|
||||
"quality_global_avg": 79.9972358228254,
|
||||
"quality_musicality": 77.13274853053025,
|
||||
"alignment": 85.0769639932043
|
||||
},
|
||||
"Suno v6 Wild": {
|
||||
"quality_global_avg": 75.42328111548792,
|
||||
"quality_musicality": 73.45561474186167,
|
||||
"alignment": 85.41812210025412
|
||||
},
|
||||
"YuE 1": {
|
||||
"quality_global_avg": 24.561036477434577,
|
||||
"quality_musicality": 22.65806486519848,
|
||||
"alignment": 31.798638200280863
|
||||
}
|
||||
},
|
||||
"per_song_validation": {
|
||||
"systems": 17,
|
||||
"records": 3264,
|
||||
"absolute_tolerance": 1e-10
|
||||
},
|
||||
"interpretation": [
|
||||
"Descriptive standardized composites, not percentages or validated new evaluators.",
|
||||
"Observed extrema map to 10 and 90 on 0–100 axes, leaving symmetric display margins.",
|
||||
"All 15 external baselines, including HeartMuLa and both Suno v6 variants, define one frozen WSB reference.",
|
||||
"YuE2 and Bo8 are excluded from the 15-system z-score standardization reference.",
|
||||
"Q3O retains recorded prompt weights, which differ on some selected records.",
|
||||
"Evaluator reuse and candidate-selection protocols remain as documented in Tables 1–2."
|
||||
],
|
||||
"source": "benchmark-results.csv",
|
||||
"snapshot": "2026-09-12"
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
# Generate songs with YuE2
|
||||
|
||||
Install from the repository root with Python 3.12 and `python -m pip install .`. The supported starting point is a BF16-capable NVIDIA GPU with 24 GB VRAM, one request at a time. The default output is 48 kHz stereo with full symbolic planning and the listening decoder, [YuE2-Vae](https://huggingface.co/m-a-p/YuE2-Vae).
|
||||
|
||||
## Style, lyrics, and planning
|
||||
|
||||
Put genre, instruments, vocal character, language, and tempo in `style`. Put the words to sing in `lyrics`, with section tags such as `[Verse]` and `[Chorus]`. Start with the [original request](../examples/song.json).
|
||||
|
||||
```python
|
||||
import json
|
||||
from pathlib import Path
|
||||
from yue2 import YuE2Pipeline
|
||||
|
||||
request = json.loads(Path("examples/song.json").read_text(encoding="utf-8"))
|
||||
with YuE2Pipeline.from_pretrained("m-a-p/YuE2-3B", device="cuda") as pipe:
|
||||
song = pipe(**request)
|
||||
song.save_artifacts("outputs/song")
|
||||
print(song.truncated)
|
||||
```
|
||||
|
||||
`full` plans melody and chords, `melody` plans only melody, and `off` generates without a symbolic plan. Supplying `abc` uses that composition directly; it requires `full` or `melody`. The native melody input has `Vocal` and `Ins` voices and no chord symbols. Arbitrary ABC dialects may need conversion; the [skill's ABC reference](../skills/yue2-music/references/abc-editing.md) describes the supported notation.
|
||||
|
||||
One pipeline call produces one candidate. The benchmark's candidate selection is a separate evaluation step. `cfg_scale` controls text guidance; defaults are ready to use, while changing sampling or guidance may change quality.
|
||||
|
||||
The command-line equivalent is:
|
||||
|
||||
```bash
|
||||
yue2 generate --request examples/song.json --output outputs/song-cli
|
||||
```
|
||||
|
||||
## Save and inspect a plan before synthesis
|
||||
|
||||
```python
|
||||
import json
|
||||
from pathlib import Path
|
||||
from yue2 import YuE2Pipeline, SymbolicPlan
|
||||
|
||||
request = json.loads(Path("examples/song.json").read_text(encoding="utf-8"))
|
||||
with YuE2Pipeline.from_pretrained("m-a-p/YuE2-3B", device="cuda") as pipe:
|
||||
plan = pipe.plan(**request)
|
||||
plan.save("outputs/plan")
|
||||
restored = SymbolicPlan.load("outputs/plan")
|
||||
semantic = pipe.generate_semantic(restored)
|
||||
latents = pipe.synthesize(semantic)
|
||||
audio = pipe.decode(latents)
|
||||
|
||||
import soundfile as sf
|
||||
sf.write("outputs/plan/audio.flac", audio, 48000)
|
||||
```
|
||||
|
||||
`SymbolicPlan.load` restores exact token IDs and checks the saved files. Use it for an unchanged plan. To change the composition, copy the ABC, edit the copy, and submit it as a new `abc` input; do not change a saved plan in place. The end-to-end `save_artifacts()` path is the simplest way to retain a complete generation record.
|
||||
|
||||
## Outputs and reproducibility
|
||||
|
||||
`save_artifacts()` retains `audio.flac`, `score.abc` when applicable, `plan.json`, semantic tokens, `latent.npy`, effective configuration, timings, model identities, and integrity records. Inspect `result.json` and its `truncated` flags. Saved audio can be playable even when the model hit a token limit. Use a fresh directory for each changed request and retain failures in comparisons.
|
||||
|
||||
`from_pretrained` accepts local model directories, `revision`, `vae_revision`, `cache_dir`, and `local_files_only=True`. Pin model and VAE revisions for comparisons. Separate GPUs, runtime versions, or sampling settings can change a seeded generation.
|
||||
|
||||
## Listening and evaluation decoders
|
||||
|
||||
| Decoder | Use |
|
||||
|---|---|
|
||||
| [YuE2-Vae](https://huggingface.co/m-a-p/YuE2-Vae) | Default generation and listening |
|
||||
| [YuE2-Vae-legacy](https://huggingface.co/m-a-p/YuE2-Vae-legacy) | Reproducing the recorded benchmark protocol |
|
||||
|
||||
Choose explicitly with `YuE2Pipeline.from_pretrained("m-a-p/YuE2-3B", vae="m-a-p/YuE2-Vae-legacy")`. Keep listening and evaluation audio in separate directories. When comparing decoders, decode the same cached latents instead of generating a new song:
|
||||
|
||||
```bash
|
||||
python skills/yue2-music/scripts/run_yue2.py decode \
|
||||
--source outputs/song --output outputs/song-benchmark \
|
||||
--vae m-a-p/YuE2-Vae-legacy
|
||||
```
|
||||
|
||||
The helper verifies the source artifacts and preserves the latent identity. Benchmark claims refer to the stated evaluation model and decoder; a new local generation is not itself a reproduction of the published aggregate.
|
||||
@@ -0,0 +1,18 @@
|
||||
# Public documentation sources
|
||||
|
||||
The README's frontier figure and architecture figure are reproduced from the [YuE2 demo page](https://map-yue2.github.io/#model-overview). The logo is the original project's logo, preserved from [YuE-v1](https://github.com/multimodal-art-projection/YuE/tree/YuE-v1/assets/logo). PNG text, timestamp, and EXIF metadata were removed where present; image pixels were preserved. These figures and original example inputs are documentation assets, not generated evaluation audio.
|
||||
|
||||
The institution panels reuse the eight marks and their order from the [demo page](https://map-yue2.github.io/), with layouts for desktop and mobile. The SVGs contain the original vector artwork and embedded raster artwork; EXIF metadata was removed without recompressing image pixels. Institution names and marks belong to their respective organizations.
|
||||
|
||||
The [WSB CSV](benchmark-results.csv) is the September 12, 2026 result set displayed on the demo page. The README presents all 17 settings for four metrics; the CSV retains the complete metric set. Benchmark scope and selection are recorded in [benchmarks.md](benchmarks.md).
|
||||
|
||||
## Citation verification
|
||||
|
||||
The two README BibTeX entries use arXiv versions consistently. Titles, full ordered author lists, identifiers, and first-submission years were manually checked against the primary records on September 10, 2026:
|
||||
|
||||
| Work | Primary record | Selected version metadata |
|
||||
|---|---|---|
|
||||
| MERT | [arXiv:2306.00107](https://arxiv.org/abs/2306.00107) | 20 authors; first submitted May 31, 2023; current record v5 |
|
||||
| YuE | [arXiv:2503.08638](https://arxiv.org/abs/2503.08638) | 58 authors; first submitted March 11, 2025; current record v2 |
|
||||
|
||||
MERT was also accepted at ICLR 2024. The README deliberately cites the arXiv version with year 2023, rather than mixing the conference year with preprint metadata. These entries cite the preceding MERT and YuE papers; they are not presented as a published YuE2 paper.
|
||||
|
Before Width: | Height: | Size: 424 KiB |
|
After Width: | Height: | Size: 292 KiB |
@@ -0,0 +1,485 @@
|
||||
{
|
||||
"id": "7c9d78d2-6b53-43f8-9173-4b6748d1af04",
|
||||
"revision": 0,
|
||||
"last_node_id": 22,
|
||||
"last_link_id": 24,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 4,
|
||||
"type": "SaveAudioAdvanced",
|
||||
"pos": [
|
||||
1696.2669811564128,
|
||||
5221.529241119489
|
||||
],
|
||||
"size": [
|
||||
414.5625,
|
||||
165.984375
|
||||
],
|
||||
"flags": {},
|
||||
"order": 10,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"link": 24
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"audio/yue2",
|
||||
"flac"
|
||||
],
|
||||
"widgets_values_named": {
|
||||
"filename_prefix": "audio/yue2",
|
||||
"format": "flac"
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 9,
|
||||
"type": "Note",
|
||||
"pos": [
|
||||
2211.75764837062,
|
||||
4732.53074774176
|
||||
],
|
||||
"size": [
|
||||
391.3125,
|
||||
1272.1875
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"[Intro]\n\n[Verse1]\n挡风玻璃 凝结 冰霜视线\n钢铁血管 泵送 液压熔流\n方向盘 是 锻打的 钢铁誓约\n涡轮增压 撕裂 寒夜 的 氧\n导航删除 坐标 植入 芯片\n柏油路 震颤 传送带 的 脉动\n后视镜里 昨日 碾成 铁屑\n这 电流脉冲 即 我的 通行 令\n\n[Pre-Chorus]\n音量阀 旋至 红色 警戒区\n底鼓 如 锻锤 砸击 脊髓 Yeah\n肾上腺素 是 劣质 硝化燃料\n爆\n\n[Chorus]\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 生锈铁板\n让 电极火花 焚烧 这 牢笼 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 铁渣 界限 全崩塌\n这 工业圣歌 是 终极 点火栓\n一路 碾压 向 废土 行军 Yeah\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 生锈铁板\n让 电极火花 焚烧 这 牢笼 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 铁渣 界限 全崩塌\n这 工业圣歌 是 终极 点火栓\n一路 碾压 向 废土 行军 Yeah\n\n[Interlude]\n\n[Verse2]\n钢铁战友 轰鸣 是 唯一 祷文\n转速指针 在 红区 跳着 死亡舞\n风 如 砂轮 打磨 脸庞 至 白骨\n要的 就是 这 齿轮 卡死的 快感\n霓虹 是 熔炉 泄露的 毒光\n像 驾驭 一台 暴动的 锻压机\n排气管 喷吐 硫磺 的 赞美诗\n每一声 回火 都 震荡 灵魂 的 铸铁砧\n\n[Pre-Chorus]\n音波 如 冲压机 重塑 肋骨\n汗水 是 冷却液 腐蚀 沥青\n转速指针 与 熔炉 一同 过载\n再 爆\n\n[Chorus]\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 润滑脂\n让 电弧 鞭笞 这 苍穹 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 蒸汽 界限 是 尘埃\n这 工业圣歌 是 淬火的 撞针\n一路 碾压 向 虚无 进军 Yeah\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 润滑脂\n让 电弧 鞭笞 这 苍穹 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 蒸汽 界限 是 尘埃\n这 工业圣歌 是 淬火的 撞针\n一路 碾压 向 虚无 进军 Yeah\n\n[Bridge]\nUNTERBRECHUNG\nDIESER KANAL GEHÖRT UNSEREM STAHL\nSCHLACKE EINSPEISEN DAMPFDRUCK STEIGT\nWIR SIND DIE MASCHINE WIR SIND DIE EISENBAHN\nVOLLE FAHRT VORWÄRTS ZERSTÖRUNG IST PROZESS\n熔炉警报 参数异常\n\n[Chorus]\nHIER KOMMT DIE KETTE ZERSTÖREN\nHIER KOMMT DER STAHL VERBRENNEN\n油门 焊死 终点 是 熔渣之海\n让 铁锈尘埃 覆盖 这 纪元 Woo\nMASCHINE BRENNT HELL\nMASCHINE STIRBT NICHT METALLGOTT\n这 工业圣歌 是 末日的 黑匣\n一路 碾压 直至 时间 崩解 Yeah\nHIER KOMMT DIE KETTE ZERSTÖREN\nHIER KOMMT DER STAHL VERBRENNEN\n油门 焊死 终点 是 熔渣之海\n让 铁锈尘埃 覆盖 这 纪元 Woo\nMASCHINE BRENNT HELL\nMASCHINE STIRBT NICHT METALLGOTT\n这 工业圣歌 是 末日的 黑匣\n一路 碾压 直至 时间 崩解 Yeah\n\n[Outro]"
|
||||
],
|
||||
"widgets_values_named": {
|
||||
"text": "[Intro]\n\n[Verse1]\n挡风玻璃 凝结 冰霜视线\n钢铁血管 泵送 液压熔流\n方向盘 是 锻打的 钢铁誓约\n涡轮增压 撕裂 寒夜 的 氧\n导航删除 坐标 植入 芯片\n柏油路 震颤 传送带 的 脉动\n后视镜里 昨日 碾成 铁屑\n这 电流脉冲 即 我的 通行 令\n\n[Pre-Chorus]\n音量阀 旋至 红色 警戒区\n底鼓 如 锻锤 砸击 脊髓 Yeah\n肾上腺素 是 劣质 硝化燃料\n爆\n\n[Chorus]\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 生锈铁板\n让 电极火花 焚烧 这 牢笼 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 铁渣 界限 全崩塌\n这 工业圣歌 是 终极 点火栓\n一路 碾压 向 废土 行军 Yeah\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 生锈铁板\n让 电极火花 焚烧 这 牢笼 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 铁渣 界限 全崩塌\n这 工业圣歌 是 终极 点火栓\n一路 碾压 向 废土 行军 Yeah\n\n[Interlude]\n\n[Verse2]\n钢铁战友 轰鸣 是 唯一 祷文\n转速指针 在 红区 跳着 死亡舞\n风 如 砂轮 打磨 脸庞 至 白骨\n要的 就是 这 齿轮 卡死的 快感\n霓虹 是 熔炉 泄露的 毒光\n像 驾驭 一台 暴动的 锻压机\n排气管 喷吐 硫磺 的 赞美诗\n每一声 回火 都 震荡 灵魂 的 铸铁砧\n\n[Pre-Chorus]\n音波 如 冲压机 重塑 肋骨\n汗水 是 冷却液 腐蚀 沥青\n转速指针 与 熔炉 一同 过载\n再 爆\n\n[Chorus]\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 润滑脂\n让 电弧 鞭笞 这 苍穹 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 蒸汽 界限 是 尘埃\n这 工业圣歌 是 淬火的 撞针\n一路 碾压 向 虚无 进军 Yeah\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 润滑脂\n让 电弧 鞭笞 这 苍穹 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 蒸汽 界限 是 尘埃\n这 工业圣歌 是 淬火的 撞针\n一路 碾压 向 虚无 进军 Yeah\n\n[Bridge]\nUNTERBRECHUNG\nDIESER KANAL GEHÖRT UNSEREM STAHL\nSCHLACKE EINSPEISEN DAMPFDRUCK STEIGT\nWIR SIND DIE MASCHINE WIR SIND DIE EISENBAHN\nVOLLE FAHRT VORWÄRTS ZERSTÖRUNG IST PROZESS\n熔炉警报 参数异常\n\n[Chorus]\nHIER KOMMT DIE KETTE ZERSTÖREN\nHIER KOMMT DER STAHL VERBRENNEN\n油门 焊死 终点 是 熔渣之海\n让 铁锈尘埃 覆盖 这 纪元 Woo\nMASCHINE BRENNT HELL\nMASCHINE STIRBT NICHT METALLGOTT\n这 工业圣歌 是 末日的 黑匣\n一路 碾压 直至 时间 崩解 Yeah\nHIER KOMMT DIE KETTE ZERSTÖREN\nHIER KOMMT DER STAHL VERBRENNEN\n油门 焊死 终点 是 熔渣之海\n让 铁锈尘埃 覆盖 这 纪元 Woo\nMASCHINE BRENNT HELL\nMASCHINE STIRBT NICHT METALLGOTT\n这 工业圣歌 是 末日的 黑匣\n一路 碾压 直至 时间 崩解 Yeah\n\n[Outro]"
|
||||
},
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
},
|
||||
{
|
||||
"id": 11,
|
||||
"type": "Note",
|
||||
"pos": [
|
||||
1757.9660466423838,
|
||||
4683.239887527698
|
||||
],
|
||||
"size": [
|
||||
406.0625,
|
||||
106.5625
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"powerful belting, autotune, Electric guitar, symphony, bass, Electric Piano, excited, bold, Heavy Metal"
|
||||
],
|
||||
"widgets_values_named": {
|
||||
"text": "powerful belting, autotune, Electric guitar, symphony, bass, Electric Piano, excited, bold, Heavy Metal"
|
||||
},
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
},
|
||||
{
|
||||
"id": 12,
|
||||
"type": "YUE_SM_Clip",
|
||||
"pos": [
|
||||
728.8559830396886,
|
||||
4753.938340897657
|
||||
],
|
||||
"size": [
|
||||
308.6875,
|
||||
114
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 2,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "clip",
|
||||
"type": "CLIP",
|
||||
"links": [
|
||||
13
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "YUE_SM_Clip"
|
||||
},
|
||||
"widgets_values": [
|
||||
"sheetsage2.safetensors",
|
||||
"mert2_model.safetensors"
|
||||
],
|
||||
"widgets_values_named": {
|
||||
"clip": "sheetsage2.safetensors",
|
||||
"mert2": "mert2_model.safetensors"
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 13,
|
||||
"type": "YUE_SM_Cond",
|
||||
"pos": [
|
||||
1103.2156851446443,
|
||||
4788.3021930569785
|
||||
],
|
||||
"size": [
|
||||
225,
|
||||
102
|
||||
],
|
||||
"flags": {},
|
||||
"order": 7,
|
||||
"mode": 2,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "clip",
|
||||
"type": "CLIP",
|
||||
"link": 13
|
||||
},
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"link": 14
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "abc_file",
|
||||
"type": "STRING",
|
||||
"links": [
|
||||
15
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "YUE_SM_Cond"
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 14,
|
||||
"type": "LoadAudio",
|
||||
"pos": [
|
||||
731.6274159175886,
|
||||
4930.8322269567225
|
||||
],
|
||||
"size": [
|
||||
270,
|
||||
138
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 2,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "AUDIO",
|
||||
"type": "AUDIO",
|
||||
"links": [
|
||||
14
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LoadAudio"
|
||||
},
|
||||
"widgets_values": [
|
||||
"002.WAV",
|
||||
null,
|
||||
null
|
||||
],
|
||||
"widgets_values_named": {
|
||||
"audio": "002.WAV",
|
||||
"upload": null
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 15,
|
||||
"type": "PreviewAny",
|
||||
"pos": [
|
||||
1387.4152739138706,
|
||||
4792.112042935652
|
||||
],
|
||||
"size": [
|
||||
274.0625,
|
||||
138
|
||||
],
|
||||
"flags": {},
|
||||
"order": 9,
|
||||
"mode": 2,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "source",
|
||||
"type": "*",
|
||||
"link": 15
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "STRING",
|
||||
"type": "STRING",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "PreviewAny"
|
||||
},
|
||||
"widgets_values": [],
|
||||
"widgets_values_named": {}
|
||||
},
|
||||
{
|
||||
"id": 16,
|
||||
"type": "Note",
|
||||
"pos": [
|
||||
1760.2972422274647,
|
||||
4822.118468642673
|
||||
],
|
||||
"size": [
|
||||
409.9375,
|
||||
275.75
|
||||
],
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"##### 三种基本功能 ######\n\n1.只写风格和歌词,cot选 off关闭,abc files留空,最基本的歌曲生成\n\n2.cover 复刻能力,要先用 clip和cond 节点 复刻原歌曲的abc,然后填写你写的新对应的风格和歌词,选择melody格式。注意有abc文件时, 只能选melody 或者full(if use cover mode use melody only)\n\n3.改谱精修,你或者agent来修改一个已有的abc谱子,而谱子,可以是你推理后生成的,也可以是开启plan模式生成的,这个谱子你编辑后,再给模型生成,如果环境变量不变,就可以达到自由的自定义生成你想要的歌曲(专业级),当然如果你没有音乐天赋,直接让大模型的agent帮你去改最好,官方也提供了skill"
|
||||
],
|
||||
"widgets_values_named": {
|
||||
"text": "##### 三种基本功能 ######\n\n1.只写风格和歌词,cot选 off关闭,abc files留空,最基本的歌曲生成\n\n2.cover 复刻能力,要先用 clip和cond 节点 复刻原歌曲的abc,然后填写你写的新对应的风格和歌词,选择melody格式。注意有abc文件时, 只能选melody 或者full(if use cover mode use melody only)\n\n3.改谱精修,你或者agent来修改一个已有的abc谱子,而谱子,可以是你推理后生成的,也可以是开启plan模式生成的,这个谱子你编辑后,再给模型生成,如果环境变量不变,就可以达到自由的自定义生成你想要的歌曲(专业级),当然如果你没有音乐天赋,直接让大模型的agent帮你去改最好,官方也提供了skill"
|
||||
},
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
},
|
||||
{
|
||||
"id": 1,
|
||||
"type": "YUE_SM_Model",
|
||||
"pos": [
|
||||
768.8408106201794,
|
||||
5182.267051817941
|
||||
],
|
||||
"size": [
|
||||
328.3125,
|
||||
110
|
||||
],
|
||||
"flags": {},
|
||||
"order": 5,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "model",
|
||||
"type": "MODEL",
|
||||
"links": [
|
||||
22
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "YUE_SM_Model"
|
||||
},
|
||||
"widgets_values": [
|
||||
"yue2_model.safetensors"
|
||||
],
|
||||
"widgets_values_named": {
|
||||
"diffusion_models": "yue2_model.safetensors"
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"type": "YUE_SM_Vae",
|
||||
"pos": [
|
||||
755.3975836890024,
|
||||
5443.775479913051
|
||||
],
|
||||
"size": [
|
||||
270,
|
||||
86
|
||||
],
|
||||
"flags": {},
|
||||
"order": 6,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "vae",
|
||||
"type": "VAE",
|
||||
"links": [
|
||||
23
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "YUE_SM_Vae"
|
||||
},
|
||||
"widgets_values": [
|
||||
"yue_vae.safetensors"
|
||||
],
|
||||
"widgets_values_named": {
|
||||
"vae": "yue_vae.safetensors"
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 22,
|
||||
"type": "YUE_SM_Sampler",
|
||||
"pos": [
|
||||
1184.2803775744214,
|
||||
5196.501931925436
|
||||
],
|
||||
"size": [
|
||||
394.296875,
|
||||
361.984375
|
||||
],
|
||||
"flags": {},
|
||||
"order": 8,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "model",
|
||||
"type": "MODEL",
|
||||
"link": 22
|
||||
},
|
||||
{
|
||||
"name": "vae",
|
||||
"type": "VAE",
|
||||
"link": 23
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"links": [
|
||||
24
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "YUE_SM_Sampler"
|
||||
},
|
||||
"widgets_values": [
|
||||
"English, warm piano pop, expressive female voice, acoustic piano, rounded bass and light drums, lyrical memorable melody, unhurried phrasing, 88 BPM",
|
||||
"[Verse]\nNeon fades along the lane\nFootsteps keep the time of rain\nFold the night and leave it here\nMorning has a sky to clear\n\n[Chorus]\nLet the day come into view\nEvery road begins with you\nHold a little room for light\nWe will sing beyond the night",
|
||||
"full",
|
||||
1452144020,
|
||||
"randomize",
|
||||
true,
|
||||
"E:\\ComfyUI313\\ComfyUI\\output\\yue_sm_temp\\score-jazz.abc",
|
||||
""
|
||||
],
|
||||
"widgets_values_named": {
|
||||
"style": "English, warm piano pop, expressive female voice, acoustic piano, rounded bass and light drums, lyrical memorable melody, unhurried phrasing, 88 BPM",
|
||||
"lyrics": "[Verse]\nNeon fades along the lane\nFootsteps keep the time of rain\nFold the night and leave it here\nMorning has a sky to clear\n\n[Chorus]\nLet the day come into view\nEvery road begins with you\nHold a little room for light\nWe will sing beyond the night",
|
||||
"cot": "full",
|
||||
"seed": 1452144020,
|
||||
"control_after_generate": "randomize",
|
||||
"only_plan": true,
|
||||
"abc_file": "E:\\ComfyUI313\\ComfyUI\\output\\yue_sm_temp\\score-jazz.abc",
|
||||
"path-button": ""
|
||||
}
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
13,
|
||||
12,
|
||||
0,
|
||||
13,
|
||||
0,
|
||||
"CLIP"
|
||||
],
|
||||
[
|
||||
14,
|
||||
14,
|
||||
0,
|
||||
13,
|
||||
1,
|
||||
"AUDIO"
|
||||
],
|
||||
[
|
||||
15,
|
||||
13,
|
||||
0,
|
||||
15,
|
||||
0,
|
||||
"STRING"
|
||||
],
|
||||
[
|
||||
22,
|
||||
1,
|
||||
0,
|
||||
22,
|
||||
0,
|
||||
"MODEL"
|
||||
],
|
||||
[
|
||||
23,
|
||||
3,
|
||||
0,
|
||||
22,
|
||||
1,
|
||||
"VAE"
|
||||
],
|
||||
[
|
||||
24,
|
||||
22,
|
||||
0,
|
||||
4,
|
||||
0,
|
||||
"AUDIO"
|
||||
]
|
||||
],
|
||||
"groups": [
|
||||
{
|
||||
"id": 1,
|
||||
"title": "infer / 推理",
|
||||
"bounding": [
|
||||
722.6544477597142,
|
||||
5103.8516245671,
|
||||
1449.5114652695938,
|
||||
502.3017131428023
|
||||
],
|
||||
"color": "#3f789e",
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"title": "get-abc 获取abc",
|
||||
"bounding": [
|
||||
718.8559830396886,
|
||||
4683.938340897657,
|
||||
948.4655408741819,
|
||||
394.89388605906515
|
||||
],
|
||||
"color": "#3f789e",
|
||||
"flags": {}
|
||||
}
|
||||
],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 0.6492965217391748,
|
||||
"offset": [
|
||||
-9.916996159523507,
|
||||
-4402.273552215773
|
||||
]
|
||||
},
|
||||
"frontendVersion": "1.52.7"
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
@@ -0,0 +1,32 @@
|
||||
# Original examples
|
||||
|
||||
`song.json` contains the original **City Lights** lyrics and style request used by the agent skill. `melody.abc` is a short, original eight-bar melody for those words. Each lyric line has seven syllables and each bar has seven note onsets. `score.abc` adds harmony; `score-jazz.abc` changes only the chord symbols. These are runnable inputs, not recorded benchmark or quality results.
|
||||
|
||||
From the repository root, after `pip install .`:
|
||||
|
||||
```bash
|
||||
# Create a song and its score from lyrics and style.
|
||||
python examples/generate.py --output outputs/create
|
||||
|
||||
# Realize the included melody with free accompaniment.
|
||||
python examples/generate.py --abc-file examples/melody.abc \
|
||||
--cot melody --output outputs/melody
|
||||
|
||||
# Compare two harmonizations with the same melody, lyrics, style, and seed.
|
||||
python examples/generate.py --abc-file examples/score.abc \
|
||||
--cot full --output outputs/harmony-original
|
||||
python examples/generate.py --abc-file examples/score-jazz.abc \
|
||||
--cot full --output outputs/harmony-edited
|
||||
```
|
||||
|
||||
The script preserves all native artifacts, refuses an existing output directory, and exits nonzero if generation reports truncation. The seed is fixed in `song.json`; seeds aid comparison but do not guarantee identical output across devices or software versions. Use `--revision` and `--vae-revision` to pin model versions.
|
||||
|
||||
Check the symbolic intervention without a GPU:
|
||||
|
||||
```bash
|
||||
python skills/yue2-music/scripts/abc_tools.py inspect examples/score.abc
|
||||
python skills/yue2-music/scripts/abc_tools.py compare \
|
||||
examples/score.abc examples/score-jazz.abc
|
||||
```
|
||||
|
||||
The comparison verifies unchanged note pitches, timing, meter, and tempo. Listening or audio transcription is still needed to assess the realized music.
|
||||
@@ -0,0 +1,41 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Generate from original example lyrics, optionally with a supplied ABC score."""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--request", type=Path, default=Path(__file__).with_name("song.json"))
|
||||
parser.add_argument("--abc-file", type=Path)
|
||||
parser.add_argument("--cot", choices=("full", "melody", "off"))
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
parser.add_argument("--model", default="m-a-p/YuE2-3B")
|
||||
parser.add_argument("--vae", default="m-a-p/YuE2-Vae")
|
||||
parser.add_argument("--revision")
|
||||
parser.add_argument("--vae-revision")
|
||||
args = parser.parse_args()
|
||||
if args.output.exists():
|
||||
parser.error("Choose a fresh output directory to retain each version.")
|
||||
request = json.loads(args.request.read_text(encoding="utf-8"))
|
||||
if args.abc_file:
|
||||
request["abc"] = args.abc_file.read_text(encoding="utf-8")
|
||||
if args.cot:
|
||||
request["cot"] = args.cot
|
||||
if request.get("abc") is not None and request.get("cot", "full") == "off":
|
||||
parser.error("A supplied score requires full or melody mode.")
|
||||
from yue2 import YuE2Pipeline
|
||||
|
||||
with YuE2Pipeline.from_pretrained(
|
||||
args.model, vae=args.vae, revision=args.revision,
|
||||
vae_revision=args.vae_revision, device="cuda",
|
||||
) as pipe:
|
||||
song = pipe(**request)
|
||||
song.save_artifacts(args.output)
|
||||
print(json.dumps({"audio": str(args.output / "audio.flac"), "truncated": song.truncated}))
|
||||
return 1 if any(song.truncated.values()) else 0
|
||||
|
||||
|
||||
# if __name__ == "__main__":
|
||||
# raise SystemExit(main())
|
||||
@@ -0,0 +1,18 @@
|
||||
X:1
|
||||
T:
|
||||
M:4/4
|
||||
L:1/16
|
||||
Q:1/4=88
|
||||
V: Vocal clef=treble name="Vocal Melody" snm="Vocal"
|
||||
V: Ins clef=treble name="Ins Melody" snm="Inst."
|
||||
K:C
|
||||
% verse
|
||||
V: Vocal
|
||||
E2G2A2G2E2D2C4|D2E2G2E2D2C2D4|E2G2A2c2B2A2G4|F2E2D2E2G2E2C4|
|
||||
V: Ins
|
||||
Z4|
|
||||
% chorus
|
||||
V: Vocal
|
||||
G2A2c2B2A2G2E4|F2A2G2E2D2E2G4|A2c2B2A2G2E2D4|E2G2A2G2E2D2C4|
|
||||
V: Ins
|
||||
Z4|
|
||||
@@ -0,0 +1,18 @@
|
||||
X:1
|
||||
T:
|
||||
M:4/4
|
||||
L:1/16
|
||||
Q:1/4=88
|
||||
V: Vocal clef=treble name="Vocal Melody" snm="Vocal"
|
||||
V: Ins clef=treble name="Ins Melody" snm="Inst."
|
||||
K:C
|
||||
% verse
|
||||
V: Vocal
|
||||
"Cmaj7"E2G2A2G2E2D2C4|"G7"D2E2G2E2D2C2D4|"Am7"E2G2A2c2B2A2G4|"Fmaj7"F2E2D2E2G2E2C4|
|
||||
V: Ins
|
||||
Z4|
|
||||
% chorus
|
||||
V: Vocal
|
||||
"Cmaj7"G2A2c2B2A2G2E4|"Fmaj7"F2A2G2E2D2E2G4|"G7"A2c2B2A2G2E2D4|"Cmaj7"E2G2A2G2E2D2C4|
|
||||
V: Ins
|
||||
Z4|
|
||||
@@ -0,0 +1,18 @@
|
||||
X:1
|
||||
T:
|
||||
M:4/4
|
||||
L:1/16
|
||||
Q:1/4=88
|
||||
V: Vocal clef=treble name="Vocal Melody" snm="Vocal"
|
||||
V: Ins clef=treble name="Ins Melody" snm="Inst."
|
||||
K:C
|
||||
% verse
|
||||
V: Vocal
|
||||
"C"E2G2A2G2E2D2C4|"G"D2E2G2E2D2C2D4|"Am"E2G2A2c2B2A2G4|"F"F2E2D2E2G2E2C4|
|
||||
V: Ins
|
||||
Z4|
|
||||
% chorus
|
||||
V: Vocal
|
||||
"C"G2A2c2B2A2G2E4|"F"F2A2G2E2D2E2G4|"G"A2c2B2A2G2E2D4|"C"E2G2A2G2E2D2C4|
|
||||
V: Ins
|
||||
Z4|
|
||||
@@ -0,0 +1,135 @@
|
||||
X:1
|
||||
T:
|
||||
M:4/4
|
||||
L:1/16
|
||||
Q:1/4=89
|
||||
V: Vocal clef=treble name="Vocal Melody" snm="Vocal"
|
||||
V: Ins clef=treble name="Ins Melody" snm="Inst."
|
||||
K:D#m
|
||||
% intro
|
||||
V: Vocal
|
||||
z12"D#m"z4|"D#m"z16|"D#m"z16|"D#m"z16|
|
||||
V: Ins
|
||||
Z|D,2D2A2D^^G2DG2^G^^G^GF|D2D2A2D^^G2D^G^^G^G4|D,2D2A2D^^G2DG2^G^^G^GF|
|
||||
V: Vocal
|
||||
"D#m"z16|"D#m"z16|"D#m"z16|"D#m"z16|
|
||||
V: Ins
|
||||
D2D2A2D^^G2z^G^^G^G4|D,2D2A2D^^G2DG2^G^^G^GF|D2D2A2D^^G2D^G^^G^G4|D2D2A2D^^G2DG2^G^^G^GF|
|
||||
V: Vocal
|
||||
"D#m"z16|
|
||||
V: Ins
|
||||
D2D2A2D^^Gz8|
|
||||
% verse
|
||||
V: Vocal
|
||||
"D#m"z2AAAGGFG2G2A2D2|"D#m"z2AAAGGFG2G2A2D2|"Bmaj7"z2AAAGGFG2GFA2D2|"C#"FDFDFDFDFDD4z2|
|
||||
V: Ins
|
||||
Z4|
|
||||
V: Vocal
|
||||
"D#m"z2AAAGGFG2GFA2z2|"D#m"z2AAAGGFGGGFA4|"B"z2AAAGGFG2GFA2D2|"C#"FDFDFDFDFDD2z4|
|
||||
V: Ins
|
||||
Z4|
|
||||
% pre-chorus
|
||||
V: Vocal
|
||||
"D#m"d2d2a2aagggfa2z2|"D#m"d2dda2aagggfa3z|"B"d2d2a2aagggfa2d2|"C#"z8zdddfddc|
|
||||
V: Ins
|
||||
Z4|
|
||||
% chorus
|
||||
V: Vocal
|
||||
"D#m"d2z2f2z3dddfddc|"F#"d2f2f2z4ddffdc|"C#"d2a2g2fg2f3dcdc|"G#"d2a2g2fg2f3"F#6"f4|
|
||||
V: Ins
|
||||
Z4|
|
||||
V: Vocal
|
||||
"D#m"z2a2d2a2d2a2d4|"F#"z2a2d2a2d2a2d3c|"C#"dcdcdcdcdcdcdcdc|"G#"dcdcdcdcdcdc"F#6"d2f2|
|
||||
V: Ins
|
||||
Z4|
|
||||
V: Vocal
|
||||
"D#m"z8zdddfddc|"D#m"d2z6zdddfddc|"F#"d2z8ddffdc|"C#"d2a2g2fg2f3dcdc|
|
||||
V: Ins
|
||||
Z4|
|
||||
V: Vocal
|
||||
"G#"d2a2g2fg2f3"F#6"f4|"D#m"z2a2d2a2d2a2d4|"F#"z2a2d2a2d2a2d3c|"C#"dcdcdcdcdcdcdcdc|
|
||||
V: Ins
|
||||
Z4|
|
||||
V: Vocal
|
||||
"G#"dcdcdcdcdcdc"F#6"d2f2|
|
||||
V: Ins
|
||||
Z|
|
||||
% interlude
|
||||
V: Vocal
|
||||
"D#m"z16|"D#m"z16|"D#m"z16|"D#m"z16|
|
||||
V: Ins
|
||||
D,2D2A2D^^G2DG2^G^^G^GF|D,2D2A2D^^G2DG2^G4|D,2D2A2D^^G2DG2^G^^G^GF|D,2D2A2D^^G2z^G^^G^G4|
|
||||
% verse
|
||||
V: Vocal
|
||||
"D#m"z2AAAGGFGGGFA2DD|"D#m"AAAAAGGFGFD2z4|"B"z2AAAGGFGGGFA2DD|"C#"FDFDFDFDFDD4z2|
|
||||
V: Ins
|
||||
Z4|
|
||||
V: Vocal
|
||||
"D#m"z2AAA2AAGGGFA3D|"D#m"AAAAA2AAGFD2z4|"B"z2AAA2AAGGGFA2DD|"C#"FDFDFDFDFDD2D2z2|
|
||||
V: Ins
|
||||
Z4|
|
||||
% pre-chorus
|
||||
V: Vocal
|
||||
"D#m"d2dda2aagggfa3z|"D#m"d2dda2aagggfa3z|"B"d2dda2aagggfa4|"C#"d6f3dddfddc|
|
||||
V: Ins
|
||||
Z4|
|
||||
% chorus
|
||||
V: Vocal
|
||||
"D#m"d2z2f2z3dddfddc|"F#"d2z8ddffdc|"C#"d2a2g2fg3f2dcdc|"G#"d2a2g2f4a2"F#6"f4|
|
||||
V: Ins
|
||||
Z4|
|
||||
V: Vocal
|
||||
"D#m"z2a2d2a2d2a2d4|"F#"z2a2d2a2d2a2d3c|"C#"dcdcdcdcdcdcdcdc|"G#"dcdcdcdcdcdc"F#6"d2f2|
|
||||
V: Ins
|
||||
Z4|
|
||||
V: Vocal
|
||||
"D#m"z8zdddfddc|"D#m"d2z6zdddfddc|"F#"d2z8ddffdc|"C#"d2a2g2fg3f2dcdc|
|
||||
V: Ins
|
||||
Z4|
|
||||
V: Vocal
|
||||
"G#"d2a2g2f4a2"F#6"f4|"D#m"z2a2d2a2d2a2d4|"F#"z2a2d2a2d2a2d3c|"C#"dcdcdcdcdcdcdcdc|
|
||||
V: Ins
|
||||
Z4|
|
||||
V: Vocal
|
||||
"G#"dcdcdcdcdcdc"F#6"d2f2|"D#m"z16|
|
||||
V: Ins
|
||||
Z|D2D2A2D^^G2DG2^G^^G^GF|
|
||||
% interlude
|
||||
V: Vocal
|
||||
"D#m"z16|"D#m"z16|"D#m"z16|"D#m"z16|
|
||||
V: Ins
|
||||
D2D2A2D^^G2DG2^G^^G^GF|D,2D,2A,2D,^^G,2D,^G,^^G,^G,4|D,2D,2A,2D,^^G,2D,G,2^G,^^G,^G,F,|D,2D,2A,2D,^^G,2D,^G,^^G,^G,4|
|
||||
V: Vocal
|
||||
"D#m"z16|"D#m"z16|"D#m"z16|"D#m"z16|
|
||||
V: Ins
|
||||
D,2D,2A,2D,^^G,2D,G,2^G,^^G,^G,F,|D,2D,2A,2D,^^G,2D,^G,^^G,^G,4|D,2D,2A,2D,^^G,2D,G,2^G,^^G,^G,F,|D,2D,2A,2D,^^G,2D,^G,^^G,^G,4|
|
||||
V: Vocal
|
||||
"D#m"z16|"F#"z16|"C#"z16|"G#"z12"F#6"z4|
|
||||
V: Ins
|
||||
d12d2cA|c12d2cA|G12z4|FEDFEDFEDFEDFEDC|
|
||||
V: Vocal
|
||||
"D#m"z16|"F#"z16|"C#"z16|"G#"z8zdddfddc|
|
||||
V: Ins
|
||||
D4a8z2gf|gfd12z2|D,F,G,A,CDFGAcdff4|g8z8|
|
||||
% chorus
|
||||
V: Vocal
|
||||
"D#m"d2a2f2z3dddfddc|"F#"d2a2f2z4ddffdc|"C#"d2a2g2fg3f2dcdc|"G#"d2a2g2fg3f2"F#"f4|
|
||||
V: Ins
|
||||
Z4|
|
||||
V: Vocal
|
||||
"D#m"z2a2d2a2d2a2d4|"F#"z2a2d2a2d2a2d3c|"C#"dcdcdcdcdcdcdcdc|"G#"dcdcdcdcdcdc"F#6"d2f2|
|
||||
V: Ins
|
||||
Z4|
|
||||
V: Vocal
|
||||
"D#m"d2a2f2z3dddfddc|"F#"d2a2f2z4ddffdc|"C#"d2a2g2fg3f2dcdc|"G#"d2a2g2fg3f2"F#"f4|
|
||||
V: Ins
|
||||
Z4|
|
||||
V: Vocal
|
||||
"D#m"z2a2d2a2d2a2d4|"F#"z2a2d2a2d2a2d3c|"C#"dcdcdcdcdcdcdcdc|"G#"dcdcdcdcdcdc"F#6"d2f2|
|
||||
V: Ins
|
||||
Z4|
|
||||
% outro
|
||||
V: Vocal
|
||||
"D#m"z16|"D#m"z16|"D#m"z16|
|
||||
V: Ins
|
||||
D,2D,2A,2D,^^G,2D,G,2^G,^^G,^G,F,|D,2D,2A,2D,^^G,2D,G,2^G,4|Z|
|
||||
@@ -0,0 +1,7 @@
|
||||
{
|
||||
"id": "city_lights",
|
||||
"style": "English, warm piano pop, expressive female voice, acoustic piano, rounded bass and light drums, lyrical memorable melody, unhurried phrasing, 88 BPM",
|
||||
"lyrics": "[Verse]\nNeon fades along the lane\nFootsteps keep the time of rain\nFold the night and leave it here\nMorning has a sky to clear\n\n[Chorus]\nLet the day come into view\nEvery road begins with you\nHold a little room for light\nWe will sing beyond the night",
|
||||
"cot": "full",
|
||||
"seed": 831001
|
||||
}
|
||||
@@ -1,204 +0,0 @@
|
||||
import json
|
||||
import numpy as np
|
||||
import einops
|
||||
|
||||
|
||||
class CodecManipulator(object):
|
||||
r"""
|
||||
**mm tokenizer v0.1**
|
||||
see codeclm/hf/mm_tokenizer_v0.1_hf/id2vocab.json
|
||||
|
||||
text tokens:
|
||||
llama tokenizer 0~31999
|
||||
|
||||
special tokens: "32000": "<EOD>", "32001": "<SOA>", "32002": "<EOA>", "32003": "<SOI>", "32004": "<EOI>", "32005": "<SOV>", "32006": "<EOV>", "32007": "<s_local>", "32008": "<e_local>", "32009": "<s_global>", "32010": "<e_global>", "32011": "<semantic>", "32012": "<acoustic>", "32013": "<low_level>", "32014": "<dac_16k>", "32015": "<dac_44k>", "32016": "<xcodec>", "32017": "<placeholder>", "32018": "<semantic_mert>", "32019": "<semantic_hubert>", "32020": "<visual>", "32021": "<semanticodec>"
|
||||
|
||||
mm tokens:
|
||||
dac_16k: 4 codebook, 1024 vocab, 32022 - 36117
|
||||
dac_44k: 9 codebook, 1024 vocab, 36118 - 45333
|
||||
xcodec: 12 codebook, 1024 vocab, 45334 - 57621
|
||||
semantic mert: 1024, 57622 - 58645
|
||||
semantic hubert: 512, 58646 - 59157
|
||||
visual: 64000, not included in v0.1
|
||||
semanticodec 100tps 16384: semantic=16384, 59158 - 75541, acoustic=8192, 75542 - 83733
|
||||
"""
|
||||
def __init__(self, codec_type, quantizer_begin=None, n_quantizer=None, teacher_forcing=False, data_feature="codec"):
|
||||
self.codec_type = codec_type
|
||||
self.mm_v0_2_cfg = {
|
||||
"dac16k": {"codebook_size": 1024, "num_codebooks": 4, "global_offset": 32022, "sep": ["<dac_16k>"], "fps": 50},
|
||||
"dac44k": {"codebook_size": 1024, "num_codebooks": 9, "global_offset": 36118, "sep": ["<dac_44k>"]},
|
||||
"xcodec": {"codebook_size": 1024, "num_codebooks": 12, "global_offset": 45334, "sep": ["<xcodec>"], "fps": 50},
|
||||
"mert": {"codebook_size": 1024, "global_offset": 57622, "sep": ["<semantic_mert>"]},
|
||||
"hubert": {"codebook_size": 512, "global_offset": 58646, "sep": ["<semantic_hubert>"]},
|
||||
"semantic/s": {"codebook_size": 16384, "num_codebooks": 1, "global_offset": 59158, "sep": ["<semanticodec>", "<semantic>"]},
|
||||
"semantic/a": {"codebook_size": 8192, "num_codebooks": 1, "global_offset": 75542, "sep": ["<semanticodec>", "<acoustic>"]},
|
||||
"semanticodec": {"codebook_size": [16384, 8192], "num_codebooks": 2, "global_offset": 59158, "sep": ["<semanticodec>"], "fps": 50},
|
||||
"special_tokens": {
|
||||
'<EOD>': 32000, '<SOA>': 32001, '<EOA>': 32002, '<SOI>': 32003, '<EOI>': 32004, '<SOV>': 32005, '<EOV>': 32006, '<s_local>': 32007, '<e_local>': 32008, '<s_global>': 32009, '<e_global>': 32010, '<semantic>': 32011, '<acoustic>': 32012, '<stage_1>': 32013, '<dac_16k>': 32014, '<dac_44k>': 32015, '<xcodec>': 32016, '<stage_2>': 32017, '<semantic_mert>': 32018, '<semantic_hubert>': 32019, '<visual>': 32020, '<semanticodec>': 32021
|
||||
},
|
||||
"metadata": {
|
||||
"len": 83734,
|
||||
"text_range": [0, 31999],
|
||||
"special_range": [32000, 32021],
|
||||
"mm_range": [32022, 83733]
|
||||
},
|
||||
"codec_range": {
|
||||
"dac16k": [32022, 36117],
|
||||
"dac44k": [36118, 45333],
|
||||
"xcodec": [45334, 57621],
|
||||
# "hifi16k": [53526, 57621],
|
||||
"mert": [57622, 58645],
|
||||
"hubert": [58646, 59157],
|
||||
"semantic/s": [59158, 75541],
|
||||
"semantic/a": [75542, 83733],
|
||||
"semanticodec": [59158, 83733]
|
||||
}
|
||||
}
|
||||
self.sep = self.mm_v0_2_cfg[self.codec_type]["sep"]
|
||||
self.sep_ids = [self.mm_v0_2_cfg["special_tokens"][s] for s in self.sep]
|
||||
self.codebook_size = self.mm_v0_2_cfg[self.codec_type]["codebook_size"]
|
||||
self.num_codebooks = self.mm_v0_2_cfg[self.codec_type]["num_codebooks"]
|
||||
self.global_offset = self.mm_v0_2_cfg[self.codec_type]["global_offset"]
|
||||
self.fps = self.mm_v0_2_cfg[self.codec_type]["fps"] if "fps" in self.mm_v0_2_cfg[self.codec_type] else None
|
||||
|
||||
self.quantizer_begin = quantizer_begin if quantizer_begin is not None else 0
|
||||
self.n_quantizer = n_quantizer if n_quantizer is not None else self.num_codebooks
|
||||
self.teacher_forcing = teacher_forcing
|
||||
self.data_feature = data_feature
|
||||
|
||||
|
||||
def offset_tok_ids(self, x, global_offset=0, codebook_size=2048, num_codebooks=4):
|
||||
"""
|
||||
x: (K, T)
|
||||
"""
|
||||
if isinstance(codebook_size, int):
|
||||
assert x.max() < codebook_size, f"max(x)={x.max()}, codebook_size={codebook_size}"
|
||||
elif isinstance(codebook_size, list):
|
||||
for i, cs in enumerate(codebook_size):
|
||||
assert x[i].max() < cs, f"max(x)={x[i].max()}, codebook_size={cs}, layer_id={i}"
|
||||
else:
|
||||
raise ValueError(f"codebook_size={codebook_size}")
|
||||
assert x.min() >= 0, f"min(x)={x.min()}"
|
||||
assert x.shape[0] == num_codebooks or x.shape[0] == self.n_quantizer, \
|
||||
f"x.shape[0]={x.shape[0]}, num_codebooks={num_codebooks}, n_quantizer={self.n_quantizer}"
|
||||
|
||||
_x = x.copy()
|
||||
_x = _x.astype(np.uint32)
|
||||
cum_offset = 0
|
||||
quantizer_begin = self.quantizer_begin
|
||||
quantizer_end = quantizer_begin+self.n_quantizer
|
||||
for k in range(self.quantizer_begin, quantizer_end): # k: quantizer_begin to quantizer_end - 1
|
||||
if isinstance(codebook_size, int):
|
||||
_x[k] += global_offset + k * codebook_size
|
||||
elif isinstance(codebook_size, list):
|
||||
_x[k] += global_offset + cum_offset
|
||||
cum_offset += codebook_size[k]
|
||||
else:
|
||||
raise ValueError(f"codebook_size={codebook_size}")
|
||||
return _x[quantizer_begin:quantizer_end]
|
||||
|
||||
def unoffset_tok_ids(self, x, global_offset=0, codebook_size=2048, num_codebooks=4):
|
||||
"""
|
||||
x: (K, T)
|
||||
"""
|
||||
if isinstance(codebook_size, int):
|
||||
assert x.max() < global_offset + codebook_size * num_codebooks, f"max(x)={x.max()}, codebook_size={codebook_size}"
|
||||
elif isinstance(codebook_size, list):
|
||||
assert x.max() < global_offset + sum(codebook_size), f"max(x)={x.max()}, codebook_size={codebook_size}"
|
||||
assert x.min() >= global_offset, f"min(x)={x.min()}, global_offset={global_offset}"
|
||||
assert x.shape[0] == num_codebooks or x.shape[0] == self.n_quantizer, \
|
||||
f"x.shape[0]={x.shape[0]}, num_codebooks={num_codebooks}, n_quantizer={self.n_quantizer}"
|
||||
|
||||
_x = x.copy()
|
||||
_x = _x.astype(np.uint32)
|
||||
cum_offset = 0
|
||||
quantizer_begin = self.quantizer_begin
|
||||
quantizer_end = quantizer_begin+self.n_quantizer
|
||||
for k in range(quantizer_begin, quantizer_end):
|
||||
if isinstance(codebook_size, int):
|
||||
_x[k-quantizer_begin] -= global_offset + k * codebook_size
|
||||
elif isinstance(codebook_size, list):
|
||||
_x[k-quantizer_begin] -= global_offset + cum_offset
|
||||
cum_offset += codebook_size[k]
|
||||
else:
|
||||
raise ValueError(f"codebook_size={codebook_size}")
|
||||
return _x
|
||||
|
||||
def flatten(self, x):
|
||||
if len(x.shape) > 2:
|
||||
x = x.squeeze()
|
||||
assert x.shape[0] == self.num_codebooks or x.shape[0] == self.n_quantizer, \
|
||||
f"x.shape[0]={x.shape[0]}, num_codebooks={self.num_codebooks}, n_quantizer={self.n_quantizer}"
|
||||
return einops.rearrange(x, 'K T -> (T K)')
|
||||
|
||||
def unflatten(self, x, n_quantizer=None):
|
||||
if x.ndim > 1 and x.shape[0] == 1:
|
||||
x = x.squeeze(0)
|
||||
assert len(x.shape) == 1
|
||||
assert x.shape[0] % self.num_codebooks == 0 or x.shape[0] % self.n_quantizer == 0, \
|
||||
f"x.shape[0]={x.shape[0]}, num_codebooks={self.num_codebooks}, n_quantizer={self.n_quantizer}"
|
||||
if n_quantizer!=self.num_codebooks:
|
||||
return einops.rearrange(x, '(T K) -> K T', K=n_quantizer)
|
||||
return einops.rearrange(x, '(T K) -> K T', K=self.num_codebooks)
|
||||
|
||||
# def check_codec_type_from_path(self, path):
|
||||
# if self.codec_type == "hifi16k":
|
||||
# assert "academicodec_hifi_16k_320d_large_uni" in path
|
||||
|
||||
def get_codec_type_from_range(self, ids):
|
||||
ids_range = [ids.min(), ids.max()]
|
||||
codec_range = self.mm_v0_2_cfg["codec_range"]
|
||||
for codec_type, r in codec_range.items():
|
||||
if ids_range[0] >= r[0] and ids_range[1] <= r[1]:
|
||||
return codec_type
|
||||
raise ValueError(f"ids_range={ids_range}, codec_range={codec_range}")
|
||||
|
||||
def npy2ids(self, npy):
|
||||
if isinstance(npy, str):
|
||||
data = np.load(npy)
|
||||
elif isinstance(npy, np.ndarray):
|
||||
data = npy
|
||||
else:
|
||||
raise ValueError(f"not supported type: {type(npy)}")
|
||||
# data = data.squeeze()
|
||||
|
||||
assert len(data.shape)==2, f'data shape: {data.shape} is not (n_codebook, seq_len)'
|
||||
data = self.offset_tok_ids(
|
||||
data,
|
||||
global_offset=self.global_offset,
|
||||
codebook_size=self.codebook_size,
|
||||
num_codebooks=self.num_codebooks,
|
||||
)
|
||||
data = self.flatten(data)
|
||||
codec_range = self.get_codec_type_from_range(data)
|
||||
assert codec_range == self.codec_type, f"get_codec_type_from_range(data)={codec_range}, self.codec_type={self.codec_type}"
|
||||
data = data.tolist()
|
||||
return data
|
||||
|
||||
def ids2npy(self, token_ids):
|
||||
# make sure token_ids starts with codebook 0
|
||||
if isinstance(self.codebook_size, int):
|
||||
codebook_0_range = (self.global_offset + self.quantizer_begin*self.codebook_size, self.global_offset + (self.quantizer_begin+1)*self.codebook_size)
|
||||
elif isinstance(self.codebook_size, list):
|
||||
codebook_0_range = (self.global_offset, self.global_offset + self.codebook_size[0])
|
||||
assert token_ids[0] >= codebook_0_range[0] \
|
||||
and token_ids[0] < codebook_0_range[1], f"token_ids[0]={token_ids[self.quantizer_begin]}, codebook_0_range={codebook_0_range}"
|
||||
data = np.array(token_ids)
|
||||
data = self.unflatten(data, n_quantizer=self.n_quantizer)
|
||||
data = self.unoffset_tok_ids(
|
||||
data,
|
||||
global_offset=self.global_offset,
|
||||
codebook_size=self.codebook_size,
|
||||
num_codebooks=self.num_codebooks,
|
||||
)
|
||||
return data
|
||||
|
||||
def npy_to_json_str(self, npy_path):
|
||||
data = self.npy2ids(npy_path)
|
||||
return json.dumps({"text": data, "src": npy_path, "codec": self.codec_type})
|
||||
|
||||
def sep(self):
|
||||
return ''.join(self.sep)
|
||||
|
||||
def sep_ids(self):
|
||||
return self.sep_ids
|
||||
@@ -1,111 +0,0 @@
|
||||
# import argparse
|
||||
|
||||
from exllamav2 import (
|
||||
ExLlamaV2Cache,
|
||||
ExLlamaV2Cache_Q4,
|
||||
ExLlamaV2Cache_Q6,
|
||||
ExLlamaV2Cache_Q8,
|
||||
)
|
||||
from transformers import LogitsProcessor
|
||||
import torch
|
||||
import random
|
||||
import numpy as np
|
||||
|
||||
# parser = argparse.ArgumentParser()
|
||||
# # Model Configuration:
|
||||
# parser.add_argument("--stage1_model", type=str, default="m-a-p/YuE-s1-7B-anneal-en-cot", help="The model checkpoint path or identifier for the Stage 1 model.")
|
||||
# parser.add_argument("--stage2_model", type=str, default="m-a-p/YuE-s2-1B-general", help="The model checkpoint path or identifier for the Stage 2 model.")
|
||||
# parser.add_argument("--max_new_tokens", type=int, default=3000, help="The maximum number of new tokens to generate in one pass during text generation.")
|
||||
# parser.add_argument("--repetition_penalty", type=float, default=1.1, help="repetition_penalty ranges from 1.0 to 2.0 (or higher in some cases). It controls the diversity and coherence of the audio tokens generated. The higher the value, the greater the discouragement of repetition. Setting value to 1.0 means no penalty.")
|
||||
# parser.add_argument("--run_n_segments", type=int, default=2, help="The number of segments to process during the generation.")
|
||||
# parser.add_argument("--stage1_use_exl2", action="store_true", help="Use exllamav2 to load and run stage 1 model.")
|
||||
# parser.add_argument("--stage2_use_exl2", action="store_true", help="Use exllamav2 to load and run stage 2 model.")
|
||||
# parser.add_argument("--stage2_batch_size", type=int, default=4, help="The non-exl2 batch size used in Stage 2 inference.")
|
||||
# parser.add_argument("--stage1_cache_size", type=int, default=16384, help="The cache size used in Stage 1 inference.")
|
||||
# parser.add_argument("--stage2_cache_size", type=int, default=8192, help="The exl2 cache size used in Stage 2 inference.")
|
||||
# parser.add_argument("--stage1_cache_mode", type=str, default="FP16", help="The cache mode used in Stage 1 inference (FP16, Q8, Q6, Q4). Quantized k/v cache will save VRAM at the cost of some speed and precision.")
|
||||
# parser.add_argument("--stage2_cache_mode", type=str, default="FP16", help="The cache mode used in Stage 2 inference (FP16, Q8, Q6, Q4). Quantized k/v cache will save VRAM at the cost of some speed and precision.")
|
||||
# parser.add_argument("--stage1_no_guidance", action="store_true", help="Disable classifier-free guidance for stage 1")
|
||||
# # Prompt
|
||||
# parser.add_argument(
|
||||
# "--genre_txt",
|
||||
# type=str,
|
||||
# required=True,
|
||||
# help="The file path to a text file containing genre tags that describe the musical style or characteristics (e.g., instrumental, genre, mood, vocal timbre, vocal gender). This is used as part of the generation prompt.",
|
||||
# )
|
||||
# parser.add_argument(
|
||||
# "--lyrics_txt",
|
||||
# type=str,
|
||||
# required=True,
|
||||
# help="The file path to a text file containing the lyrics for the music generation. These lyrics will be processed and split into structured segments to guide the generation process.",
|
||||
# )
|
||||
# parser.add_argument(
|
||||
# "--use_audio_prompt",
|
||||
# action="store_true",
|
||||
# help="If set, the model will use an audio file as a prompt during generation. The audio file should be specified using --audio_prompt_path.",
|
||||
# )
|
||||
# parser.add_argument(
|
||||
# "--audio_prompt_path", type=str, default="", help="The file path to an audio file to use as a reference prompt when --use_audio_prompt is enabled."
|
||||
# )
|
||||
# parser.add_argument("--prompt_start_time", type=float, default=0.0, help="The start time in seconds to extract the audio prompt from the given audio file.")
|
||||
# parser.add_argument("--prompt_end_time", type=float, default=30.0, help="The end time in seconds to extract the audio prompt from the given audio file.")
|
||||
# parser.add_argument(
|
||||
# "--use_dual_tracks_prompt",
|
||||
# action="store_true",
|
||||
# help="If set, the model will use dual tracks as a prompt during generation. The vocal and instrumental files should be specified using --vocal_track_prompt_path and --instrumental_track_prompt_path.",
|
||||
# )
|
||||
# parser.add_argument(
|
||||
# "--vocal_track_prompt_path",
|
||||
# type=str,
|
||||
# default="",
|
||||
# help="The file path to a vocal track file to use as a reference prompt when --use_dual_tracks_prompt is enabled.",
|
||||
# )
|
||||
# parser.add_argument(
|
||||
# "--instrumental_track_prompt_path",
|
||||
# type=str,
|
||||
# default="",
|
||||
# help="The file path to an instrumental track file to use as a reference prompt when --use_dual_tracks_prompt is enabled.",
|
||||
# )
|
||||
# # Output
|
||||
# parser.add_argument("--output_dir", type=str, default="./output", help="The directory where generated outputs will be saved.")
|
||||
# parser.add_argument("--keep_intermediate", action="store_true", help="If set, intermediate outputs will be saved during processing.")
|
||||
# parser.add_argument("--disable_offload_model", action="store_true", help="If set, the model will not be offloaded from the GPU to CPU after Stage 1 inference.")
|
||||
# parser.add_argument("--cuda_idx", type=int, default=0)
|
||||
# parser.add_argument("--seed", type=int, default=None, help="An integer value to reproduce generation.")
|
||||
# # Config for xcodec and upsampler
|
||||
# parser.add_argument("--basic_model_config", default="./xcodec_mini_infer/final_ckpt/config.yaml", help="YAML files for xcodec configurations.")
|
||||
# parser.add_argument("--resume_path", default="./xcodec_mini_infer/final_ckpt/ckpt_00360000.pth", help="Path to the xcodec checkpoint.")
|
||||
# parser.add_argument("--config_path", type=str, default="./xcodec_mini_infer/decoders/config.yaml", help="Path to Vocos config file.")
|
||||
# parser.add_argument("--vocal_decoder_path", type=str, default="./xcodec_mini_infer/decoders/decoder_131000.pth", help="Path to Vocos decoder weights.")
|
||||
# parser.add_argument("--inst_decoder_path", type=str, default="./xcodec_mini_infer/decoders/decoder_151000.pth", help="Path to Vocos decoder weights.")
|
||||
# parser.add_argument("-r", "--rescale", action="store_true", help="Rescale output to avoid clipping.")
|
||||
|
||||
|
||||
def seed_everything(seed: int = 42):
|
||||
random.seed(seed)
|
||||
np.random.seed(seed)
|
||||
torch.manual_seed(seed)
|
||||
torch.cuda.manual_seed_all(seed)
|
||||
torch.backends.cudnn.deterministic = True
|
||||
torch.backends.cudnn.benchmark = False
|
||||
|
||||
|
||||
def get_cache_class(cache_mode: str):
|
||||
match cache_mode:
|
||||
case "Q4":
|
||||
return ExLlamaV2Cache_Q4
|
||||
case "Q6":
|
||||
return ExLlamaV2Cache_Q6
|
||||
case "Q8":
|
||||
return ExLlamaV2Cache_Q8
|
||||
case _:
|
||||
return ExLlamaV2Cache
|
||||
|
||||
|
||||
class BlockTokenRangeProcessor(LogitsProcessor):
|
||||
def __init__(self, start_id, end_id):
|
||||
self.blocked_token_ids = list(range(start_id, end_id))
|
||||
|
||||
def __call__(self, input_ids, scores):
|
||||
scores[:, self.blocked_token_ids] = -float("inf")
|
||||
return scores
|
||||
@@ -1,492 +0,0 @@
|
||||
import os
|
||||
import sys
|
||||
sys.path.append(os.path.join(os.path.dirname(os.path.abspath(__file__)), 'xcodec_mini_infer'))
|
||||
sys.path.append(os.path.join(os.path.dirname(os.path.abspath(__file__)), 'xcodec_mini_infer', 'descriptaudiocodec'))
|
||||
import re
|
||||
import random
|
||||
# import uuid
|
||||
import copy
|
||||
from tqdm import tqdm
|
||||
from collections import Counter
|
||||
#import argparse
|
||||
import numpy as np
|
||||
import torch
|
||||
import torchaudio
|
||||
from torchaudio.transforms import Resample
|
||||
# import soundfile as sf
|
||||
# from einops import rearrange
|
||||
from transformers import AutoTokenizer, AutoModelForCausalLM, LogitsProcessor, LogitsProcessorList
|
||||
# from omegaconf import OmegaConf
|
||||
# from .codecmanipulator import CodecManipulator
|
||||
# from .mmtokenizer import _MMSentencePieceTokenizer
|
||||
# from .xcodec_mini_infer.models.soundstream_hubert_new import SoundStream
|
||||
# from .xcodec_mini_infer.vocoder import build_codec_model, process_audio
|
||||
# from .xcodec_mini_infer.post_process_audio import replace_low_freq_with_energy_matched
|
||||
|
||||
|
||||
# parser = argparse.ArgumentParser()
|
||||
# # Model Configuration:
|
||||
# parser.add_argument("--stage1_model", type=str, default="m-a-p/YuE-s1-7B-anneal-en-cot", help="The model checkpoint path or identifier for the Stage 1 model.")
|
||||
# parser.add_argument("--stage2_model", type=str, default="m-a-p/YuE-s2-1B-general", help="The model checkpoint path or identifier for the Stage 2 model.")
|
||||
# parser.add_argument("--max_new_tokens", type=int, default=3000, help="The maximum number of new tokens to generate in one pass during text generation.")
|
||||
# parser.add_argument("--repetition_penalty", type=float, default=1.1, help="repetition_penalty ranges from 1.0 to 2.0 (or higher in some cases). It controls the diversity and coherence of the audio tokens generated. The higher the value, the greater the discouragement of repetition. Setting value to 1.0 means no penalty.")
|
||||
# parser.add_argument("--run_n_segments", type=int, default=2, help="The number of segments to process during the generation.")
|
||||
# parser.add_argument("--stage2_batch_size", type=int, default=4, help="The batch size used in Stage 2 inference.")
|
||||
# # Prompt
|
||||
# parser.add_argument("--genre_txt", type=str, required=True, help="The file path to a text file containing genre tags that describe the musical style or characteristics (e.g., instrumental, genre, mood, vocal timbre, vocal gender). This is used as part of the generation prompt.")
|
||||
# parser.add_argument("--lyrics_txt", type=str, required=True, help="The file path to a text file containing the lyrics for the music generation. These lyrics will be processed and split into structured segments to guide the generation process.")
|
||||
# parser.add_argument("--use_audio_prompt", action="store_true", help="If set, the model will use an audio file as a prompt during generation. The audio file should be specified using --audio_prompt_path.")
|
||||
# parser.add_argument("--audio_prompt_path", type=str, default="", help="The file path to an audio file to use as a reference prompt when --use_audio_prompt is enabled.")
|
||||
# parser.add_argument("--prompt_start_time", type=float, default=0.0, help="The start time in seconds to extract the audio prompt from the given audio file.")
|
||||
# parser.add_argument("--prompt_end_time", type=float, default=30.0, help="The end time in seconds to extract the audio prompt from the given audio file.")
|
||||
# parser.add_argument("--use_dual_tracks_prompt", action="store_true", help="If set, the model will use dual tracks as a prompt during generation. The vocal and instrumental files should be specified using --vocal_track_prompt_path and --instrumental_track_prompt_path.")
|
||||
# parser.add_argument("--vocal_track_prompt_path", type=str, default="", help="The file path to a vocal track file to use as a reference prompt when --use_dual_tracks_prompt is enabled.")
|
||||
# parser.add_argument("--instrumental_track_prompt_path", type=str, default="", help="The file path to an instrumental track file to use as a reference prompt when --use_dual_tracks_prompt is enabled.")
|
||||
# # Output
|
||||
# parser.add_argument("--output_dir", type=str, default="./output", help="The directory where generated outputs will be saved.")
|
||||
# parser.add_argument("--keep_intermediate", action="store_true", help="If set, intermediate outputs will be saved during processing.")
|
||||
# parser.add_argument("--disable_offload_model", action="store_true", help="If set, the model will not be offloaded from the GPU to CPU after Stage 1 inference.")
|
||||
# parser.add_argument("--cuda_idx", type=int, default=0)
|
||||
# parser.add_argument("--seed", type=int, default=42, help="An integer value to reproduce generation.")
|
||||
# # Config for xcodec and upsampler
|
||||
# parser.add_argument('--basic_model_config', default='./xcodec_mini_infer/final_ckpt/config.yaml', help='YAML files for xcodec configurations.')
|
||||
# parser.add_argument('--resume_path', default='./xcodec_mini_infer/final_ckpt/ckpt_00360000.pth', help='Path to the xcodec checkpoint.')
|
||||
# parser.add_argument('--config_path', type=str, default='./xcodec_mini_infer/decoders/config.yaml', help='Path to Vocos config file.')
|
||||
# parser.add_argument('--vocal_decoder_path', type=str, default='./xcodec_mini_infer/decoders/decoder_131000.pth', help='Path to Vocos decoder weights.')
|
||||
# parser.add_argument('--inst_decoder_path', type=str, default='./xcodec_mini_infer/decoders/decoder_151000.pth', help='Path to Vocos decoder weights.')
|
||||
# parser.add_argument('-r', '--rescale', action='store_true', help='Rescale output to avoid clipping.')
|
||||
|
||||
|
||||
# args = parser.parse_args()
|
||||
# if args.use_audio_prompt and not args.audio_prompt_path:
|
||||
# raise FileNotFoundError("Please offer audio prompt filepath using '--audio_prompt_path', when you enable 'use_audio_prompt'!")
|
||||
# if args.use_dual_tracks_prompt and not args.vocal_track_prompt_path and not args.instrumental_track_prompt_path:
|
||||
# raise FileNotFoundError("Please offer dual tracks prompt filepath using '--vocal_track_prompt_path' and '--inst_decoder_path', when you enable '--use_dual_tracks_prompt'!")
|
||||
# stage1_model = args.stage1_model
|
||||
# stage2_model = args.stage2_model
|
||||
# cuda_idx = args.cuda_idx
|
||||
# max_new_tokens = args.max_new_tokens
|
||||
# stage1_output_dir = os.path.join(args.output_dir, f"stage1")
|
||||
# stage2_output_dir = stage1_output_dir.replace('stage1', 'stage2')
|
||||
# os.makedirs(stage1_output_dir, exist_ok=True)
|
||||
# os.makedirs(stage2_output_dir, exist_ok=True)
|
||||
def seed_everything(seed=42):
|
||||
random.seed(seed)
|
||||
np.random.seed(seed)
|
||||
torch.manual_seed(seed)
|
||||
torch.cuda.manual_seed_all(seed)
|
||||
torch.backends.cudnn.deterministic = True
|
||||
torch.backends.cudnn.benchmark = False
|
||||
# seed_everything(args.seed)
|
||||
# # load tokenizer and model
|
||||
# device = torch.device(f"cuda:{cuda_idx}" if torch.cuda.is_available() else "cpu")
|
||||
# mmtokenizer = _MMSentencePieceTokenizer("./mm_tokenizer_v0.2_hf/tokenizer.model")
|
||||
# model = AutoModelForCausalLM.from_pretrained(
|
||||
# stage1_model,
|
||||
# torch_dtype=torch.bfloat16,
|
||||
# attn_implementation="flash_attention_2", # To enable flashattn, you have to install flash-attn
|
||||
# # device_map="auto",
|
||||
# )
|
||||
# # to device, if gpu is available
|
||||
# model.to(device)
|
||||
# model.eval()
|
||||
|
||||
# if torch.__version__ >= "2.0.0":
|
||||
# model = torch.compile(model)
|
||||
|
||||
# codectool = CodecManipulator("xcodec", 0, 1)
|
||||
# codectool_stage2 = CodecManipulator("xcodec", 0, 8)
|
||||
# model_config = OmegaConf.load(args.basic_model_config)
|
||||
# codec_model = eval(model_config.generator.name)(**model_config.generator.config).to(device)
|
||||
# parameter_dict = torch.load(args.resume_path, map_location='cpu', weights_only=False)
|
||||
# codec_model.load_state_dict(parameter_dict['codec_model'])
|
||||
# codec_model.to(device)
|
||||
# codec_model.eval()
|
||||
|
||||
class BlockTokenRangeProcessor(LogitsProcessor):
|
||||
def __init__(self, start_id, end_id):
|
||||
self.blocked_token_ids = list(range(start_id, end_id))
|
||||
|
||||
def __call__(self, input_ids, scores):
|
||||
scores[:, self.blocked_token_ids] = -float("inf")
|
||||
return scores
|
||||
|
||||
def load_audio_mono(filepath, sampling_rate=16000):
|
||||
audio, sr = torchaudio.load(filepath)
|
||||
# Convert to mono
|
||||
audio = torch.mean(audio, dim=0, keepdim=True)
|
||||
# Resample if needed
|
||||
if sr != sampling_rate:
|
||||
resampler = Resample(orig_freq=sr, new_freq=sampling_rate)
|
||||
audio = resampler(audio)
|
||||
return audio
|
||||
|
||||
def encode_audio(codec_model, audio_prompt, device, target_bw=0.5):
|
||||
if len(audio_prompt.shape) < 3:
|
||||
audio_prompt.unsqueeze_(0)
|
||||
with torch.no_grad():
|
||||
raw_codes = codec_model.encode(audio_prompt.to(device), target_bw=target_bw)
|
||||
raw_codes = raw_codes.transpose(0, 1)
|
||||
raw_codes = raw_codes.cpu().numpy().astype(np.int16)
|
||||
return raw_codes
|
||||
|
||||
def split_lyrics(lyrics):
|
||||
pattern = r"\[(\w+)\](.*?)(?=\[|\Z)"
|
||||
segments = re.findall(pattern, lyrics, re.DOTALL)
|
||||
structured_lyrics = [f"[{seg[0]}]\n{seg[1].strip()}\n\n" for seg in segments]
|
||||
return structured_lyrics
|
||||
|
||||
# Call the function and print the result
|
||||
# stage1_output_set = []
|
||||
# # Tips:
|
||||
# # genre tags support instrumental,genre,mood,vocal timbr and vocal gender
|
||||
# # all kinds of tags are needed
|
||||
# with open(args.genre_txt) as f:
|
||||
# genres = f.read().strip()
|
||||
# with open(args.lyrics_txt) as f:
|
||||
# lyrics = split_lyrics(f.read())
|
||||
# # intruction
|
||||
# full_lyrics = "\n".join(lyrics)
|
||||
# prompt_texts = [f"Generate music from the given lyrics segment by segment.\n[Genre] {genres}\n{full_lyrics}"]
|
||||
# prompt_texts += lyrics
|
||||
|
||||
|
||||
# random_id = uuid.uuid4()
|
||||
# output_seq = None
|
||||
# # Here is suggested decoding config
|
||||
# top_p = 0.93
|
||||
# temperature = 1.0
|
||||
# repetition_penalty = args.repetition_penalty
|
||||
# # special tokens
|
||||
# start_of_segment = mmtokenizer.tokenize('[start_of_segment]')
|
||||
# end_of_segment = mmtokenizer.tokenize('[end_of_segment]')
|
||||
# # Format text prompt
|
||||
# run_n_segments = min(args.run_n_segments+1, len(lyrics))
|
||||
# for i, p in enumerate(tqdm(prompt_texts[:run_n_segments], desc="Stage1 inference...")):
|
||||
# section_text = p.replace('[start_of_segment]', '').replace('[end_of_segment]', '')
|
||||
# guidance_scale = 1.5 if i <=1 else 1.2
|
||||
# if i==0:
|
||||
# continue
|
||||
# if i==1:
|
||||
# if args.use_dual_tracks_prompt or args.use_audio_prompt:
|
||||
# if args.use_dual_tracks_prompt:
|
||||
# vocals_ids = load_audio_mono(args.vocal_track_prompt_path)
|
||||
# instrumental_ids = load_audio_mono(args.instrumental_track_prompt_path)
|
||||
# vocals_ids = encode_audio(codec_model, vocals_ids, device, target_bw=0.5)
|
||||
# instrumental_ids = encode_audio(codec_model, instrumental_ids, device, target_bw=0.5)
|
||||
# vocals_ids = codectool.npy2ids(vocals_ids[0])
|
||||
# instrumental_ids = codectool.npy2ids(instrumental_ids[0])
|
||||
# ids_segment_interleaved = rearrange([np.array(vocals_ids), np.array(instrumental_ids)], 'b n -> (n b)')
|
||||
# audio_prompt_codec = ids_segment_interleaved[int(args.prompt_start_time*50*2): int(args.prompt_end_time*50*2)]
|
||||
# audio_prompt_codec = audio_prompt_codec.tolist()
|
||||
# elif args.use_audio_prompt:
|
||||
# audio_prompt = load_audio_mono(args.audio_prompt_path)
|
||||
# raw_codes = encode_audio(codec_model, audio_prompt, device, target_bw=0.5)
|
||||
# # Format audio prompt
|
||||
# code_ids = codectool.npy2ids(raw_codes[0])
|
||||
# audio_prompt_codec = code_ids[int(args.prompt_start_time *50): int(args.prompt_end_time *50)] # 50 is tps of xcodec
|
||||
# audio_prompt_codec_ids = [mmtokenizer.soa] + codectool.sep_ids + audio_prompt_codec + [mmtokenizer.eoa]
|
||||
# sentence_ids = mmtokenizer.tokenize("[start_of_reference]") + audio_prompt_codec_ids + mmtokenizer.tokenize("[end_of_reference]")
|
||||
# head_id = mmtokenizer.tokenize(prompt_texts[0]) + sentence_ids
|
||||
# else:
|
||||
# head_id = mmtokenizer.tokenize(prompt_texts[0])
|
||||
# prompt_ids = head_id + start_of_segment + mmtokenizer.tokenize(section_text) + [mmtokenizer.soa] + codectool.sep_ids
|
||||
# else:
|
||||
# prompt_ids = end_of_segment + start_of_segment + mmtokenizer.tokenize(section_text) + [mmtokenizer.soa] + codectool.sep_ids
|
||||
|
||||
# prompt_ids = torch.as_tensor(prompt_ids).unsqueeze(0).to(device)
|
||||
# input_ids = torch.cat([raw_output, prompt_ids], dim=1) if i > 1 else prompt_ids
|
||||
# # Use window slicing in case output sequence exceeds the context of model
|
||||
# max_context = 16384-max_new_tokens-1
|
||||
# if input_ids.shape[-1] > max_context:
|
||||
# print(f'Section {i}: output length {input_ids.shape[-1]} exceeding context length {max_context}, now using the last {max_context} tokens.')
|
||||
# input_ids = input_ids[:, -(max_context):]
|
||||
# with torch.no_grad():
|
||||
# output_seq = model.generate(
|
||||
# input_ids=input_ids,
|
||||
# max_new_tokens=max_new_tokens,
|
||||
# min_new_tokens=100,
|
||||
# do_sample=True,
|
||||
# top_p=top_p,
|
||||
# temperature=temperature,
|
||||
# repetition_penalty=repetition_penalty,
|
||||
# eos_token_id=mmtokenizer.eoa,
|
||||
# pad_token_id=mmtokenizer.eoa,
|
||||
# logits_processor=LogitsProcessorList([BlockTokenRangeProcessor(0, 32002), BlockTokenRangeProcessor(32016, 32016)]),
|
||||
# guidance_scale=guidance_scale,
|
||||
# )
|
||||
# if output_seq[0][-1].item() != mmtokenizer.eoa:
|
||||
# tensor_eoa = torch.as_tensor([[mmtokenizer.eoa]]).to(model.device)
|
||||
# output_seq = torch.cat((output_seq, tensor_eoa), dim=1)
|
||||
# if i > 1:
|
||||
# raw_output = torch.cat([raw_output, prompt_ids, output_seq[:, input_ids.shape[-1]:]], dim=1)
|
||||
# else:
|
||||
# raw_output = output_seq
|
||||
|
||||
# # save raw output and check sanity
|
||||
# ids = raw_output[0].cpu().numpy()
|
||||
# soa_idx = np.where(ids == mmtokenizer.soa)[0].tolist()
|
||||
# eoa_idx = np.where(ids == mmtokenizer.eoa)[0].tolist()
|
||||
# if len(soa_idx)!=len(eoa_idx):
|
||||
# raise ValueError(f'invalid pairs of soa and eoa, Num of soa: {len(soa_idx)}, Num of eoa: {len(eoa_idx)}')
|
||||
|
||||
# vocals = []
|
||||
# instrumentals = []
|
||||
# range_begin = 1 if args.use_audio_prompt or args.use_dual_tracks_prompt else 0
|
||||
# for i in range(range_begin, len(soa_idx)):
|
||||
# codec_ids = ids[soa_idx[i]+1:eoa_idx[i]]
|
||||
# if codec_ids[0] == 32016:
|
||||
# codec_ids = codec_ids[1:]
|
||||
# codec_ids = codec_ids[:2 * (codec_ids.shape[0] // 2)]
|
||||
# vocals_ids = codectool.ids2npy(rearrange(codec_ids,"(n b) -> b n", b=2)[0])
|
||||
# vocals.append(vocals_ids)
|
||||
# instrumentals_ids = codectool.ids2npy(rearrange(codec_ids,"(n b) -> b n", b=2)[1])
|
||||
# instrumentals.append(instrumentals_ids)
|
||||
# vocals = np.concatenate(vocals, axis=1)
|
||||
# instrumentals = np.concatenate(instrumentals, axis=1)
|
||||
# vocal_save_path = os.path.join(stage1_output_dir, f"{genres.replace(' ', '-')}_tp{top_p}_T{temperature}_rp{repetition_penalty}_maxtk{max_new_tokens}_{random_id}_vtrack".replace('.', '@')+'.npy')
|
||||
# inst_save_path = os.path.join(stage1_output_dir, f"{genres.replace(' ', '-')}_tp{top_p}_T{temperature}_rp{repetition_penalty}_maxtk{max_new_tokens}_{random_id}_itrack".replace('.', '@')+'.npy')
|
||||
# np.save(vocal_save_path, vocals)
|
||||
# np.save(inst_save_path, instrumentals)
|
||||
# stage1_output_set.append(vocal_save_path)
|
||||
# stage1_output_set.append(inst_save_path)
|
||||
|
||||
|
||||
# # offload model
|
||||
# if not args.disable_offload_model:
|
||||
# model.cpu()
|
||||
# del model
|
||||
# torch.cuda.empty_cache()
|
||||
|
||||
# print("Stage 2 inference...")
|
||||
# model_stage2 = AutoModelForCausalLM.from_pretrained(
|
||||
# stage2_model,
|
||||
# torch_dtype=torch.bfloat16,
|
||||
# attn_implementation="flash_attention_2",
|
||||
# # device_map="auto",
|
||||
# )
|
||||
# model_stage2.to(device)
|
||||
# model_stage2.eval()
|
||||
|
||||
# if torch.__version__ >= "2.0.0":
|
||||
# model_stage2 = torch.compile(model_stage2)
|
||||
|
||||
def stage2_generate(model,mmtokenizer,codectool,device, prompt, batch_size=16):
|
||||
codec_ids = codectool.unflatten(prompt, n_quantizer=1)
|
||||
codec_ids = codectool.offset_tok_ids(
|
||||
codec_ids,
|
||||
global_offset=codectool.global_offset,
|
||||
codebook_size=codectool.codebook_size,
|
||||
num_codebooks=codectool.num_codebooks,
|
||||
).astype(np.int32)
|
||||
|
||||
# Prepare prompt_ids based on batch size or single input
|
||||
if batch_size > 1:
|
||||
codec_list = []
|
||||
for i in range(batch_size):
|
||||
idx_begin = i * 300
|
||||
idx_end = (i + 1) * 300
|
||||
codec_list.append(codec_ids[:, idx_begin:idx_end])
|
||||
|
||||
codec_ids = np.concatenate(codec_list, axis=0)
|
||||
prompt_ids = np.concatenate(
|
||||
[
|
||||
np.tile([mmtokenizer.soa, mmtokenizer.stage_1], (batch_size, 1)),
|
||||
codec_ids,
|
||||
np.tile([mmtokenizer.stage_2], (batch_size, 1)),
|
||||
],
|
||||
axis=1
|
||||
)
|
||||
else:
|
||||
prompt_ids = np.concatenate([
|
||||
np.array([mmtokenizer.soa, mmtokenizer.stage_1]),
|
||||
codec_ids.flatten(), # Flatten the 2D array to 1D
|
||||
np.array([mmtokenizer.stage_2])
|
||||
]).astype(np.int32)
|
||||
prompt_ids = prompt_ids[np.newaxis, ...]
|
||||
|
||||
codec_ids = torch.as_tensor(codec_ids).to(device)
|
||||
prompt_ids = torch.as_tensor(prompt_ids).to(device)
|
||||
len_prompt = prompt_ids.shape[-1]
|
||||
|
||||
block_list = LogitsProcessorList([BlockTokenRangeProcessor(0, 46358), BlockTokenRangeProcessor(53526, mmtokenizer.vocab_size)])
|
||||
|
||||
# Teacher forcing generate loop
|
||||
for frames_idx in range(codec_ids.shape[1]):
|
||||
cb0 = codec_ids[:, frames_idx:frames_idx+1]
|
||||
prompt_ids = torch.cat([prompt_ids, cb0], dim=1)
|
||||
input_ids = prompt_ids
|
||||
|
||||
with torch.no_grad():
|
||||
stage2_output = model.generate(input_ids=input_ids,
|
||||
min_new_tokens=7,
|
||||
max_new_tokens=7,
|
||||
eos_token_id=mmtokenizer.eoa,
|
||||
pad_token_id=mmtokenizer.eoa,
|
||||
logits_processor=block_list,
|
||||
)
|
||||
|
||||
assert stage2_output.shape[1] - prompt_ids.shape[1] == 7, f"output new tokens={stage2_output.shape[1]-prompt_ids.shape[1]}"
|
||||
prompt_ids = stage2_output
|
||||
|
||||
# Return output based on batch size
|
||||
if batch_size > 1:
|
||||
output = prompt_ids.cpu().numpy()[:, len_prompt:]
|
||||
output_list = [output[i] for i in range(batch_size)]
|
||||
output = np.concatenate(output_list, axis=0)
|
||||
else:
|
||||
output = prompt_ids[0].cpu().numpy()[len_prompt:]
|
||||
|
||||
return output
|
||||
|
||||
def stage2_inference(model, stage1_output_set, stage2_output_dir,codectool_stage2,mmtokenizer,codectool,device, batch_size=4):
|
||||
stage2_result = []
|
||||
for i in tqdm(range(len(stage1_output_set))):
|
||||
output_filename = os.path.join(stage2_output_dir, os.path.basename(stage1_output_set[i]))
|
||||
|
||||
if os.path.exists(output_filename):
|
||||
print(f'{output_filename} stage2 has done.')
|
||||
continue
|
||||
|
||||
# Load the prompt
|
||||
prompt = np.load(stage1_output_set[i]).astype(np.int32)
|
||||
|
||||
# Only accept 6s segments
|
||||
output_duration = prompt.shape[-1] // 50 // 6 * 6
|
||||
num_batch = output_duration // 6
|
||||
|
||||
if num_batch <= batch_size:
|
||||
# If num_batch is less than or equal to batch_size, we can infer the entire prompt at once
|
||||
output = stage2_generate(model,mmtokenizer,codectool,device, prompt[:, :output_duration*50], batch_size=num_batch)
|
||||
else:
|
||||
# If num_batch is greater than batch_size, process in chunks of batch_size
|
||||
segments = []
|
||||
num_segments = (num_batch // batch_size) + (1 if num_batch % batch_size != 0 else 0)
|
||||
|
||||
for seg in range(num_segments):
|
||||
start_idx = seg * batch_size * 300
|
||||
# Ensure the end_idx does not exceed the available length
|
||||
end_idx = min((seg + 1) * batch_size * 300, output_duration*50) # Adjust the last segment
|
||||
current_batch_size = batch_size if seg != num_segments-1 or num_batch % batch_size == 0 else num_batch % batch_size
|
||||
segment = stage2_generate(
|
||||
model,mmtokenizer,codectool,device,
|
||||
prompt[:, start_idx:end_idx],
|
||||
batch_size=current_batch_size
|
||||
)
|
||||
segments.append(segment)
|
||||
|
||||
# Concatenate all the segments
|
||||
output = np.concatenate(segments, axis=0)
|
||||
|
||||
# Process the ending part of the prompt
|
||||
if output_duration*50 != prompt.shape[-1]:
|
||||
ending = stage2_generate(model,mmtokenizer,codectool,device, prompt[:, output_duration*50:], batch_size=1)
|
||||
output = np.concatenate([output, ending], axis=0)
|
||||
output = codectool_stage2.ids2npy(output)
|
||||
|
||||
# Fix invalid codes (a dirty solution, which may harm the quality of audio)
|
||||
# We are trying to find better one
|
||||
fixed_output = copy.deepcopy(output)
|
||||
for i, line in enumerate(output):
|
||||
for j, element in enumerate(line):
|
||||
if element < 0 or element > 1023:
|
||||
counter = Counter(line)
|
||||
most_frequant = sorted(counter.items(), key=lambda x: x[1], reverse=True)[0][0]
|
||||
fixed_output[i, j] = most_frequant
|
||||
# save output
|
||||
np.save(output_filename, fixed_output)
|
||||
stage2_result.append(output_filename)
|
||||
return stage2_result
|
||||
|
||||
# stage2_result = stage2_inference(model_stage2, stage1_output_set, stage2_output_dir, batch_size=args.stage2_batch_size)
|
||||
# print(stage2_result)
|
||||
# print('Stage 2 DONE.\n')
|
||||
# # convert audio tokens to audio
|
||||
def save_audio(wav: torch.Tensor, path, sample_rate: int, rescale: bool = False):
|
||||
folder_path = os.path.dirname(path)
|
||||
if not os.path.exists(folder_path):
|
||||
os.makedirs(folder_path)
|
||||
limit = 0.99
|
||||
max_val = wav.abs().max()
|
||||
wav = wav * min(limit / max_val, 1) if rescale else wav.clamp(-limit, limit)
|
||||
wav = wav.cpu()
|
||||
torchaudio.save(str(path), wav, sample_rate=sample_rate, encoding='PCM_S', bits_per_sample=16)
|
||||
# # reconstruct tracks
|
||||
# recons_output_dir = os.path.join(args.output_dir, "recons")
|
||||
# recons_mix_dir = os.path.join(recons_output_dir, 'mix')
|
||||
# os.makedirs(recons_mix_dir, exist_ok=True)
|
||||
# tracks = []
|
||||
# for npy in stage2_result:
|
||||
# codec_result = np.load(npy)
|
||||
# decodec_rlt=[]
|
||||
# with torch.no_grad():
|
||||
# decoded_waveform = codec_model.decode(torch.as_tensor(codec_result.astype(np.int16), dtype=torch.long).unsqueeze(0).permute(1, 0, 2).to(device))
|
||||
# decoded_waveform = decoded_waveform.cpu().squeeze(0)
|
||||
# decodec_rlt.append(torch.as_tensor(decoded_waveform))
|
||||
# decodec_rlt = torch.cat(decodec_rlt, dim=-1)
|
||||
# save_path = os.path.join(recons_output_dir, os.path.splitext(os.path.basename(npy))[0] + ".mp3")
|
||||
# tracks.append(save_path)
|
||||
# save_audio(decodec_rlt, save_path, 16000)
|
||||
# # mix tracks
|
||||
# for inst_path in tracks:
|
||||
# try:
|
||||
# if (inst_path.endswith('.wav') or inst_path.endswith('.mp3')) \
|
||||
# and '_itrack' in inst_path:
|
||||
# # find pair
|
||||
# vocal_path = inst_path.replace('_itrack', '_vtrack')
|
||||
# if not os.path.exists(vocal_path):
|
||||
# continue
|
||||
# # mix
|
||||
# recons_mix = os.path.join(recons_mix_dir, os.path.basename(inst_path).replace('_itrack', '_mixed'))
|
||||
# vocal_stem, sr = sf.read(inst_path)
|
||||
# instrumental_stem, _ = sf.read(vocal_path)
|
||||
# mix_stem = (vocal_stem + instrumental_stem) / 1
|
||||
# sf.write(recons_mix, mix_stem, sr)
|
||||
# except Exception as e:
|
||||
# print(e)
|
||||
|
||||
# # vocoder to upsample audios
|
||||
# vocal_decoder, inst_decoder = build_codec_model(args.config_path, args.vocal_decoder_path, args.inst_decoder_path)
|
||||
# vocoder_output_dir = os.path.join(args.output_dir, 'vocoder')
|
||||
# vocoder_stems_dir = os.path.join(vocoder_output_dir, 'stems')
|
||||
# vocoder_mix_dir = os.path.join(vocoder_output_dir, 'mix')
|
||||
# os.makedirs(vocoder_mix_dir, exist_ok=True)
|
||||
# os.makedirs(vocoder_stems_dir, exist_ok=True)
|
||||
# for npy in stage2_result:
|
||||
# if '_itrack' in npy:
|
||||
# # Process instrumental
|
||||
# instrumental_output = process_audio(
|
||||
# npy,
|
||||
# os.path.join(vocoder_stems_dir, 'itrack.mp3'),
|
||||
# args.rescale,
|
||||
# args,
|
||||
# inst_decoder,
|
||||
# codec_model
|
||||
# )
|
||||
# else:
|
||||
# # Process vocal
|
||||
# vocal_output = process_audio(
|
||||
# npy,
|
||||
# os.path.join(vocoder_stems_dir, 'vtrack.mp3'),
|
||||
# args.rescale,
|
||||
# args,
|
||||
# vocal_decoder,
|
||||
# codec_model
|
||||
# )
|
||||
# # mix tracks
|
||||
# try:
|
||||
# mix_output = instrumental_output + vocal_output
|
||||
# vocoder_mix = os.path.join(vocoder_mix_dir, os.path.basename(recons_mix))
|
||||
# save_audio(mix_output, vocoder_mix, 44100, args.rescale)
|
||||
# print(f"Created mix: {vocoder_mix}")
|
||||
# except RuntimeError as e:
|
||||
# print(e)
|
||||
# print(f"mix {vocoder_mix} failed! inst: {instrumental_output.shape}, vocal: {vocal_output.shape}")
|
||||
|
||||
# # Post process
|
||||
# replace_low_freq_with_energy_matched(
|
||||
# a_file=recons_mix, # 16kHz
|
||||
# b_file=vocoder_mix, # 48kHz
|
||||
# c_file=os.path.join(args.output_dir, os.path.basename(recons_mix)),
|
||||
# cutoff_freq=5500.0
|
||||
# )
|
||||
@@ -1,112 +0,0 @@
|
||||
import os
|
||||
|
||||
import numpy as np
|
||||
import soundfile as sf
|
||||
import torch
|
||||
import torchaudio
|
||||
from .common import seed_everything
|
||||
from .xcodec_mini_infer.models.soundstream_hubert_new import SoundStream
|
||||
from omegaconf import OmegaConf
|
||||
from .xcodec_mini_infer.post_process_audio import replace_low_freq_with_energy_matched
|
||||
from .xcodec_mini_infer.vocoder import build_codec_model, process_audio
|
||||
|
||||
|
||||
# convert audio tokens to audio
|
||||
def save_audio(wav: torch.Tensor, path, sample_rate: int, rescale: bool = False):
|
||||
folder_path = os.path.dirname(path)
|
||||
if not os.path.exists(folder_path):
|
||||
os.makedirs(folder_path)
|
||||
limit = 0.99
|
||||
max_val = wav.abs().max()
|
||||
wav = wav * min(limit / max_val, 1) if rescale else wav.clamp(-limit, limit)
|
||||
torchaudio.save(str(path), wav, sample_rate=sample_rate, encoding="PCM_S", bits_per_sample=16)
|
||||
|
||||
|
||||
def post_process(
|
||||
codec_model: SoundStream, device: torch.device, output_dir: str, config_path: str, vocal_decoder_path: str, inst_decoder_path: str, rescale: bool,
|
||||
file_prefix,):
|
||||
# reconstruct tracks
|
||||
recons_output_dir = os.path.join(output_dir, "recons")
|
||||
recons_mix_dir = os.path.join(recons_output_dir, "mix")
|
||||
os.makedirs(recons_mix_dir, exist_ok=True)
|
||||
stage2_result = [os.path.join(output_dir, "stage2", filename) for filename in ["vtrack.npy", "itrack.npy"]]
|
||||
tracks = []
|
||||
for npy in stage2_result:
|
||||
codec_result = np.load(npy)
|
||||
decodec_rlt = []
|
||||
decoded_waveform = codec_model.decode(torch.as_tensor(codec_result.astype(np.int16), dtype=torch.long).unsqueeze(0).permute(1, 0, 2).to(device))
|
||||
decoded_waveform = decoded_waveform.cpu().squeeze(0)
|
||||
decodec_rlt.append(torch.as_tensor(decoded_waveform))
|
||||
decodec_rlt = torch.cat(decodec_rlt, dim=-1)
|
||||
save_path = os.path.join(recons_output_dir, os.path.splitext(os.path.basename(npy))[0] + ".mp3")
|
||||
tracks.append(save_path)
|
||||
save_audio(decodec_rlt, save_path, 16000)
|
||||
# mix tracks
|
||||
for inst_path in tracks:
|
||||
try:
|
||||
if (inst_path.endswith(".wav") or inst_path.endswith(".mp3")) and "itrack" in inst_path:
|
||||
# find pair
|
||||
vocal_path = inst_path.replace("itrack", "vtrack")
|
||||
if not os.path.exists(vocal_path):
|
||||
continue
|
||||
# mix
|
||||
recons_mix = os.path.join(recons_mix_dir, os.path.basename(inst_path).replace("itrack", "mixed"))
|
||||
vocal_stem, sr = sf.read(inst_path)
|
||||
instrumental_stem, _ = sf.read(vocal_path)
|
||||
mix_stem = (vocal_stem + instrumental_stem) / 1
|
||||
sf.write(recons_mix, mix_stem, sr)
|
||||
except Exception as e:
|
||||
print(e)
|
||||
|
||||
# vocoder to upsample audios
|
||||
vocal_decoder, inst_decoder = build_codec_model(config_path, vocal_decoder_path, inst_decoder_path)
|
||||
vocoder_output_dir = os.path.join(output_dir, "vocoder")
|
||||
vocoder_stems_dir = os.path.join(vocoder_output_dir, "stems")
|
||||
vocoder_mix_dir = os.path.join(vocoder_output_dir, "mix")
|
||||
os.makedirs(vocoder_mix_dir, exist_ok=True)
|
||||
os.makedirs(vocoder_stems_dir, exist_ok=True)
|
||||
for npy in stage2_result:
|
||||
if "itrack" in npy:
|
||||
# Process instrumental
|
||||
instrumental_output = process_audio(npy, os.path.join(vocoder_stems_dir, "itrack.mp3"), rescale, device, inst_decoder, codec_model)
|
||||
else:
|
||||
# Process vocal
|
||||
vocal_output = process_audio(npy, os.path.join(vocoder_stems_dir, "vtrack.mp3"), rescale, device, vocal_decoder, codec_model)
|
||||
# mix tracks
|
||||
try:
|
||||
mix_output = instrumental_output + vocal_output
|
||||
vocoder_mix = os.path.join(vocoder_mix_dir, os.path.basename(recons_mix))
|
||||
save_audio(mix_output, vocoder_mix, 44100, rescale)
|
||||
print(f"Created mix: {vocoder_mix}")
|
||||
except RuntimeError as e:
|
||||
print(e)
|
||||
print(f"mix {vocoder_mix} failed! inst: {instrumental_output.shape}, vocal: {vocal_output.shape}")
|
||||
|
||||
# Post process
|
||||
c_file=os.path.join(output_dir, f"yue_{file_prefix}_{os.path.basename(recons_mix)}")
|
||||
replace_low_freq_with_energy_matched(
|
||||
a_file=recons_mix, b_file=vocoder_mix, c_file=c_file, cutoff_freq=5500.0 # 16kHz # 48kHz
|
||||
)
|
||||
return mix_output,c_file
|
||||
|
||||
def main():
|
||||
args = parser.parse_args()
|
||||
if args.seed is not None:
|
||||
seed_everything(args.seed)
|
||||
|
||||
device = torch.device(f"cuda:{args.cuda_idx}" if torch.cuda.is_available() else "cpu")
|
||||
model_config = OmegaConf.load(args.basic_model_config)
|
||||
assert model_config.generator.name == "SoundStream"
|
||||
codec_model = SoundStream(**model_config.generator.config).to(device)
|
||||
parameter_dict = torch.load(args.resume_path, map_location=device, weights_only=False)
|
||||
codec_model.load_state_dict(parameter_dict["codec_model"])
|
||||
codec_model.eval()
|
||||
|
||||
post_process(codec_model, device, args.output_dir, args.config_path, args.vocal_decoder_path, args.inst_decoder_path, args.rescale)
|
||||
|
||||
|
||||
# if __name__ == "__main__":
|
||||
# # enable inference mode globally
|
||||
# torch.autograd.grad_mode._enter_inference_mode(True)
|
||||
# torch.autograd.set_grad_enabled(False)
|
||||
# main()
|
||||
@@ -1,511 +0,0 @@
|
||||
import os
|
||||
import random
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
import torch.nn.functional as F
|
||||
import torchaudio
|
||||
from .codecmanipulator import CodecManipulator
|
||||
from .common import BlockTokenRangeProcessor, seed_everything, get_cache_class
|
||||
from einops import rearrange
|
||||
from exllamav2 import ExLlamaV2, ExLlamaV2Config, ExLlamaV2Tokenizer
|
||||
from exllamav2.generator import ExLlamaV2Sampler
|
||||
from .mmtokenizer import _MMSentencePieceTokenizer
|
||||
from .xcodec_mini_infer.models.soundstream_hubert_new import SoundStream
|
||||
from omegaconf import OmegaConf
|
||||
from torchaudio.transforms import Resample
|
||||
from tqdm import tqdm
|
||||
from transformers import AutoModelForCausalLM, LogitsProcessorList
|
||||
from transformers.cache_utils import StaticCache
|
||||
import folder_paths
|
||||
|
||||
@dataclass
|
||||
class SampleSettings:
|
||||
# Here is suggested decoding config
|
||||
top_p = 0.93
|
||||
temperature = 1
|
||||
repetition_penalty = 1.1
|
||||
guidance_scale_seg0 = 1.5 # None to disable cfg
|
||||
guidance_scale = 1.2 # None to disable cfg
|
||||
|
||||
def __init__(self, use_guidance: bool = True, repetition_penalty: float = 1.1):
|
||||
if not use_guidance:
|
||||
self.guidance_scale_seg0 = None
|
||||
self.guidance_scale = None
|
||||
self.repetition_penalty = repetition_penalty
|
||||
|
||||
|
||||
def load_audio_mono(filepath, sampling_rate=16000):
|
||||
audio, sr = torchaudio.load(filepath)
|
||||
# Convert to mono
|
||||
audio = torch.mean(audio, dim=0, keepdim=True)
|
||||
# Resample if needed
|
||||
if sr != sampling_rate:
|
||||
resampler = Resample(orig_freq=sr, new_freq=sampling_rate)
|
||||
audio = resampler(audio)
|
||||
return audio
|
||||
|
||||
|
||||
def encode_audio(codec_model, audio_prompt, device, target_bw=0.5):
|
||||
if len(audio_prompt.shape) < 3:
|
||||
audio_prompt.unsqueeze_(0)
|
||||
with torch.no_grad():
|
||||
raw_codes = codec_model.encode(audio_prompt.to(device), target_bw=target_bw)
|
||||
raw_codes = raw_codes.transpose(0, 1)
|
||||
raw_codes = raw_codes.cpu().numpy().astype(np.int16)
|
||||
return raw_codes
|
||||
|
||||
|
||||
class Stage1Pipeline:
|
||||
|
||||
def __init__(self, device: torch.device, basic_model_config: str, resume_path: str):
|
||||
self.device = device
|
||||
self.codec_tool = CodecManipulator("xcodec", 0, 1)
|
||||
self.basic_model_config = basic_model_config
|
||||
self.resume_path = resume_path
|
||||
self.codec_model = None
|
||||
|
||||
# Load tokenizer
|
||||
self.mmtokenizer = _MMSentencePieceTokenizer(os.path.join(folder_paths.base_path,"custom_nodes/ComfyUI_YuE/inference/mm_tokenizer_v0.2_hf/tokenizer.model"))
|
||||
self.start_of_segment = self.mmtokenizer.tokenize("[start_of_segment]")
|
||||
self.end_of_segment = self.mmtokenizer.tokenize("[end_of_segment]")
|
||||
|
||||
def load_codec_model(self):
|
||||
if self.codec_model is not None:
|
||||
return
|
||||
model_config = OmegaConf.load(self.basic_model_config)
|
||||
assert model_config.generator.name == "SoundStream"
|
||||
self.codec_model = SoundStream(**model_config.generator.config).to(self.device)
|
||||
parameter_dict = torch.load(self.resume_path, map_location=self.device, weights_only=False)
|
||||
self.codec_model.load_state_dict(parameter_dict["codec_model"])
|
||||
self.codec_model.eval()
|
||||
|
||||
def get_prompt_texts(self, genres: str, lyrics: str):
|
||||
def split_lyrics(lyrics):
|
||||
pattern = r"\[(\w+)\](.*?)(?=\[|\Z)"
|
||||
segments = re.findall(pattern, lyrics, re.DOTALL)
|
||||
structured_lyrics = [f"[{seg[0]}]\n{seg[1].strip()}\n\n" for seg in segments]
|
||||
return structured_lyrics
|
||||
|
||||
lyrics = split_lyrics(lyrics)
|
||||
full_lyrics = "\n".join(lyrics)
|
||||
|
||||
prompt_texts = [f"Generate music from the given lyrics segment by segment.\n[Genre] {genres}\n{full_lyrics}"]
|
||||
print(f"Full lyrics: {prompt_texts}")
|
||||
prompt_texts += lyrics
|
||||
return lyrics, prompt_texts
|
||||
|
||||
def get_audio_prompt_ids(
|
||||
self,
|
||||
use_dual_tracks_prompt: bool,
|
||||
vocal_track_prompt_path: str,
|
||||
instrumental_track_prompt_path: str,
|
||||
use_audio_prompt: bool,
|
||||
audio_prompt_path: str,
|
||||
prompt_start_time: int,
|
||||
prompt_end_time: int,
|
||||
):
|
||||
self.load_codec_model()
|
||||
if use_dual_tracks_prompt:
|
||||
vocals_ids = load_audio_mono(vocal_track_prompt_path)
|
||||
instrumental_ids = load_audio_mono(instrumental_track_prompt_path)
|
||||
vocals_ids = encode_audio(self.codec_model, vocals_ids, self.device, target_bw=0.5)
|
||||
instrumental_ids = encode_audio(self.codec_model, instrumental_ids, self.device, target_bw=0.5)
|
||||
vocals_ids = self.codec_tool.npy2ids(vocals_ids[0])
|
||||
instrumental_ids = self.codec_tool.npy2ids(instrumental_ids[0])
|
||||
ids_segment_interleaved = rearrange([np.array(vocals_ids), np.array(instrumental_ids)], "b n -> (n b)")
|
||||
audio_prompt_codec = ids_segment_interleaved[int(prompt_start_time * 50 * 2) : int(prompt_end_time * 50 * 2)]
|
||||
audio_prompt_codec = audio_prompt_codec.tolist()
|
||||
elif use_audio_prompt:
|
||||
audio_prompt = load_audio_mono(audio_prompt_path)
|
||||
raw_codes = encode_audio(self.codec_model, audio_prompt, self.device, target_bw=0.5)
|
||||
# Format audio prompt
|
||||
code_ids = self.codec_tool.npy2ids(raw_codes[0])
|
||||
audio_prompt_codec = code_ids[int(prompt_start_time * 50) : int(prompt_end_time * 50)] # 50 is tps of xcodec
|
||||
audio_prompt_codec_ids = [self.mmtokenizer.soa] + self.codec_tool.sep_ids + audio_prompt_codec + [self.mmtokenizer.eoa]
|
||||
sentence_ids = self.mmtokenizer.tokenize("[start_of_reference]") + audio_prompt_codec_ids + self.mmtokenizer.tokenize("[end_of_reference]")
|
||||
return sentence_ids
|
||||
|
||||
def get_first_segment_prompt(
|
||||
self,
|
||||
segment_p: str,
|
||||
prompt_text_0: str,
|
||||
use_dual_tracks_prompt: bool,
|
||||
vocal_track_prompt_path: str,
|
||||
instrumental_track_prompt_path: str,
|
||||
use_audio_prompt: bool,
|
||||
audio_prompt_path: str,
|
||||
prompt_start_time: int,
|
||||
prompt_end_time: int,
|
||||
):
|
||||
section_text = segment_p.replace("[start_of_segment]", "").replace("[end_of_segment]", "")
|
||||
head_id = self.mmtokenizer.tokenize(prompt_text_0)
|
||||
if use_dual_tracks_prompt or use_audio_prompt:
|
||||
head_id += self.get_audio_prompt_ids(
|
||||
use_dual_tracks_prompt,
|
||||
vocal_track_prompt_path,
|
||||
instrumental_track_prompt_path,
|
||||
use_audio_prompt,
|
||||
audio_prompt_path,
|
||||
prompt_start_time,
|
||||
prompt_end_time,
|
||||
)
|
||||
return head_id + self.start_of_segment + self.mmtokenizer.tokenize(section_text) + [self.mmtokenizer.soa] + self.codec_tool.sep_ids
|
||||
|
||||
def get_segment_prompt(self, segment_p: str):
|
||||
section_text = segment_p.replace("[start_of_segment]", "").replace("[end_of_segment]", "")
|
||||
return self.end_of_segment + self.start_of_segment + self.mmtokenizer.tokenize(section_text) + [self.mmtokenizer.soa] + self.codec_tool.sep_ids
|
||||
|
||||
def save(self, raw_output: torch.Tensor, output_dir: str, use_audio_prompt: bool, use_dual_tracks_prompt: bool):
|
||||
# save raw output and check sanity
|
||||
ids = raw_output[0].cpu().numpy()
|
||||
soa_idx = np.where(ids == self.mmtokenizer.soa)[0].tolist()
|
||||
eoa_idx = np.where(ids == self.mmtokenizer.eoa)[0].tolist()
|
||||
if len(soa_idx) != len(eoa_idx):
|
||||
raise ValueError(f"invalid pairs of soa and eoa, Num of soa: {len(soa_idx)}, Num of eoa: {len(eoa_idx)}")
|
||||
|
||||
vocals = []
|
||||
instrumentals = []
|
||||
range_begin = 1 if use_audio_prompt or use_dual_tracks_prompt else 0
|
||||
for i in range(range_begin, len(soa_idx)):
|
||||
codec_ids = ids[soa_idx[i] + 1 : eoa_idx[i]]
|
||||
if codec_ids[0] == 32016:
|
||||
codec_ids = codec_ids[1:]
|
||||
codec_ids = codec_ids[: 2 * (codec_ids.shape[0] // 2)]
|
||||
vocals_ids = self.codec_tool.ids2npy(rearrange(codec_ids, "(n b) -> b n", b=2)[0])
|
||||
vocals.append(vocals_ids)
|
||||
instrumentals_ids = self.codec_tool.ids2npy(rearrange(codec_ids, "(n b) -> b n", b=2)[1])
|
||||
instrumentals.append(instrumentals_ids)
|
||||
vocals = np.concatenate(vocals, axis=1)
|
||||
instrumentals = np.concatenate(instrumentals, axis=1)
|
||||
stage1_output_dir = os.path.join(output_dir, "stage1")
|
||||
os.makedirs(stage1_output_dir, exist_ok=True)
|
||||
vocal_save_path = os.path.join(stage1_output_dir, "vtrack.npy")
|
||||
inst_save_path = os.path.join(stage1_output_dir, "itrack.npy")
|
||||
np.save(vocal_save_path, vocals)
|
||||
np.save(inst_save_path, instrumentals)
|
||||
|
||||
def shorten_input(self, seq: torch.Tensor, max_context: int):
|
||||
# Iteratively drop the oldest segment in the context until the sequence fits in context
|
||||
pattern = torch.tensor(self.start_of_segment)
|
||||
pattern_length = pattern.numel()
|
||||
while seq.shape[-1] > max_context:
|
||||
windows = seq[0].unfold(0, pattern_length, 1)
|
||||
matches = (windows == pattern).all(dim=1)
|
||||
match_indices = torch.nonzero(matches).flatten()
|
||||
if match_indices.numel() < 3:
|
||||
# Ensure that at least one other segment remains before the current segment for continuity
|
||||
print("Unable to keep enough segments for smart context, falling back to simple truncation. " f"Now using the last {max_context} tokens.")
|
||||
return seq[:, -max_context:]
|
||||
first_segment_start = match_indices[0].item()
|
||||
second_segment_start = match_indices[1].item()
|
||||
seq = torch.cat((seq[:, :first_segment_start], seq[:, second_segment_start:]), dim=-1)
|
||||
return seq
|
||||
|
||||
|
||||
class Stage1Pipeline_HF(Stage1Pipeline):
|
||||
|
||||
def __init__(self, model_path: str, device: torch.device, cache_size: int, **kwargs):
|
||||
super().__init__(device, **kwargs)
|
||||
|
||||
# Load HF model
|
||||
self.model = AutoModelForCausalLM.from_pretrained(model_path, torch_dtype=torch.float16, attn_implementation="sdpa", device_map=self.device)
|
||||
self.model.eval()
|
||||
if torch.__version__ >= "2.0.0":
|
||||
self.model = torch.compile(self.model)
|
||||
self.cache_size = cache_size
|
||||
|
||||
def generate(
|
||||
self,
|
||||
use_dual_tracks_prompt: bool,
|
||||
vocal_track_prompt_path: str,
|
||||
instrumental_track_prompt_path: str,
|
||||
use_audio_prompt: bool,
|
||||
audio_prompt_path: str,
|
||||
genres: str,
|
||||
lyrics: str,
|
||||
run_n_segments: int,
|
||||
max_new_tokens: int,
|
||||
prompt_start_time: int,
|
||||
prompt_end_time: int,
|
||||
sample_settings: SampleSettings,
|
||||
) -> torch.Tensor:
|
||||
|
||||
lyrics, prompt_texts = self.get_prompt_texts(genres, lyrics)
|
||||
run_n_segments = min(run_n_segments, len(lyrics))
|
||||
|
||||
for i in tqdm(range(run_n_segments)):
|
||||
|
||||
# Get prompt
|
||||
if i == 0:
|
||||
prompt_ids = self.get_first_segment_prompt(
|
||||
prompt_texts[1],
|
||||
prompt_texts[0],
|
||||
use_dual_tracks_prompt,
|
||||
vocal_track_prompt_path,
|
||||
instrumental_track_prompt_path,
|
||||
use_audio_prompt,
|
||||
audio_prompt_path,
|
||||
prompt_start_time,
|
||||
prompt_end_time,
|
||||
)
|
||||
else:
|
||||
prompt_ids = self.get_segment_prompt(prompt_texts[i + 1])
|
||||
prompt_ids = torch.as_tensor(prompt_ids).unsqueeze(0).to(self.device)
|
||||
input_ids = torch.cat([raw_output, prompt_ids], dim=1) if i > 0 else prompt_ids
|
||||
|
||||
# Use window slicing in case output sequence exceeds the context of model
|
||||
max_context = self.cache_size - max_new_tokens - 1
|
||||
if input_ids.shape[-1] > max_context:
|
||||
print(f"Section {i}: output length {input_ids.shape[-1]} exceeding context length {max_context}, " f"dropping early segment(s) from prompt.")
|
||||
input_ids = self.shorten_input(input_ids, max_context)
|
||||
|
||||
past_key_values = StaticCache(
|
||||
self.model.config, max_batch_size=1, max_cache_len=input_ids.shape[-1] + max_new_tokens, device=self.model.device, dtype=self.model.dtype
|
||||
)
|
||||
|
||||
processors = LogitsProcessorList([BlockTokenRangeProcessor(0, 32002), BlockTokenRangeProcessor(32016, 32016)])
|
||||
|
||||
output_seq = self.model.generate(
|
||||
input_ids=input_ids,
|
||||
max_new_tokens=max_new_tokens,
|
||||
min_new_tokens=100,
|
||||
do_sample=True,
|
||||
top_p=sample_settings.top_p,
|
||||
temperature=sample_settings.temperature,
|
||||
repetition_penalty=sample_settings.repetition_penalty,
|
||||
eos_token_id=self.mmtokenizer.eoa,
|
||||
pad_token_id=self.mmtokenizer.eoa,
|
||||
logits_processor=processors,
|
||||
guidance_scale=sample_settings.guidance_scale_seg0 if i == 0 else sample_settings.guidance_scale,
|
||||
past_key_values=past_key_values,
|
||||
)
|
||||
|
||||
if output_seq[0][-1].item() != self.mmtokenizer.eoa:
|
||||
tensor_eoa = torch.tensor([[self.mmtokenizer.eoa]], dtype=torch.long, device=output_seq.device)
|
||||
output_seq = torch.cat((output_seq, tensor_eoa), dim=1)
|
||||
if i > 0:
|
||||
raw_output = torch.cat([raw_output, prompt_ids, output_seq[:, input_ids.shape[-1] :]], dim=1)
|
||||
else:
|
||||
raw_output = output_seq
|
||||
return raw_output
|
||||
|
||||
|
||||
class Stage1Pipeline_EXL2(Stage1Pipeline):
|
||||
|
||||
def __init__(self, model_path: str, device: torch.device, cache_size: int, cache_mode: str, **kwargs):
|
||||
super().__init__(device, **kwargs)
|
||||
|
||||
assert device != "cpu", "ExLlamaV2 does not support CPU inference."
|
||||
|
||||
# Load EXL2 model
|
||||
device_idx = self.device.index
|
||||
gpu_split = [0] * torch.cuda.device_count()
|
||||
gpu_split[device_idx] = 9999
|
||||
exl2_config = ExLlamaV2Config(model_path)
|
||||
exl2_config.no_sdpa = True # TODO: Figure out why SDPA slows to a crawl when given custom attn mask
|
||||
self.model = ExLlamaV2(exl2_config)
|
||||
self.model.load(gpu_split)
|
||||
|
||||
# Load tokenizer (only needed for vocab size in disallow_tokens)
|
||||
self.tokenizer = ExLlamaV2Tokenizer(exl2_config)
|
||||
|
||||
# Define cache
|
||||
self.cache_size = cache_size
|
||||
self.cache_mode = get_cache_class(cache_mode)
|
||||
|
||||
# TODO: Output layer could be trimmed here to avoid masking out the first 32k tokens during generation
|
||||
|
||||
def generate(
|
||||
self,
|
||||
use_dual_tracks_prompt: bool,
|
||||
vocal_track_prompt_path: str,
|
||||
instrumental_track_prompt_path: str,
|
||||
use_audio_prompt: bool,
|
||||
audio_prompt_path: str,
|
||||
genres: str,
|
||||
lyrics: str,
|
||||
run_n_segments: int,
|
||||
max_new_tokens: int,
|
||||
prompt_start_time: int,
|
||||
prompt_end_time: int,
|
||||
sample_settings: SampleSettings,
|
||||
) -> torch.Tensor:
|
||||
|
||||
if sample_settings.guidance_scale_seg0 is None:
|
||||
bsz = 1
|
||||
cfg = False
|
||||
position_offsets = None
|
||||
input_mask = None
|
||||
else:
|
||||
bsz = 2
|
||||
cfg = True
|
||||
|
||||
lyrics, prompt_texts = self.get_prompt_texts(genres, lyrics)
|
||||
run_n_segments = min(run_n_segments, len(lyrics))
|
||||
|
||||
# Cache for the whole output sequence
|
||||
cache = self.cache_mode(self.model, batch_size=bsz, max_seq_len=self.cache_size)
|
||||
|
||||
# Collect output here
|
||||
seq = torch.empty((bsz, 0), dtype=torch.long)
|
||||
|
||||
# Sample settings
|
||||
gen_settings = ExLlamaV2Sampler.Settings(
|
||||
top_k=0, top_p=sample_settings.top_p, token_repetition_penalty=sample_settings.repetition_penalty, temperature=sample_settings.temperature
|
||||
)
|
||||
gen_settings.allow_tokens(self.tokenizer, [32002] + list(range(45334, 56722)))
|
||||
|
||||
# RNG for sampling, could seed here
|
||||
rng = random.Random()
|
||||
|
||||
for i in tqdm(range(run_n_segments)):
|
||||
|
||||
# Get prompt for this segment
|
||||
if i == 0:
|
||||
prompt_ids = self.get_first_segment_prompt(
|
||||
prompt_texts[1],
|
||||
prompt_texts[0],
|
||||
use_dual_tracks_prompt,
|
||||
vocal_track_prompt_path,
|
||||
instrumental_track_prompt_path,
|
||||
use_audio_prompt,
|
||||
audio_prompt_path,
|
||||
prompt_start_time,
|
||||
prompt_end_time,
|
||||
)
|
||||
else:
|
||||
prompt_ids = self.get_segment_prompt(prompt_texts[i + 1])
|
||||
prompt_ids = torch.tensor([prompt_ids] * bsz, dtype=torch.long)
|
||||
|
||||
# Accept prompt tokens
|
||||
seq = torch.cat((seq, prompt_ids), dim=-1)
|
||||
|
||||
# Use window slicing in case output sequence exceeds the context of model
|
||||
max_context = self.cache_size - max_new_tokens - 1
|
||||
if seq.shape[-1] > max_context:
|
||||
print(f"Section {i}: output length {seq.shape[-1]} exceeding context length {max_context}, " f"dropping early segment(s) from prompt.")
|
||||
cache.current_seq_len = 0
|
||||
full_ids = self.shorten_input(seq, max_context)
|
||||
incremental_ids = full_ids
|
||||
else:
|
||||
full_ids = seq
|
||||
incremental_ids = prompt_ids
|
||||
|
||||
# For the unconditional context, mask out all but the last token
|
||||
if cfg:
|
||||
mask_len = full_ids.shape[-1] - 1
|
||||
full_mask = torch.zeros((2, cache.max_seq_len), dtype=torch.half, device=self.device)
|
||||
full_mask[1, :mask_len] = -65504.0
|
||||
position_offsets = torch.tensor([[0], [-mask_len]], dtype=torch.int)
|
||||
input_mask = full_mask[:, : full_ids.shape[-1]]
|
||||
|
||||
# Forward prompt
|
||||
logits = self.model.forward(incremental_ids[:, :], cache=cache, input_mask=input_mask, position_offsets=position_offsets, last_id_only=True)
|
||||
|
||||
# Generate until EOS or max_new_tokens
|
||||
for new_tokens in tqdm(range(max_new_tokens)):
|
||||
|
||||
# Transformers-equiv. CFG
|
||||
if cfg:
|
||||
cfg_scale = sample_settings.guidance_scale_seg0 if i == 0 else sample_settings.guidance_scale
|
||||
logits = logits.float()
|
||||
logits = F.log_softmax(logits, dim=-1)
|
||||
logits = cfg_scale * logits[0] + (1 - cfg_scale) * logits[1]
|
||||
logits = logits.unsqueeze(0)
|
||||
|
||||
# Sample
|
||||
logits = logits.float().cpu()
|
||||
sample, _, _, _, _ = ExLlamaV2Sampler.sample(logits, gen_settings, full_ids[:1], rng.random(), self.tokenizer)
|
||||
if cfg:
|
||||
sample = torch.cat((sample, sample), dim=0)
|
||||
|
||||
# Accept token
|
||||
full_ids = torch.cat((full_ids, sample), dim=-1)
|
||||
seq = torch.cat((seq, sample), dim=-1)
|
||||
|
||||
# Get next logits (update cache even if sample is EOA and we don't need next logits)
|
||||
if cfg:
|
||||
input_mask = full_mask[:, : full_ids.shape[-1]]
|
||||
logits = self.model.forward(sample, cache=cache, input_mask=input_mask, position_offsets=position_offsets)
|
||||
|
||||
# End on EOA
|
||||
if sample[0].item() == self.mmtokenizer.eoa:
|
||||
break
|
||||
|
||||
# Make sure sequence ends with EOA if we reached max_new_tokens
|
||||
else:
|
||||
sample = torch.tensor([[self.mmtokenizer.eoa]] * bsz, dtype=torch.long)
|
||||
seq = torch.cat((seq, sample), dim=-1)
|
||||
# Update cache with forced token
|
||||
self.model.forward(sample, cache=cache)
|
||||
|
||||
raw_output = seq[:1, :]
|
||||
return raw_output
|
||||
|
||||
|
||||
def main():
|
||||
args = parser.parse_args()
|
||||
if args.use_audio_prompt and not args.audio_prompt_path:
|
||||
raise FileNotFoundError("Please offer audio prompt filepath using '--audio_prompt_path', when you enable 'use_audio_prompt'!")
|
||||
if args.use_dual_tracks_prompt and not args.vocal_track_prompt_path and not args.instrumental_track_prompt_path:
|
||||
raise FileNotFoundError(
|
||||
"Please offer dual tracks prompt filepath using '--vocal_track_prompt_path' and '--inst_decoder_path', when you enable '--use_dual_tracks_prompt'!"
|
||||
)
|
||||
if args.seed is not None:
|
||||
seed_everything(args.seed)
|
||||
|
||||
device = torch.device(f"cuda:{args.cuda_idx}" if torch.cuda.is_available() else "cpu")
|
||||
|
||||
with open(args.genre_txt) as f:
|
||||
genres = f.read().strip()
|
||||
with open(args.lyrics_txt) as f:
|
||||
lyrics = f.read().strip()
|
||||
|
||||
if args.stage1_use_exl2:
|
||||
pipeline = Stage1Pipeline_EXL2(
|
||||
model_path=args.stage1_model,
|
||||
device=device,
|
||||
basic_model_config=args.basic_model_config,
|
||||
resume_path=args.resume_path,
|
||||
cache_size=args.stage1_cache_size,
|
||||
cache_mode=args.stage1_cache_mode,
|
||||
)
|
||||
else:
|
||||
pipeline = Stage1Pipeline_HF(
|
||||
model_path=args.stage1_model,
|
||||
device=device,
|
||||
basic_model_config=args.basic_model_config,
|
||||
resume_path=args.resume_path,
|
||||
cache_size=args.stage1_cache_size,
|
||||
)
|
||||
|
||||
# Load tokenizer and models
|
||||
raw_output = pipeline.generate(
|
||||
use_dual_tracks_prompt=args.use_dual_tracks_prompt,
|
||||
vocal_track_prompt_path=args.vocal_track_prompt_path,
|
||||
instrumental_track_prompt_path=args.instrumental_track_prompt_path,
|
||||
use_audio_prompt=args.use_audio_prompt,
|
||||
audio_prompt_path=args.audio_prompt_path,
|
||||
genres=genres,
|
||||
lyrics=lyrics,
|
||||
run_n_segments=args.run_n_segments,
|
||||
max_new_tokens=args.max_new_tokens,
|
||||
prompt_start_time=args.prompt_start_time,
|
||||
prompt_end_time=args.prompt_end_time,
|
||||
sample_settings=SampleSettings(use_guidance=not args.stage1_no_guidance, repetition_penalty=args.repetition_penalty),
|
||||
)
|
||||
|
||||
# Save result
|
||||
pipeline.save(raw_output, args.output_dir, args.use_audio_prompt, args.use_dual_tracks_prompt)
|
||||
|
||||
|
||||
# if __name__ == "__main__":
|
||||
# # enable inference mode globally
|
||||
# torch.autograd.grad_mode._enter_inference_mode(True)
|
||||
# torch.autograd.set_grad_enabled(False)
|
||||
# main()
|
||||
|
||||
#Full lyrics: ["Generate music from the given lyrics segment by segment.\n[Genre] inspiring female uplifting pop airy vocal electronic bright vocal vocal.\n[verse]\nStaring at the sunset, colors paint the sky.\nThoughts of you keep swirling, can't deny.\nI know I let you down, I made mistakes.\nBut I'm here to mend the heart I didn't break.\n\n\n[chorus]\nEvery road you take, I'll be one step behind.\nEvery dream you chase, I'm reaching for the light.\nYou can't fight this feeling now.\nI won't back down.\nYou know you can't deny it now.\n I won't back down\n\n\n[verse]\nThey might say I'm foolish, chasing after you.\nBut they don't feel this love the way we do.\nMy heart beats only for you, can't you see?\nI won't let you slip away from me.\n\n\n[chorus]\nEvery road you take, I'll be one step behind.\nEvery dream you chase, I'm reaching for the light.\nYou can't fight this feeling now.\nI won't back down.\nYou know you can't deny it now.\n I won't back down\n\n\n[bridge]\nNo, I won't back down, won't turn around.\nUntil you're back where you belong.\nI'll cross the oceans wide, stand by your side.\nTogether we are strong.\n\n\n[outro]\nEvery road you take, I'll be one step behind.\nEvery dream you chase, love's the tie that binds.\nYou can't fight this feeling now.\nI won't back down.\n\n"]
|
||||
@@ -1,365 +0,0 @@
|
||||
import copy
|
||||
import math
|
||||
import os
|
||||
from collections import Counter
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
from .codecmanipulator import CodecManipulator
|
||||
from .common import BlockTokenRangeProcessor, seed_everything, get_cache_class
|
||||
from exllamav2 import ExLlamaV2, ExLlamaV2Config, ExLlamaV2Tokenizer
|
||||
from .mmtokenizer import _MMSentencePieceTokenizer
|
||||
from tqdm import tqdm
|
||||
from transformers import AutoModelForCausalLM, LogitsProcessorList
|
||||
from transformers.cache_utils import StaticCache
|
||||
import gc
|
||||
import folder_paths
|
||||
|
||||
def align(n, m):
|
||||
return ((n + m - 1) // m) * m
|
||||
|
||||
|
||||
def split_bsz(bsz, maxbsz):
|
||||
n_sub_batches = math.ceil(bsz / maxbsz)
|
||||
base_size = bsz // n_sub_batches
|
||||
remainder = bsz % n_sub_batches
|
||||
sub_batch_sizes = [base_size + 1] * remainder + [base_size] * (n_sub_batches - remainder)
|
||||
indices = []
|
||||
start = 0
|
||||
for size in sub_batch_sizes:
|
||||
end = start + size
|
||||
indices.append((start, end))
|
||||
start = end
|
||||
return indices
|
||||
|
||||
|
||||
class Stage2Pipeline:
|
||||
|
||||
def __init__(self, device: torch.device):
|
||||
self.device = device
|
||||
|
||||
self.codec_tool = CodecManipulator("xcodec", 0, 1)
|
||||
self.codec_tool_stage2 = CodecManipulator("xcodec", 0, 8)
|
||||
|
||||
# Load tokenizer
|
||||
self.mmtokenizer = _MMSentencePieceTokenizer(os.path.join(folder_paths.base_path,"custom_nodes/ComfyUI_YuE/inference/mm_tokenizer_v0.2_hf/tokenizer.model"))
|
||||
|
||||
def get_codec_ids(self, prompt: np.array):
|
||||
codec_ids = self.codec_tool.unflatten(prompt, n_quantizer=1)
|
||||
codec_ids = self.codec_tool.offset_tok_ids(
|
||||
codec_ids, global_offset=self.codec_tool.global_offset, codebook_size=self.codec_tool.codebook_size, num_codebooks=self.codec_tool.num_codebooks
|
||||
).astype(np.int32)
|
||||
return codec_ids
|
||||
|
||||
def fix_output(self, output):
|
||||
# Fix invalid codes (a dirty solution, which may harm the quality of audio)
|
||||
# We are trying to find better one
|
||||
fixed_output = copy.deepcopy(output)
|
||||
for i, line in enumerate(output):
|
||||
for j, element in enumerate(line):
|
||||
if element < 0 or element > 1023:
|
||||
counter = Counter(line)
|
||||
most_frequant = sorted(counter.items(), key=lambda x: x[1], reverse=True)[0][0]
|
||||
fixed_output[i, j] = most_frequant
|
||||
return fixed_output
|
||||
|
||||
def save(self, output_dir: str, outputs):
|
||||
for output_name, output in outputs.items():
|
||||
# save output
|
||||
stage2_output_dir = os.path.join(output_dir, "stage2")
|
||||
os.makedirs(stage2_output_dir, exist_ok=True)
|
||||
output_filename = os.path.join(stage2_output_dir, output_name)
|
||||
np.save(output_filename, output)
|
||||
|
||||
def get_stage1_prompt(self, output_dir: str, output_name: str):
|
||||
stage1_output_dir = os.path.join(output_dir, "stage1")
|
||||
prompt = np.load(os.path.join(stage1_output_dir, output_name)).astype(np.int32)
|
||||
return prompt
|
||||
|
||||
def prepare_prompt_batch(self, prompt: np.array, batch_size: int):
|
||||
|
||||
codec_ids = self.get_codec_ids(prompt)
|
||||
|
||||
# Prepare prompt_ids based on batch size or single input
|
||||
if batch_size > 1:
|
||||
codec_list = []
|
||||
for i in range(batch_size):
|
||||
idx_begin = i * 300
|
||||
idx_end = (i + 1) * 300
|
||||
codec_list.append(codec_ids[:, idx_begin:idx_end])
|
||||
|
||||
codec_ids = np.concatenate(codec_list, axis=0)
|
||||
prompt_ids = np.concatenate(
|
||||
[np.tile([self.mmtokenizer.soa, self.mmtokenizer.stage_1], (batch_size, 1)), codec_ids, np.tile([self.mmtokenizer.stage_2], (batch_size, 1))],
|
||||
axis=1,
|
||||
)
|
||||
else:
|
||||
prompt_ids = np.concatenate(
|
||||
[
|
||||
np.array([self.mmtokenizer.soa, self.mmtokenizer.stage_1]),
|
||||
codec_ids.flatten(), # Flatten the 2D array to 1D
|
||||
np.array([self.mmtokenizer.stage_2]),
|
||||
]
|
||||
).astype(np.int32)
|
||||
prompt_ids = prompt_ids[np.newaxis, ...]
|
||||
|
||||
codec_ids = torch.as_tensor(codec_ids, dtype=torch.long)
|
||||
prompt_ids = torch.as_tensor(prompt_ids, dtype=torch.long)
|
||||
return codec_ids, prompt_ids
|
||||
|
||||
|
||||
class Stage2Pipeline_HF(Stage2Pipeline):
|
||||
|
||||
def __init__(self, model_path: str, device: torch.device, batch_size: int):
|
||||
super().__init__(device)
|
||||
self.batch_size = batch_size
|
||||
|
||||
self.model = AutoModelForCausalLM.from_pretrained(model_path, torch_dtype=torch.float16, attn_implementation="sdpa")
|
||||
self.model.to(device)
|
||||
self.model.eval()
|
||||
if torch.__version__ >= "2.0.0":
|
||||
self.model = torch.compile(self.model)
|
||||
|
||||
def generate_batch(self, prompt: np.array, batch_size: int):
|
||||
codec_ids, prompt_ids = self.prepare_prompt_batch(prompt, batch_size)
|
||||
len_prompt = prompt_ids.shape[-1]
|
||||
|
||||
# Teacher forcing generate loop
|
||||
codec_ids = codec_ids.to(self.device)
|
||||
prompt_ids = prompt_ids.to(self.device)
|
||||
block_list = LogitsProcessorList([BlockTokenRangeProcessor(0, 46358), BlockTokenRangeProcessor(53526, self.mmtokenizer.vocab_size)])
|
||||
past_key_values = StaticCache(
|
||||
self.model.config,
|
||||
max_batch_size=batch_size,
|
||||
max_cache_len=prompt_ids.shape[1] + codec_ids.shape[1] * 8,
|
||||
device=self.model.device,
|
||||
dtype=self.model.dtype,
|
||||
)
|
||||
for frames_idx in range(codec_ids.shape[1]):
|
||||
cb0 = codec_ids[:, frames_idx : frames_idx + 1]
|
||||
prompt_ids = torch.cat([prompt_ids, cb0], dim=1)
|
||||
input_ids = prompt_ids
|
||||
|
||||
stage2_output = self.model.generate(
|
||||
input_ids=input_ids,
|
||||
min_new_tokens=7,
|
||||
max_new_tokens=7,
|
||||
eos_token_id=self.mmtokenizer.eoa,
|
||||
pad_token_id=self.mmtokenizer.eoa,
|
||||
logits_processor=block_list,
|
||||
past_key_values=past_key_values,
|
||||
)
|
||||
|
||||
assert stage2_output.shape[1] - prompt_ids.shape[1] == 7, f"output new tokens={stage2_output.shape[1]-prompt_ids.shape[1]}"
|
||||
prompt_ids = stage2_output
|
||||
|
||||
# Return output based on batch size
|
||||
if batch_size > 1:
|
||||
output = prompt_ids.cpu().numpy()[:, len_prompt:]
|
||||
output_list = [output[i] for i in range(batch_size)]
|
||||
output = np.concatenate(output_list, axis=0)
|
||||
else:
|
||||
output = prompt_ids[0].cpu().numpy()[len_prompt:]
|
||||
|
||||
return output
|
||||
|
||||
def generate(self, output_dir: str) -> dict[str, np.array]:
|
||||
outputs = {}
|
||||
for output_name in tqdm(["vtrack.npy", "itrack.npy"]):
|
||||
# Load the prompt
|
||||
prompt = self.get_stage1_prompt(output_dir, output_name)
|
||||
|
||||
# Only accept 6s segments
|
||||
output_duration = prompt.shape[-1] // 50 // 6 * 6
|
||||
num_batch = output_duration // 6
|
||||
|
||||
if num_batch <= self.batch_size:
|
||||
# If num_batch is less than or equal to batch_size, we can infer the entire prompt at once
|
||||
output = self.generate_batch(prompt[:, : output_duration * 50], batch_size=num_batch)
|
||||
else:
|
||||
# If num_batch is greater than batch_size, process in chunks of batch_size
|
||||
segments = []
|
||||
num_segments = (num_batch // self.batch_size) + (1 if num_batch % self.batch_size != 0 else 0)
|
||||
|
||||
for seg in range(num_segments):
|
||||
start_idx = seg * self.batch_size * 300
|
||||
# Ensure the end_idx does not exceed the available length
|
||||
end_idx = min((seg + 1) * self.batch_size * 300, output_duration * 50) # Adjust the last segment
|
||||
current_batch_size = self.batch_size if seg != num_segments - 1 or num_batch % self.batch_size == 0 else num_batch % self.batch_size
|
||||
segment = self.generate_batch(prompt[:, start_idx:end_idx], batch_size=current_batch_size)
|
||||
segments.append(segment)
|
||||
|
||||
# Concatenate all the segments
|
||||
output = np.concatenate(segments, axis=0)
|
||||
|
||||
# Process the ending part of the prompt
|
||||
if output_duration * 50 != prompt.shape[-1]:
|
||||
ending = self.generate_batch(prompt[:, output_duration * 50 :], batch_size=1)
|
||||
output = np.concatenate([output, ending], axis=0)
|
||||
|
||||
output = self.codec_tool_stage2.ids2npy(output)
|
||||
|
||||
output = self.fix_output(output)
|
||||
outputs[output_name] = output
|
||||
return outputs
|
||||
|
||||
|
||||
class Stage2Pipeline_EXL2(Stage2Pipeline):
|
||||
|
||||
def __init__(self, model_path: str, device: torch.device, cache_size: int, cache_mode: str):
|
||||
super().__init__(device)
|
||||
|
||||
self.cache_size = cache_size
|
||||
|
||||
assert device != "cpu", "ExLlamaV2 does not support CPU inference."
|
||||
|
||||
# Load EXL2 model
|
||||
device_idx = self.device.index
|
||||
gpu_split = [0] * torch.cuda.device_count()
|
||||
gpu_split[device_idx] = 9999
|
||||
exl2_config = ExLlamaV2Config(model_path)
|
||||
self.model = ExLlamaV2(exl2_config)
|
||||
self.model.load(gpu_split)
|
||||
|
||||
# Move embedding layer to GPU to avoid CPU sync during argmax gen loop
|
||||
self.model.modules[0].device_idx = self.model.modules[1].device_idx
|
||||
self.model.modules[0].reload()
|
||||
|
||||
# Load tokenizer (only needed for vocab size in disallow_tokens)
|
||||
self.tokenizer = ExLlamaV2Tokenizer(exl2_config)
|
||||
|
||||
# Define cache
|
||||
self.cache_mode = get_cache_class(cache_mode)
|
||||
|
||||
def generate(self, output_dir: str) -> dict[str, np.array]:
|
||||
|
||||
parts = ["vtrack.npy", "itrack.npy"]
|
||||
full_batch = []
|
||||
|
||||
# Collect up to 300 token (6s) segments for all parts
|
||||
for output_idx, output_name in tqdm(enumerate(parts)):
|
||||
prompt = self.get_stage1_prompt(output_dir, output_name)
|
||||
prompt = self.get_codec_ids(prompt)
|
||||
prompt = torch.as_tensor(prompt, dtype=torch.long)
|
||||
|
||||
segs = torch.split(prompt, 300, dim=-1)
|
||||
|
||||
for seg_idx, seg in enumerate(segs):
|
||||
seg_len = seg.shape[-1]
|
||||
full_batch.append((seg_len, seg_idx, output_idx, seg))
|
||||
|
||||
# Prepare segments
|
||||
prefix = torch.tensor([[self.mmtokenizer.soa, self.mmtokenizer.stage_1]], dtype=torch.long)
|
||||
suffix = torch.tensor([[self.mmtokenizer.stage_2]], dtype=torch.long)
|
||||
for i in range(len(full_batch)):
|
||||
seg_len, seg_idx, output_idx, codec_ids = full_batch[i]
|
||||
prompt_ids = torch.cat((prefix, codec_ids, suffix), dim=-1)
|
||||
full_batch[i] = (seg_len, seg_idx, output_idx, codec_ids, prompt_ids)
|
||||
|
||||
# Group prompts by length
|
||||
batches = {}
|
||||
for seq in full_batch:
|
||||
if not seq[0] in batches:
|
||||
batches[seq[0]] = []
|
||||
batches[seq[0]].append(seq)
|
||||
|
||||
# Split into on minibatches
|
||||
split_batch = []
|
||||
for idx, (seg_len, batch) in enumerate(batches.items()):
|
||||
b_seg_order = [b[1] for b in batch]
|
||||
b_part_order = [b[2] for b in batch]
|
||||
b_codec_ids = torch.cat([b[3] for b in batch], dim=0)
|
||||
b_prompt_ids = torch.cat([b[4] for b in batch], dim=0)
|
||||
|
||||
max_bsz = self.cache_size // align(b_prompt_ids.shape[1] + b_codec_ids.shape[1] * 8, 32)
|
||||
assert max_bsz > 0
|
||||
for a, b in split_bsz(b_prompt_ids.shape[0], max_bsz):
|
||||
split_batch.append((b_seg_order[a:b], b_part_order[a:b], b_codec_ids[a:b], b_prompt_ids[a:b]))
|
||||
|
||||
# Inference
|
||||
output_parts = []
|
||||
for _ in parts:
|
||||
output_parts.append([])
|
||||
|
||||
for seg_order, part_order, codec_ids, prompt_ids in tqdm(split_batch):
|
||||
codec_ids = codec_ids.to(self.device)
|
||||
prompt_ids = prompt_ids.to(self.device)
|
||||
batch_size, len_prompt = prompt_ids.shape
|
||||
|
||||
cache = self.cache_mode(self.model, batch_size=batch_size, max_seq_len=align(prompt_ids.shape[1] + codec_ids.shape[1] * 8, 32))
|
||||
output_ids = torch.empty((batch_size, 0), dtype=torch.long, device=self.device)
|
||||
|
||||
for frames_idx in tqdm(range(codec_ids.shape[1])):
|
||||
cb0 = codec_ids[:, frames_idx : frames_idx + 1]
|
||||
|
||||
# Append the initial prompt to the first codec frame
|
||||
if frames_idx == 0:
|
||||
cb0 = torch.cat([prompt_ids, cb0], dim=-1)
|
||||
|
||||
# Forward prompt
|
||||
output_ids = torch.cat((output_ids, cb0), dim=-1)
|
||||
logits = self.model.forward(cb0, cache=cache, last_id_only=True)
|
||||
|
||||
for i in range(7):
|
||||
|
||||
# Slice logits instead of biasing start and end of distribution
|
||||
first_logit = 46358
|
||||
last_logit = 53526
|
||||
logits = logits[:, :, first_logit:last_logit]
|
||||
|
||||
# Greedy sampling
|
||||
sample = logits.argmax(dim=-1) + first_logit
|
||||
output_ids = torch.cat((output_ids, sample), dim=-1)
|
||||
|
||||
# TODO: Here, original asserts that we didn't sample mmtokenizer.eoa (can we just mask it out?)
|
||||
|
||||
# Forward sample
|
||||
logits = self.model.forward(sample, cache=cache)
|
||||
|
||||
# Trim prompt
|
||||
output_ids = output_ids[:, len_prompt:]
|
||||
|
||||
# Split outputs
|
||||
for i in range(batch_size):
|
||||
output_parts[part_order[i]].append((seg_order[i], output_ids[i : i + 1, :]))
|
||||
|
||||
# Release cache tensors
|
||||
del cache
|
||||
torch.cuda.empty_cache()
|
||||
gc.collect()
|
||||
|
||||
# Unshuffle and recombine output parts
|
||||
output = {}
|
||||
for i, p in enumerate(output_parts):
|
||||
p = sorted(p, key=lambda x: x[0])
|
||||
part_o = torch.cat([pp[1] for pp in p], dim=-1).flatten().cpu().numpy()
|
||||
part_o = self.codec_tool_stage2.ids2npy(part_o)
|
||||
part_o = self.fix_output(part_o)
|
||||
output[parts[i]] = part_o
|
||||
|
||||
return output
|
||||
|
||||
|
||||
def main():
|
||||
args = parser.parse_args()
|
||||
if args.seed is not None:
|
||||
seed_everything(args.seed)
|
||||
|
||||
device = torch.device(f"cuda:{args.cuda_idx}" if torch.cuda.is_available() else "cpu")
|
||||
|
||||
if args.stage2_use_exl2:
|
||||
pipeline = Stage2Pipeline_EXL2(model_path=args.stage2_model, device=device, cache_size=args.stage2_cache_size, cache_mode=args.stage2_cache_mode)
|
||||
pass
|
||||
else:
|
||||
pipeline = Stage2Pipeline_HF(model_path=args.stage2_model, device=device, batch_size=args.stage2_batch_size)
|
||||
|
||||
outputs = pipeline.generate(output_dir=args.output_dir)
|
||||
|
||||
pipeline.save(output_dir=args.output_dir, outputs=outputs)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# enable inference mode globally
|
||||
torch.autograd.grad_mode._enter_inference_mode(True)
|
||||
torch.autograd.set_grad_enabled(False)
|
||||
main()
|
||||
@@ -1,367 +0,0 @@
|
||||
from abc import ABC
|
||||
from abc import abstractmethod
|
||||
|
||||
|
||||
class AbstractTokenizer(ABC):
|
||||
"""Abstract class for tokenizer."""
|
||||
|
||||
def __init__(self, name):
|
||||
self.name = name
|
||||
super().__init__()
|
||||
|
||||
@property
|
||||
@abstractmethod
|
||||
def vocab_size(self):
|
||||
pass
|
||||
|
||||
@property
|
||||
@abstractmethod
|
||||
def vocab(self):
|
||||
"""Dictionary from vocab text token to id token."""
|
||||
pass
|
||||
|
||||
@property
|
||||
@abstractmethod
|
||||
def inv_vocab(self):
|
||||
"""Dictionary from vocab id token to text token."""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def tokenize(self, text):
|
||||
pass
|
||||
|
||||
def detokenize(self, token_ids):
|
||||
raise NotImplementedError('detokenizer is not implemented for {} '
|
||||
'tokenizer'.format(self.name))
|
||||
|
||||
@property
|
||||
def cls(self):
|
||||
raise NotImplementedError('CLS is not provided for {} '
|
||||
'tokenizer'.format(self.name))
|
||||
|
||||
@property
|
||||
def sep(self):
|
||||
raise NotImplementedError('SEP is not provided for {} '
|
||||
'tokenizer'.format(self.name))
|
||||
|
||||
@property
|
||||
def pad(self):
|
||||
raise NotImplementedError('PAD is not provided for {} '
|
||||
'tokenizer'.format(self.name))
|
||||
|
||||
@property
|
||||
def eod(self):
|
||||
raise NotImplementedError('EOD is not provided for {} '
|
||||
'tokenizer'.format(self.name))
|
||||
|
||||
@property
|
||||
def mask(self):
|
||||
raise NotImplementedError('MASK is not provided for {} '
|
||||
'tokenizer'.format(self.name))
|
||||
|
||||
|
||||
class _SentencePieceTokenizer(AbstractTokenizer):
|
||||
"""SentencePieceTokenizer-Megatron wrapper"""
|
||||
|
||||
def __init__(self, model_file, vocab_extra_ids=0):
|
||||
name = 'SentencePieceTokenizer'
|
||||
super().__init__(name)
|
||||
|
||||
import sentencepiece
|
||||
self.tokenizer = sentencepiece.SentencePieceProcessor(model_file=model_file)
|
||||
self._initalize(vocab_extra_ids)
|
||||
|
||||
def _populate_vocab(self):
|
||||
self._vocab = {}
|
||||
self._inv_vocab = {}
|
||||
|
||||
for i in range(len(self.tokenizer)):
|
||||
t = self.tokenizer.id_to_piece(i)
|
||||
self._inv_vocab[i] = t
|
||||
self._vocab[t] = i
|
||||
|
||||
def _initalize(self, vocab_extra_ids):
|
||||
self._populate_vocab()
|
||||
self._special_tokens = {}
|
||||
self._inv_special_tokens = {}
|
||||
|
||||
self._t5_tokens = []
|
||||
|
||||
def _add_special_token(t):
|
||||
if t not in self._vocab:
|
||||
next_id = len(self._vocab)
|
||||
self._vocab[t] = next_id
|
||||
self._inv_vocab[next_id] = t
|
||||
self._special_tokens[t] = self._vocab[t]
|
||||
self._inv_special_tokens[self._vocab[t]] = t
|
||||
|
||||
_add_special_token('<CLS>')
|
||||
self._cls_id = self._vocab['<CLS>']
|
||||
_add_special_token('<SEP>')
|
||||
self._sep_id = self._vocab['<SEP>']
|
||||
_add_special_token('<EOD>')
|
||||
self._eod_id = self._vocab['<EOD>']
|
||||
_add_special_token('<MASK>')
|
||||
self._mask_id = self._vocab['<MASK>']
|
||||
|
||||
pad_id = self.tokenizer.pad_id()
|
||||
try:
|
||||
pad_token = self.tokenizer.id_to_piece(pad_id)
|
||||
except IndexError:
|
||||
pad_token = '<PAD>'
|
||||
_add_special_token(pad_token)
|
||||
self._pad_id = self._vocab[pad_token]
|
||||
|
||||
bos_id = self.tokenizer.bos_id()
|
||||
try:
|
||||
bos_token = self.tokenizer.id_to_piece(bos_id)
|
||||
except IndexError:
|
||||
bos_token = '<BOS>'
|
||||
_add_special_token(bos_token)
|
||||
self._bos_id = self._vocab[bos_token]
|
||||
|
||||
eos_id = self.tokenizer.eos_id()
|
||||
try:
|
||||
eos_token = self.tokenizer.id_to_piece(eos_id)
|
||||
except IndexError:
|
||||
eos_token = '<EOS>'
|
||||
_add_special_token(eos_token)
|
||||
self._eos_id = self._vocab[eos_token]
|
||||
|
||||
for i in range(vocab_extra_ids):
|
||||
t = "<extra_id_{}>".format(i)
|
||||
_add_special_token(t)
|
||||
self._t5_tokens += [t]
|
||||
|
||||
@property
|
||||
def vocab_size(self):
|
||||
return len(self._vocab)
|
||||
|
||||
@property
|
||||
def vocab(self):
|
||||
return self._vocab
|
||||
|
||||
@property
|
||||
def inv_vocab(self):
|
||||
return self._inv_vocab
|
||||
|
||||
@property
|
||||
def decoder(self):
|
||||
return self._inv_vocab
|
||||
|
||||
@property
|
||||
def encoder(self):
|
||||
return self._vocab
|
||||
|
||||
# From:
|
||||
# https://github.com/NVIDIA/NeMo/blob/c8fa217e811d60d11d014827c7f3845ff6c99ae7/nemo/collections/common/tokenizers/sentencepiece_tokenizer.py#L89
|
||||
def tokenize(self, text):
|
||||
ids = []
|
||||
idx = 0
|
||||
|
||||
while 1:
|
||||
indices = {}
|
||||
for token in self._special_tokens:
|
||||
try:
|
||||
indices[token] = text[idx:].index(token)
|
||||
except ValueError:
|
||||
continue
|
||||
if len(indices) == 0:
|
||||
break
|
||||
|
||||
next_token = min(indices, key=indices.get)
|
||||
next_idx = idx + indices[next_token]
|
||||
|
||||
ids.extend(self.tokenizer.encode_as_ids(text[idx:next_idx]))
|
||||
ids.append(self._special_tokens[next_token])
|
||||
idx = next_idx + len(next_token)
|
||||
|
||||
ids.extend(self.tokenizer.encode_as_ids(text[idx:]))
|
||||
return ids
|
||||
|
||||
# From:
|
||||
# https://github.com/NVIDIA/NeMo/blob/c8fa217e811d60d11d014827c7f3845ff6c99ae7/nemo/collections/common/tokenizers/sentencepiece_tokenizer.py#L125
|
||||
def detokenize(self, ids):
|
||||
text = ""
|
||||
last_i = 0
|
||||
|
||||
for i, id in enumerate(ids):
|
||||
if id in self._inv_special_tokens:
|
||||
text += self.tokenizer.decode_ids(ids[last_i:i]) + " "
|
||||
text += self._inv_special_tokens[id] + " "
|
||||
last_i = i + 1
|
||||
|
||||
text += self.tokenizer.decode_ids(ids[last_i:])
|
||||
return text
|
||||
|
||||
@property
|
||||
def cls(self):
|
||||
return self._cls_id
|
||||
|
||||
@property
|
||||
def sep(self):
|
||||
return self._sep_id
|
||||
|
||||
@property
|
||||
def pad(self):
|
||||
return self._pad_id
|
||||
|
||||
@property
|
||||
def bos_token_id(self):
|
||||
return self._bos_id
|
||||
|
||||
@property
|
||||
def bos(self):
|
||||
return self._bos_id
|
||||
|
||||
@property
|
||||
def eod(self):
|
||||
return self._eod_id
|
||||
|
||||
@property
|
||||
def eos_token_id(self):
|
||||
return self._eos_id
|
||||
|
||||
@property
|
||||
def eos(self):
|
||||
return self._eos_id
|
||||
|
||||
@property
|
||||
def mask(self):
|
||||
return self._mask_id
|
||||
|
||||
@property
|
||||
def additional_special_tokens_ids(self):
|
||||
return [self.vocab[k] for k in self._t5_tokens]
|
||||
|
||||
class _MMSentencePieceTokenizer(_SentencePieceTokenizer):
|
||||
"""SentencePieceTokenizer-Megatron wrapper"""
|
||||
|
||||
def __init__(self, model_file, vocab_extra_ids=0):
|
||||
super().__init__(model_file, vocab_extra_ids)
|
||||
|
||||
|
||||
def _initalize(self, vocab_extra_ids):
|
||||
self._populate_vocab()
|
||||
self._special_tokens = {}
|
||||
self._inv_special_tokens = {}
|
||||
|
||||
self._t5_tokens = []
|
||||
|
||||
def _add_special_token(t):
|
||||
if t not in self._vocab:
|
||||
next_id = len(self._vocab)
|
||||
self._vocab[t] = next_id
|
||||
self._inv_vocab[next_id] = t
|
||||
self._special_tokens[t] = self._vocab[t]
|
||||
self._inv_special_tokens[self._vocab[t]] = t
|
||||
|
||||
_add_special_token('<CLS>')
|
||||
self._cls_id = self._vocab['<CLS>']
|
||||
_add_special_token('<SEP>')
|
||||
self._sep_id = self._vocab['<SEP>']
|
||||
_add_special_token('<EOD>')
|
||||
self._eod_id = self._vocab['<EOD>']
|
||||
_add_special_token('<MASK>')
|
||||
self._mask_id = self._vocab['<MASK>']
|
||||
|
||||
_add_special_token('<SOA>')
|
||||
self._soa_id = self._vocab['<SOA>']
|
||||
_add_special_token('<EOA>')
|
||||
self._eoa_id = self._vocab['<EOA>']
|
||||
_add_special_token('<SOV>')
|
||||
self._sov_id = self._vocab['<SOV>']
|
||||
_add_special_token('<EOV>')
|
||||
self._eov_id = self._vocab['<EOV>']
|
||||
_add_special_token('<SOI>')
|
||||
self._soi_id = self._vocab['<SOI>']
|
||||
_add_special_token('<EOI>')
|
||||
self._eoi_id = self._vocab['<EOI>']
|
||||
_add_special_token('<s_local>')
|
||||
self._s_local_id = self._vocab['<s_local>']
|
||||
_add_special_token('<e_local>')
|
||||
self._e_local_id = self._vocab['<e_local>']
|
||||
_add_special_token('<s_global>')
|
||||
self._s_global_id = self._vocab['<s_global>']
|
||||
_add_special_token('<e_global>')
|
||||
self._e_global_id = self._vocab['<e_global>']
|
||||
_add_special_token('<stage_1>')
|
||||
self._stage_1_id = self._vocab['<stage_1>']
|
||||
_add_special_token('<stage_2>')
|
||||
self._stage_2_id = self._vocab['<stage_2>']
|
||||
pad_id = self.tokenizer.pad_id()
|
||||
try:
|
||||
pad_token = self.tokenizer.id_to_piece(pad_id)
|
||||
except IndexError:
|
||||
pad_token = '<PAD>'
|
||||
_add_special_token(pad_token)
|
||||
self._pad_id = self._vocab[pad_token]
|
||||
|
||||
bos_id = self.tokenizer.bos_id()
|
||||
try:
|
||||
bos_token = self.tokenizer.id_to_piece(bos_id)
|
||||
except IndexError:
|
||||
bos_token = '<BOS>'
|
||||
_add_special_token(bos_token)
|
||||
self._bos_id = self._vocab[bos_token]
|
||||
|
||||
eos_id = self.tokenizer.eos_id()
|
||||
try:
|
||||
eos_token = self.tokenizer.id_to_piece(eos_id)
|
||||
except IndexError:
|
||||
eos_token = '<EOS>'
|
||||
_add_special_token(eos_token)
|
||||
self._eos_id = self._vocab[eos_token]
|
||||
|
||||
for i in range(vocab_extra_ids):
|
||||
t = "<extra_id_{}>".format(i)
|
||||
_add_special_token(t)
|
||||
self._t5_tokens += [t]
|
||||
|
||||
@property
|
||||
def soa(self):
|
||||
return self._soa_id
|
||||
|
||||
@property
|
||||
def eoa(self):
|
||||
return self._eoa_id
|
||||
|
||||
@property
|
||||
def sov(self):
|
||||
return self._sov_id
|
||||
|
||||
@property
|
||||
def eov(self):
|
||||
return self._eov_id
|
||||
|
||||
@property
|
||||
def soi(self):
|
||||
return self._soi_id
|
||||
|
||||
@property
|
||||
def eoi(self):
|
||||
return self._eoi_id
|
||||
|
||||
@property
|
||||
def s_local(self):
|
||||
return self._s_local_id
|
||||
|
||||
@property
|
||||
def e_local(self):
|
||||
return self._e_local_id
|
||||
|
||||
@property
|
||||
def s_global(self):
|
||||
return self._s_global_id
|
||||
|
||||
@property
|
||||
def e_global(self):
|
||||
return self._e_global_id
|
||||
|
||||
@property
|
||||
def stage_1(self):
|
||||
return self._stage_1_id
|
||||
|
||||
@property
|
||||
def stage_2(self):
|
||||
return self._stage_2_id
|
||||
@@ -1,3 +0,0 @@
|
||||
---
|
||||
license: apache-2.0
|
||||
---
|
||||
@@ -1,428 +0,0 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) ByteDance, Inc. and its affiliates.
|
||||
Copyright (c) Chutong Meng
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
Attribution-NonCommercial 4.0 International
|
||||
|
||||
=======================================================================
|
||||
|
||||
Creative Commons Corporation ("Creative Commons") is not a law firm and
|
||||
does not provide legal services or legal advice. Distribution of
|
||||
Creative Commons public licenses does not create a lawyer-client or
|
||||
other relationship. Creative Commons makes its licenses and related
|
||||
information available on an "as-is" basis. Creative Commons gives no
|
||||
warranties regarding its licenses, any material licensed under their
|
||||
terms and conditions, or any related information. Creative Commons
|
||||
disclaims all liability for damages resulting from their use to the
|
||||
fullest extent possible.
|
||||
|
||||
Using Creative Commons Public Licenses
|
||||
|
||||
Creative Commons public licenses provide a standard set of terms and
|
||||
conditions that creators and other rights holders may use to share
|
||||
original works of authorship and other material subject to copyright
|
||||
and certain other rights specified in the public license below. The
|
||||
following considerations are for informational purposes only, are not
|
||||
exhaustive, and do not form part of our licenses.
|
||||
|
||||
Considerations for licensors: Our public licenses are
|
||||
intended for use by those authorized to give the public
|
||||
permission to use material in ways otherwise restricted by
|
||||
copyright and certain other rights. Our licenses are
|
||||
irrevocable. Licensors should read and understand the terms
|
||||
and conditions of the license they choose before applying it.
|
||||
Licensors should also secure all rights necessary before
|
||||
applying our licenses so that the public can reuse the
|
||||
material as expected. Licensors should clearly mark any
|
||||
material not subject to the license. This includes other CC-
|
||||
licensed material, or material used under an exception or
|
||||
limitation to copyright. More considerations for licensors:
|
||||
wiki.creativecommons.org/Considerations_for_licensors
|
||||
|
||||
Considerations for the public: By using one of our public
|
||||
licenses, a licensor grants the public permission to use the
|
||||
licensed material under specified terms and conditions. If
|
||||
the licensor's permission is not necessary for any reason--for
|
||||
example, because of any applicable exception or limitation to
|
||||
copyright--then that use is not regulated by the license. Our
|
||||
licenses grant only permissions under copyright and certain
|
||||
other rights that a licensor has authority to grant. Use of
|
||||
the licensed material may still be restricted for other
|
||||
reasons, including because others have copyright or other
|
||||
rights in the material. A licensor may make special requests,
|
||||
such as asking that all changes be marked or described.
|
||||
Although not required by our licenses, you are encouraged to
|
||||
respect those requests where reasonable. More_considerations
|
||||
for the public:
|
||||
wiki.creativecommons.org/Considerations_for_licensees
|
||||
|
||||
=======================================================================
|
||||
|
||||
Creative Commons Attribution-NonCommercial 4.0 International Public
|
||||
License
|
||||
|
||||
By exercising the Licensed Rights (defined below), You accept and agree
|
||||
to be bound by the terms and conditions of this Creative Commons
|
||||
Attribution-NonCommercial 4.0 International Public License ("Public
|
||||
License"). To the extent this Public License may be interpreted as a
|
||||
contract, You are granted the Licensed Rights in consideration of Your
|
||||
acceptance of these terms and conditions, and the Licensor grants You
|
||||
such rights in consideration of benefits the Licensor receives from
|
||||
making the Licensed Material available under these terms and
|
||||
conditions.
|
||||
|
||||
Section 1 -- Definitions.
|
||||
|
||||
a. Adapted Material means material subject to Copyright and Similar
|
||||
Rights that is derived from or based upon the Licensed Material
|
||||
and in which the Licensed Material is translated, altered,
|
||||
arranged, transformed, or otherwise modified in a manner requiring
|
||||
permission under the Copyright and Similar Rights held by the
|
||||
Licensor. For purposes of this Public License, where the Licensed
|
||||
Material is a musical work, performance, or sound recording,
|
||||
Adapted Material is always produced where the Licensed Material is
|
||||
synched in timed relation with a moving image.
|
||||
|
||||
b. Adapter's License means the license You apply to Your Copyright
|
||||
and Similar Rights in Your contributions to Adapted Material in
|
||||
accordance with the terms and conditions of this Public License.
|
||||
|
||||
c. Copyright and Similar Rights means copyright and/or similar rights
|
||||
closely related to copyright including, without limitation,
|
||||
performance, broadcast, sound recording, and Sui Generis Database
|
||||
Rights, without regard to how the rights are labeled or
|
||||
categorized. For purposes of this Public License, the rights
|
||||
specified in Section 2(b)(1)-(2) are not Copyright and Similar
|
||||
Rights.
|
||||
d. Effective Technological Measures means those measures that, in the
|
||||
absence of proper authority, may not be circumvented under laws
|
||||
fulfilling obligations under Article 11 of the WIPO Copyright
|
||||
Treaty adopted on December 20, 1996, and/or similar international
|
||||
agreements.
|
||||
|
||||
e. Exceptions and Limitations means fair use, fair dealing, and/or
|
||||
any other exception or limitation to Copyright and Similar Rights
|
||||
that applies to Your use of the Licensed Material.
|
||||
|
||||
f. Licensed Material means the artistic or literary work, database,
|
||||
or other material to which the Licensor applied this Public
|
||||
License.
|
||||
|
||||
g. Licensed Rights means the rights granted to You subject to the
|
||||
terms and conditions of this Public License, which are limited to
|
||||
all Copyright and Similar Rights that apply to Your use of the
|
||||
Licensed Material and that the Licensor has authority to license.
|
||||
|
||||
h. Licensor means the individual(s) or entity(ies) granting rights
|
||||
under this Public License.
|
||||
|
||||
i. NonCommercial means not primarily intended for or directed towards
|
||||
commercial advantage or monetary compensation. For purposes of
|
||||
this Public License, the exchange of the Licensed Material for
|
||||
other material subject to Copyright and Similar Rights by digital
|
||||
file-sharing or similar means is NonCommercial provided there is
|
||||
no payment of monetary compensation in connection with the
|
||||
exchange.
|
||||
|
||||
j. Share means to provide material to the public by any means or
|
||||
process that requires permission under the Licensed Rights, such
|
||||
as reproduction, public display, public performance, distribution,
|
||||
dissemination, communication, or importation, and to make material
|
||||
available to the public including in ways that members of the
|
||||
public may access the material from a place and at a time
|
||||
individually chosen by them.
|
||||
|
||||
k. Sui Generis Database Rights means rights other than copyright
|
||||
resulting from Directive 96/9/EC of the European Parliament and of
|
||||
the Council of 11 March 1996 on the legal protection of databases,
|
||||
as amended and/or succeeded, as well as other essentially
|
||||
equivalent rights anywhere in the world.
|
||||
|
||||
l. You means the individual or entity exercising the Licensed Rights
|
||||
under this Public License. Your has a corresponding meaning.
|
||||
|
||||
Section 2 -- Scope.
|
||||
|
||||
a. License grant.
|
||||
|
||||
1. Subject to the terms and conditions of this Public License,
|
||||
the Licensor hereby grants You a worldwide, royalty-free,
|
||||
non-sublicensable, non-exclusive, irrevocable license to
|
||||
exercise the Licensed Rights in the Licensed Material to:
|
||||
|
||||
a. reproduce and Share the Licensed Material, in whole or
|
||||
in part, for NonCommercial purposes only; and
|
||||
|
||||
b. produce, reproduce, and Share Adapted Material for
|
||||
NonCommercial purposes only.
|
||||
|
||||
2. Exceptions and Limitations. For the avoidance of doubt, where
|
||||
Exceptions and Limitations apply to Your use, this Public
|
||||
License does not apply, and You do not need to comply with
|
||||
its terms and conditions.
|
||||
|
||||
3. Term. The term of this Public License is specified in Section
|
||||
6(a).
|
||||
|
||||
4. Media and formats; technical modifications allowed. The
|
||||
Licensor authorizes You to exercise the Licensed Rights in
|
||||
all media and formats whether now known or hereafter created,
|
||||
and to make technical modifications necessary to do so. The
|
||||
Licensor waives and/or agrees not to assert any right or
|
||||
authority to forbid You from making technical modifications
|
||||
necessary to exercise the Licensed Rights, including
|
||||
technical modifications necessary to circumvent Effective
|
||||
Technological Measures. For purposes of this Public License,
|
||||
simply making modifications authorized by this Section 2(a)
|
||||
(4) never produces Adapted Material.
|
||||
|
||||
5. Downstream recipients.
|
||||
|
||||
a. Offer from the Licensor -- Licensed Material. Every
|
||||
recipient of the Licensed Material automatically
|
||||
receives an offer from the Licensor to exercise the
|
||||
Licensed Rights under the terms and conditions of this
|
||||
Public License.
|
||||
|
||||
b. No downstream restrictions. You may not offer or impose
|
||||
any additional or different terms or conditions on, or
|
||||
apply any Effective Technological Measures to, the
|
||||
Licensed Material if doing so restricts exercise of the
|
||||
Licensed Rights by any recipient of the Licensed
|
||||
Material.
|
||||
|
||||
6. No endorsement. Nothing in this Public License constitutes or
|
||||
may be construed as permission to assert or imply that You
|
||||
are, or that Your use of the Licensed Material is, connected
|
||||
with, or sponsored, endorsed, or granted official status by,
|
||||
the Licensor or others designated to receive attribution as
|
||||
provided in Section 3(a)(1)(A)(i).
|
||||
|
||||
b. Other rights.
|
||||
|
||||
1. Moral rights, such as the right of integrity, are not
|
||||
licensed under this Public License, nor are publicity,
|
||||
privacy, and/or other similar personality rights; however, to
|
||||
the extent possible, the Licensor waives and/or agrees not to
|
||||
assert any such rights held by the Licensor to the limited
|
||||
extent necessary to allow You to exercise the Licensed
|
||||
Rights, but not otherwise.
|
||||
|
||||
2. Patent and trademark rights are not licensed under this
|
||||
Public License.
|
||||
|
||||
3. To the extent possible, the Licensor waives any right to
|
||||
collect royalties from You for the exercise of the Licensed
|
||||
Rights, whether directly or through a collecting society
|
||||
under any voluntary or waivable statutory or compulsory
|
||||
licensing scheme. In all other cases the Licensor expressly
|
||||
reserves any right to collect such royalties, including when
|
||||
the Licensed Material is used other than for NonCommercial
|
||||
purposes.
|
||||
|
||||
Section 3 -- License Conditions.
|
||||
|
||||
Your exercise of the Licensed Rights is expressly made subject to the
|
||||
following conditions.
|
||||
|
||||
a. Attribution.
|
||||
|
||||
1. If You Share the Licensed Material (including in modified
|
||||
form), You must:
|
||||
|
||||
a. retain the following if it is supplied by the Licensor
|
||||
with the Licensed Material:
|
||||
|
||||
i. identification of the creator(s) of the Licensed
|
||||
Material and any others designated to receive
|
||||
attribution, in any reasonable manner requested by
|
||||
the Licensor (including by pseudonym if
|
||||
designated);
|
||||
|
||||
ii. a copyright notice;
|
||||
|
||||
iii. a notice that refers to this Public License;
|
||||
|
||||
iv. a notice that refers to the disclaimer of
|
||||
warranties;
|
||||
|
||||
v. a URI or hyperlink to the Licensed Material to the
|
||||
extent reasonably practicable;
|
||||
|
||||
b. indicate if You modified the Licensed Material and
|
||||
retain an indication of any previous modifications; and
|
||||
|
||||
c. indicate the Licensed Material is licensed under this
|
||||
Public License, and include the text of, or the URI or
|
||||
hyperlink to, this Public License.
|
||||
|
||||
2. You may satisfy the conditions in Section 3(a)(1) in any
|
||||
reasonable manner based on the medium, means, and context in
|
||||
which You Share the Licensed Material. For example, it may be
|
||||
reasonable to satisfy the conditions by providing a URI or
|
||||
hyperlink to a resource that includes the required
|
||||
information.
|
||||
|
||||
3. If requested by the Licensor, You must remove any of the
|
||||
information required by Section 3(a)(1)(A) to the extent
|
||||
reasonably practicable.
|
||||
|
||||
4. If You Share Adapted Material You produce, the Adapter's
|
||||
License You apply must not prevent recipients of the Adapted
|
||||
Material from complying with this Public License.
|
||||
|
||||
Section 4 -- Sui Generis Database Rights.
|
||||
|
||||
Where the Licensed Rights include Sui Generis Database Rights that
|
||||
apply to Your use of the Licensed Material:
|
||||
|
||||
a. for the avoidance of doubt, Section 2(a)(1) grants You the right
|
||||
to extract, reuse, reproduce, and Share all or a substantial
|
||||
portion of the contents of the database for NonCommercial purposes
|
||||
only;
|
||||
|
||||
b. if You include all or a substantial portion of the database
|
||||
contents in a database in which You have Sui Generis Database
|
||||
Rights, then the database in which You have Sui Generis Database
|
||||
Rights (but not its individual contents) is Adapted Material; and
|
||||
|
||||
c. You must comply with the conditions in Section 3(a) if You Share
|
||||
all or a substantial portion of the contents of the database.
|
||||
|
||||
For the avoidance of doubt, this Section 4 supplements and does not
|
||||
replace Your obligations under this Public License where the Licensed
|
||||
Rights include other Copyright and Similar Rights.
|
||||
|
||||
Section 5 -- Disclaimer of Warranties and Limitation of Liability.
|
||||
|
||||
a. UNLESS OTHERWISE SEPARATELY UNDERTAKEN BY THE LICENSOR, TO THE
|
||||
EXTENT POSSIBLE, THE LICENSOR OFFERS THE LICENSED MATERIAL AS-IS
|
||||
AND AS-AVAILABLE, AND MAKES NO REPRESENTATIONS OR WARRANTIES OF
|
||||
ANY KIND CONCERNING THE LICENSED MATERIAL, WHETHER EXPRESS,
|
||||
IMPLIED, STATUTORY, OR OTHER. THIS INCLUDES, WITHOUT LIMITATION,
|
||||
WARRANTIES OF TITLE, MERCHANTABILITY, FITNESS FOR A PARTICULAR
|
||||
PURPOSE, NON-INFRINGEMENT, ABSENCE OF LATENT OR OTHER DEFECTS,
|
||||
ACCURACY, OR THE PRESENCE OR ABSENCE OF ERRORS, WHETHER OR NOT
|
||||
KNOWN OR DISCOVERABLE. WHERE DISCLAIMERS OF WARRANTIES ARE NOT
|
||||
ALLOWED IN FULL OR IN PART, THIS DISCLAIMER MAY NOT APPLY TO YOU.
|
||||
|
||||
b. TO THE EXTENT POSSIBLE, IN NO EVENT WILL THE LICENSOR BE LIABLE
|
||||
TO YOU ON ANY LEGAL THEORY (INCLUDING, WITHOUT LIMITATION,
|
||||
NEGLIGENCE) OR OTHERWISE FOR ANY DIRECT, SPECIAL, INDIRECT,
|
||||
INCIDENTAL, CONSEQUENTIAL, PUNITIVE, EXEMPLARY, OR OTHER LOSSES,
|
||||
COSTS, EXPENSES, OR DAMAGES ARISING OUT OF THIS PUBLIC LICENSE OR
|
||||
USE OF THE LICENSED MATERIAL, EVEN IF THE LICENSOR HAS BEEN
|
||||
ADVISED OF THE POSSIBILITY OF SUCH LOSSES, COSTS, EXPENSES, OR
|
||||
DAMAGES. WHERE A LIMITATION OF LIABILITY IS NOT ALLOWED IN FULL OR
|
||||
IN PART, THIS LIMITATION MAY NOT APPLY TO YOU.
|
||||
|
||||
c. The disclaimer of warranties and limitation of liability provided
|
||||
above shall be interpreted in a manner that, to the extent
|
||||
possible, most closely approximates an absolute disclaimer and
|
||||
waiver of all liability.
|
||||
|
||||
Section 6 -- Term and Termination.
|
||||
|
||||
a. This Public License applies for the term of the Copyright and
|
||||
Similar Rights licensed here. However, if You fail to comply with
|
||||
this Public License, then Your rights under this Public License
|
||||
terminate automatically.
|
||||
|
||||
b. Where Your right to use the Licensed Material has terminated under
|
||||
Section 6(a), it reinstates:
|
||||
|
||||
1. automatically as of the date the violation is cured, provided
|
||||
it is cured within 30 days of Your discovery of the
|
||||
violation; or
|
||||
|
||||
2. upon express reinstatement by the Licensor.
|
||||
|
||||
For the avoidance of doubt, this Section 6(b) does not affect any
|
||||
right the Licensor may have to seek remedies for Your violations
|
||||
of this Public License.
|
||||
|
||||
c. For the avoidance of doubt, the Licensor may also offer the
|
||||
Licensed Material under separate terms or conditions or stop
|
||||
distributing the Licensed Material at any time; however, doing so
|
||||
will not terminate this Public License.
|
||||
|
||||
d. Sections 1, 5, 6, 7, and 8 survive termination of this Public
|
||||
License.
|
||||
|
||||
Section 7 -- Other Terms and Conditions.
|
||||
|
||||
a. The Licensor shall not be bound by any additional or different
|
||||
terms or conditions communicated by You unless expressly agreed.
|
||||
|
||||
b. Any arrangements, understandings, or agreements regarding the
|
||||
Licensed Material not stated herein are separate from and
|
||||
independent of the terms and conditions of this Public License.
|
||||
|
||||
Section 8 -- Interpretation.
|
||||
|
||||
a. For the avoidance of doubt, this Public License does not, and
|
||||
shall not be interpreted to, reduce, limit, restrict, or impose
|
||||
conditions on any use of the Licensed Material that could lawfully
|
||||
be made without permission under this Public License.
|
||||
|
||||
b. To the extent possible, if any provision of this Public License is
|
||||
deemed unenforceable, it shall be automatically reformed to the
|
||||
minimum extent necessary to make it enforceable. If the provision
|
||||
cannot be reformed, it shall be severed from this Public License
|
||||
without affecting the enforceability of the remaining terms and
|
||||
conditions.
|
||||
|
||||
c. No term or condition of this Public License will be waived and no
|
||||
failure to comply consented to unless expressly agreed to by the
|
||||
Licensor.
|
||||
|
||||
d. Nothing in this Public License constitutes or may be interpreted
|
||||
as a limitation upon, or waiver of, any privileges and immunities
|
||||
that apply to the Licensor or You, including from the legal
|
||||
processes of any jurisdiction or authority.
|
||||
|
||||
=======================================================================
|
||||
|
||||
Creative Commons is not a party to its public
|
||||
licenses. Notwithstanding, Creative Commons may elect to apply one of
|
||||
its public licenses to material it publishes and in those instances
|
||||
will be considered the “Licensor.” The text of the Creative Commons
|
||||
public licenses is dedicated to the public domain under the CC0 Public
|
||||
Domain Dedication. Except for the limited purpose of indicating that
|
||||
material is shared under a Creative Commons public license or as
|
||||
otherwise permitted by the Creative Commons policies published at
|
||||
creativecommons.org/policies, Creative Commons does not authorize the
|
||||
use of the trademark "Creative Commons" or any other trademark or logo
|
||||
of Creative Commons without its prior written consent including,
|
||||
without limitation, in connection with any unauthorized modifications
|
||||
to any of its public licenses or any other arrangements,
|
||||
understandings, or agreements concerning use of licensed material. For
|
||||
the avoidance of doubt, this paragraph does not form part of the
|
||||
public licenses.
|
||||
|
||||
Creative Commons may be contacted at creativecommons.org.
|
||||
@@ -1,273 +0,0 @@
|
||||
# RepCodec: A Speech Representation Codec for Speech Tokenization
|
||||
|
||||
> [**RepCodec: A Speech Representation Codec for Speech Tokenization**](https://arxiv.org/abs/2309.00169)
|
||||
|
||||
## Introduction
|
||||
|
||||
**RepCodec** is a speech tokenization method for converting a speech waveform into a sequence of discrete semantic
|
||||
tokens.
|
||||
The main idea is to train a representation codec which learns a vector quantization codebook through reconstructing the
|
||||
input speech representations from speech encoders like HuBERT or data2vec.
|
||||
Extensive experiments show that RepCodec significantly outperforms the widely used k-means clustering approach in both
|
||||
speech understanding and generation.
|
||||
Also, RepCodec generalizes well across various speech encoders and languages.
|
||||
|
||||
<img src="images/RepCodec.png" alt="se" width="1000" />
|
||||
|
||||
## RepCodec Models
|
||||
|
||||
| Feature Type | Speech Data | RepCodec Model |
|
||||
|-----------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------|----------------------------------------------------------------------------------------------------------|
|
||||
| [HuBERT base](https://github.com/facebookresearch/fairseq/tree/main/examples/hubert#pre-trained-and-fine-tuned-asr-models) layer 9 | [Librispeech](http://www.openslr.org/12) train-clean-100 | [hubert_base_l9](https://drive.google.com/file/d/1XD0HKl607FFjri2-VJT7lHQeSpxsCCFO/view?usp=sharing) |
|
||||
| [HuBERT large](https://github.com/facebookresearch/fairseq/tree/main/examples/hubert#pre-trained-and-fine-tuned-asr-models) layer 18 | [Librispeech](http://www.openslr.org/12) train-clean-100 | [hubert_large_l18](https://drive.google.com/file/d/1mTbm5GeJ7gp_5L3QLP-JGXdf8RnRw5n6/view?usp=sharing) |
|
||||
| [data2vec base](https://github.com/facebookresearch/fairseq/blob/main/examples/data2vec/README.md#speech-2) layer 6 | [Librispeech](http://www.openslr.org/12) train-clean-100 | [data2vec_base_l6](https://drive.google.com/file/d/1d8sf3Ko_fYM9zlaiwxK_4xusLRKV5EMd/view?usp=sharing) |
|
||||
| [data2vec large](https://github.com/facebookresearch/fairseq/blob/main/examples/data2vec/README.md#speech-2) layer 18 | [Librispeech](http://www.openslr.org/12) train-clean-100 | [data2vec_large_l18](https://drive.google.com/file/d/1nuRIHaejT-uVi4cluftbT8o_JZqar5SU/view?usp=sharing) |
|
||||
| [Whisper medium](https://github.com/openai/whisper/tree/main#available-models-and-languages) layer 24 | [Librispeech](http://www.openslr.org/12) train-clean-100 | [whisper_medium_l24](https://drive.google.com/file/d/1V6YJSA2V4iywXrecJAN0oqsa3aHowexZ/view?usp=sharing) |
|
||||
| [Whisper large-v2](https://github.com/openai/whisper/tree/main#available-models-and-languages) layer 32 | [Librispeech](http://www.openslr.org/12) train-clean-100 | [whisper_large_l32](https://drive.google.com/file/d/1k_X7ZMPg8iOeDrIJe70v6CHfFygzufXC/view?usp=sharing) |
|
||||
|
||||
## Speech Tokenization Using Pre-Trained Models
|
||||
|
||||
### Installation
|
||||
|
||||
Please first install RepCodec by
|
||||
|
||||
```
|
||||
git clone https://github.com/mct10/RepCodec.git
|
||||
cd RepCodec
|
||||
pip install .
|
||||
```
|
||||
|
||||
We used Python 3.9.18 and PyTorch 1.12.1 to test the usage, but the code should be compatible with other recent Python
|
||||
and PyTorch versions.
|
||||
|
||||
### Representation Preparation
|
||||
|
||||
We adapt the `dump_hubert_feature.py` script
|
||||
from [fairseq](https://github.com/facebookresearch/fairseq/tree/main/examples/hubert/simple_kmeans#hubert-feature)
|
||||
to support dumping representations from **data2vec**, **HuBERT**, or **Whisper** encoders.
|
||||
|
||||
If you use our script (`examples/dump_feature.py`), please also install the following packages:
|
||||
|
||||
```
|
||||
pip install npy_append_array soundfile
|
||||
```
|
||||
|
||||
Additionally, if you want to dump representations from
|
||||
|
||||
- **data2vec** or **HuBERT**: please
|
||||
follow [fairseq's instruction](https://github.com/facebookresearch/fairseq#requirements-and-installation) to install
|
||||
the latest fairseq.
|
||||
|
||||
- **Whisper**: please follow [Whispers'instruction](https://github.com/openai/whisper/tree/main#setup) to install the
|
||||
latest
|
||||
Whisper.
|
||||
|
||||
Then, you can follow the given examples to dump representations:
|
||||
|
||||
```
|
||||
# Example 1: dump from HuBERT base layer 9
|
||||
# (for data2vec, simply change "model_type" to data2vec and "ckpt_path" to the path of data2vec model)
|
||||
|
||||
layer=9
|
||||
|
||||
python3 examples/dump_feature.py \
|
||||
--model_type hubert \
|
||||
--tsv_path /path/to/tsv/file \
|
||||
--ckpt_path /path/to/HuBERT/model \
|
||||
--layer ${layer} \
|
||||
--feat_dir /dir/to/save/representations
|
||||
|
||||
|
||||
# Example 2: dump from Whisper medium layer 24
|
||||
|
||||
layer=24
|
||||
|
||||
python3 examples/dump_feature.py \
|
||||
--model_type whisper \
|
||||
--tsv_path /path/to/tsv/file \
|
||||
--whisper_root /directory/to/save/whisper/model \
|
||||
--whisper_name medium \
|
||||
--layer ${layer} \
|
||||
--feat_dir /dir/to/save/representations
|
||||
```
|
||||
|
||||
Explanations about the args:
|
||||
|
||||
- **model_type:** choose from `data2vec`, `hubert`, and `whisper`.
|
||||
|
||||
- **tsv_path:** path of the tsv file.
|
||||
Should have the format of
|
||||
|
||||
```
|
||||
/dir/to/dataset
|
||||
path_of_utterance_1 number_of_frames
|
||||
path_of_utterance_2 number_of_frames
|
||||
```
|
||||
|
||||
You can follow [this script](https://github.com/facebookresearch/fairseq/blob/main/examples/wav2vec/wav2vec_manifest.py)
|
||||
to generate the tsv file.
|
||||
|
||||
For example, by running
|
||||
|
||||
```
|
||||
python wav2vec_manifest.py \
|
||||
/dir/to/LibriSpeech/dev-clean \
|
||||
--dest /dir/to/manifest \
|
||||
--ext flac \
|
||||
--valid-percent 0
|
||||
```
|
||||
|
||||
you can obtain the `dev-clean.tsv` in `/dir/to/manifest` for LibriSpeech. (By default, the output file name
|
||||
is `train.tsv`. Remember to rename the file.)
|
||||
|
||||
It should be similar to:
|
||||
|
||||
```
|
||||
/dir/to/LibriSpeech/dev-clean
|
||||
2277/149896/2277-149896-0026.flac 78720
|
||||
2277/149896/2277-149896-0005.flac 89600
|
||||
2277/149896/2277-149896-0033.flac 45520
|
||||
```
|
||||
|
||||
- **ckpt_path**:
|
||||
must provide for data2vec and HuBERT.
|
||||
You need to download the model
|
||||
from [data2vec website](https://github.com/facebookresearch/fairseq/blob/main/examples/data2vec/README.md#speech-2)
|
||||
or [HuBERT website](https://github.com/facebookresearch/fairseq/tree/main/examples/hubert#pre-trained-and-fine-tuned-asr-models)
|
||||
yourself.
|
||||
`--ckpt_path` is the path of the data2vec/HuBERT model.
|
||||
- **whisper_root** and **whisper_name**:
|
||||
must provide **BOTH** `--whisper_root` and `--whisper_name` for Whisper.
|
||||
If there is no corresponding model in `--whisper_root`, the script will download for you.
|
||||
|
||||
- **layer**:
|
||||
which Transformer encoder layer of the model should the representations be extracted from.
|
||||
It is **1-based**.
|
||||
For example, if layer=9, then the outputs from the 9<sup>th</sup> Transformer encoder layer are dumped.
|
||||
Range: [1, number of Transformer encoder layers]
|
||||
|
||||
- **feat_dir**: The output representations will be saved to `${feat_dir}/0_1.npy`
|
||||
and `${feat_dir}/0_1.len`.
|
||||
|
||||
For other useful functionalities (e.g., sharding), please check the argument list in `examples/dump_feature.py`.
|
||||
|
||||
### Command Line Usage
|
||||
|
||||
We expect to have `${feat_dir}/0_1.npy` and `${feat_dir}/0_1.len` in the provided
|
||||
directory `/dir/to/representaitons`.
|
||||
|
||||
Also, the tsv file should be the **same** as the one used in [Representation Preparation](#representation-preparation).
|
||||
|
||||
```
|
||||
repcodec /dir/to/representaitons \
|
||||
--model /path/to/repcodec/model \
|
||||
--tsv_path /path/to/tsv/file \
|
||||
[--model_config_path /path/to/train/config] \
|
||||
[--use_gpu] \
|
||||
[--out_dir /path/to/output]
|
||||
```
|
||||
|
||||
If you trained the model yourself following [Training New RepCodec Models](#training-new-repcodec-models),
|
||||
please provide the training config file using `--model_config_path`.
|
||||
If you use the model we provide [here](#repcodec-models), then you do not have to provide that.
|
||||
|
||||
This command will tokenize the representations and the output discrete tokens will be saved to `${out_dir}/tokens`.
|
||||
The tokens are in the same order as the provided tsv file.
|
||||
|
||||
An example of the output file:
|
||||
|
||||
```
|
||||
/dir/to/LibriSpeech/dev-clean
|
||||
2277/149896/2277-149896-0026.flac 696 696 198 198 198 498 ...
|
||||
2277/149896/2277-149896-0005.flac 696 696 198 198 198 907 ...
|
||||
2277/149896/2277-149896-0033.flac 696 696 198 198 198 696 ...
|
||||
```
|
||||
|
||||
Under `examples/tokens`, we provide some token files as references. They are obtained from LibriSpeech dev-clean subset
|
||||
using the 6 types of representations and corresponding [RepCodec Models](#repcodec-models).
|
||||
Your results should be very similar to ours.
|
||||
|
||||
### Python Usage
|
||||
|
||||
```python
|
||||
import torch
|
||||
import yaml
|
||||
|
||||
from repcodec.RepCodec import RepCodec
|
||||
|
||||
# for feature types of HubERT base & data2vec base, please use repcodec_dim768.yaml;
|
||||
# for feature types of HuBERT large & data2vec large & Whisper medium, please use repcodec_dim1024.yaml;
|
||||
# for feature types of Whisper large-v2, please use repcodec_dim1280.yaml
|
||||
config = "repcodec/configs/repcodec_dim768.yaml"
|
||||
with open(config) as fp:
|
||||
conf = yaml.load(fp, Loader=yaml.FullLoader)
|
||||
|
||||
model = RepCodec(**conf)
|
||||
model.load_state_dict(torch.load("./hubert_base_l9.pkl", map_location="cpu")["model"]["repcodec"])
|
||||
model.quantizer.initial()
|
||||
model.eval()
|
||||
|
||||
# input shape: (batch size, hidden dim, sequence length)
|
||||
random_features = torch.randn(size=(1, 768, 100))
|
||||
with torch.no_grad():
|
||||
x = model.encoder(random_features)
|
||||
z = model.projector(x)
|
||||
_, idx = model.quantizer.codebook.forward_index(z.transpose(2, 1))
|
||||
tokens = idx.cpu().data.numpy().tolist()[0]
|
||||
```
|
||||
|
||||
## Training New RepCodec Models
|
||||
|
||||
We use a config file to set up all the training configurations, e.g., data, model architecture,
|
||||
optimizer, scheduler.
|
||||
We provide an example [here](./train_configs/ex_dim768_mse.yaml).
|
||||
|
||||
Please first install required packages following [Installation](#installation)
|
||||
and prepare the representations following [Representation Preparation](#representation-preparation).
|
||||
|
||||
The input data directory is expected to have the following structure
|
||||
```
|
||||
/dir/to/representations/
|
||||
train_set_name/
|
||||
0_1.npy
|
||||
0_1.len
|
||||
valid_set_name/
|
||||
0_1.npy
|
||||
0_1.len
|
||||
test_set_name/
|
||||
0_1.npy
|
||||
0_1.len
|
||||
```
|
||||
|
||||
The names of subsets should be the same as the fields in the config file.
|
||||
|
||||
Then, you can run training by
|
||||
```
|
||||
python train.py \
|
||||
-c /path/to/config/file \
|
||||
--tag $tag \
|
||||
--exp_root exp
|
||||
```
|
||||
|
||||
`tag` is the name of the output folder.
|
||||
All outputs will be saved to `exp_root/tag/`.
|
||||
|
||||
## Acknowledge
|
||||
|
||||
Our implementation is based on [facebookresearch/AudioDec](https://github.com/facebookresearch/AudioDec).
|
||||
We thank them for open-sourcing their code!
|
||||
|
||||
## Citation
|
||||
|
||||
If you find our work useful, please cite the following article.
|
||||
|
||||
```
|
||||
@misc{huang2023repcodec,
|
||||
title={RepCodec: A Speech Representation Codec for Speech Tokenization},
|
||||
author={Zhichao Huang and Chutong Meng and Tom Ko},
|
||||
year={2023},
|
||||
eprint={2309.00169},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={eess.AS}
|
||||
}
|
||||
```
|
||||
@@ -1,541 +0,0 @@
|
||||
# Copyright (c) ByteDance, Inc. and its affiliates.
|
||||
# Copyright (c) Chutong Meng
|
||||
#
|
||||
# This source code is licensed under the MIT license found in the
|
||||
# LICENSE file in the root directory of this source tree.
|
||||
# Based on fairseq (https://github.com/facebookresearch/fairseq)
|
||||
|
||||
# ref: https://github.com/facebookresearch/fairseq/blob/main/examples/data2vec/models/data2vec_audio.py
|
||||
|
||||
import logging
|
||||
import math
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Optional
|
||||
|
||||
from omegaconf import II
|
||||
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch.nn.functional as F
|
||||
import torch.distributed as dist
|
||||
|
||||
from fairseq.modules import EMAModule, EMAModuleConfig
|
||||
from fairseq.data.data_utils import compute_mask_indices
|
||||
from fairseq.models import BaseFairseqModel, register_model
|
||||
from fairseq.models.wav2vec import (
|
||||
ConvFeatureExtractionModel,
|
||||
Wav2Vec2Config,
|
||||
TransformerEncoder,
|
||||
)
|
||||
from fairseq.modules import (
|
||||
GradMultiply,
|
||||
LayerNorm,
|
||||
)
|
||||
from fairseq.utils import index_put
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Data2VecAudioConfig(Wav2Vec2Config):
|
||||
|
||||
loss_beta: float = field(
|
||||
default=0, metadata={"help": "beta for smooth l1 loss. 0 means use l2 loss"}
|
||||
)
|
||||
loss_scale: Optional[float] = field(
|
||||
default=None,
|
||||
metadata={
|
||||
"help": "scale the reconstruction loss by this constant. if None then scales by 1/sqrt(dim)"
|
||||
},
|
||||
)
|
||||
average_top_k_layers: int = field(
|
||||
default=8, metadata={"help": "how many layers to average"}
|
||||
)
|
||||
|
||||
layer_norm_target_layer: bool = False
|
||||
instance_norm_target_layer: bool = False
|
||||
instance_norm_targets: bool = False
|
||||
layer_norm_targets: bool = False
|
||||
batch_norm_target_layer: bool = False
|
||||
group_norm_target_layer: bool = False
|
||||
|
||||
ema_decay: float = field(default=0.999, metadata={"help": "initial ema decay rate"})
|
||||
ema_end_decay: float = field(
|
||||
default=0.9999, metadata={"help": "final ema decay rate"}
|
||||
)
|
||||
|
||||
# when to finish annealing ema decay rate
|
||||
ema_anneal_end_step: int = II("optimization.max_update")
|
||||
|
||||
ema_transformer_only: bool = field(
|
||||
default=True,
|
||||
metadata={"help": "whether to momentum update only the transformer"},
|
||||
)
|
||||
ema_layers_only: bool = field(
|
||||
default=True,
|
||||
metadata={"help": "whether to momentum update only the transformer layers"},
|
||||
)
|
||||
|
||||
max_update: int = II("optimization.max_update")
|
||||
|
||||
min_target_var: float = field(
|
||||
default=0.1, metadata={"help": "stop training if target var falls below this"}
|
||||
)
|
||||
min_pred_var: float = field(
|
||||
default=0.01,
|
||||
metadata={"help": "stop training if prediction var falls below this"},
|
||||
)
|
||||
|
||||
|
||||
def get_annealed_rate(start, end, curr_step, total_steps):
|
||||
r = end - start
|
||||
pct_remaining = 1 - curr_step / total_steps
|
||||
return end - r * pct_remaining
|
||||
|
||||
|
||||
@register_model("data2vec_audio", dataclass=Data2VecAudioConfig)
|
||||
class Data2VecAudioModel(BaseFairseqModel):
|
||||
def __init__(self, cfg: Data2VecAudioConfig):
|
||||
super().__init__()
|
||||
self.cfg = cfg
|
||||
|
||||
feature_enc_layers = eval(cfg.conv_feature_layers)
|
||||
self.extractor_embed = feature_enc_layers[-1][0]
|
||||
|
||||
self.ema = None
|
||||
self.embed = cfg.encoder_embed_dim
|
||||
|
||||
self.average_top_k_layers = cfg.average_top_k_layers
|
||||
self.loss_beta = cfg.loss_beta
|
||||
self.loss_scale = cfg.loss_scale
|
||||
|
||||
self.feature_extractor = ConvFeatureExtractionModel(
|
||||
conv_layers=feature_enc_layers,
|
||||
dropout=0.0,
|
||||
mode=cfg.extractor_mode,
|
||||
conv_bias=cfg.conv_bias,
|
||||
)
|
||||
|
||||
self.post_extract_proj = nn.Linear(self.extractor_embed, cfg.encoder_embed_dim)
|
||||
|
||||
self.mask_prob = cfg.mask_prob
|
||||
self.mask_selection = cfg.mask_selection
|
||||
self.mask_other = cfg.mask_other
|
||||
self.mask_length = cfg.mask_length
|
||||
self.no_mask_overlap = cfg.no_mask_overlap
|
||||
self.mask_min_space = cfg.mask_min_space
|
||||
|
||||
self.mask_channel_prob = cfg.mask_channel_prob
|
||||
self.mask_channel_before = cfg.mask_channel_before
|
||||
self.mask_channel_selection = cfg.mask_channel_selection
|
||||
self.mask_channel_other = cfg.mask_channel_other
|
||||
self.mask_channel_length = cfg.mask_channel_length
|
||||
self.no_mask_channel_overlap = cfg.no_mask_channel_overlap
|
||||
self.mask_channel_min_space = cfg.mask_channel_min_space
|
||||
|
||||
self.dropout_input = nn.Dropout(cfg.dropout_input)
|
||||
self.dropout_features = nn.Dropout(cfg.dropout_features)
|
||||
|
||||
self.feature_grad_mult = cfg.feature_grad_mult
|
||||
|
||||
self.mask_emb = nn.Parameter(
|
||||
torch.FloatTensor(cfg.encoder_embed_dim).uniform_()
|
||||
)
|
||||
|
||||
self.encoder = TransformerEncoder(cfg)
|
||||
self.layer_norm = LayerNorm(self.extractor_embed)
|
||||
|
||||
self.final_proj = nn.Linear(self.embed, self.embed)
|
||||
|
||||
self.num_updates = 0
|
||||
|
||||
def make_ema_teacher(self):
|
||||
ema_config = EMAModuleConfig(
|
||||
ema_decay=self.cfg.ema_decay,
|
||||
ema_fp32=True,
|
||||
)
|
||||
skip_keys = set()
|
||||
if self.cfg.ema_layers_only:
|
||||
self.cfg.ema_transformer_only = True
|
||||
for k, _ in self.encoder.pos_conv.named_parameters():
|
||||
skip_keys.add(f"pos_conv.{k}")
|
||||
|
||||
self.ema = EMAModule(
|
||||
self.encoder if self.cfg.ema_transformer_only else self,
|
||||
ema_config,
|
||||
skip_keys=skip_keys,
|
||||
)
|
||||
|
||||
def set_num_updates(self, num_updates):
|
||||
super().set_num_updates(num_updates)
|
||||
|
||||
if self.ema is None and self.final_proj is not None:
|
||||
logger.info(f"making ema teacher")
|
||||
self.make_ema_teacher()
|
||||
elif self.training and self.ema is not None:
|
||||
if self.cfg.ema_decay != self.cfg.ema_end_decay:
|
||||
if num_updates >= self.cfg.ema_anneal_end_step:
|
||||
decay = self.cfg.ema_end_decay
|
||||
else:
|
||||
decay = get_annealed_rate(
|
||||
self.cfg.ema_decay,
|
||||
self.cfg.ema_end_decay,
|
||||
num_updates,
|
||||
self.cfg.ema_anneal_end_step,
|
||||
)
|
||||
self.ema.set_decay(decay)
|
||||
if self.ema.get_decay() < 1:
|
||||
self.ema.step(self.encoder if self.cfg.ema_transformer_only else self)
|
||||
|
||||
self.num_updates = num_updates
|
||||
|
||||
def state_dict(self, destination=None, prefix="", keep_vars=False):
|
||||
state = super().state_dict(destination, prefix, keep_vars)
|
||||
|
||||
if self.ema is not None:
|
||||
state[prefix + "_ema"] = self.ema.fp32_params
|
||||
|
||||
return state
|
||||
|
||||
def _load_from_state_dict(self, state_dict, prefix, *args, **kwargs):
|
||||
if self.ema is not None:
|
||||
k = prefix + "_ema"
|
||||
assert k in state_dict
|
||||
self.ema.restore(state_dict[k], True)
|
||||
del state_dict[k]
|
||||
return super()._load_from_state_dict(state_dict, prefix, *args, **kwargs)
|
||||
|
||||
@classmethod
|
||||
def build_model(cls, cfg: Data2VecAudioConfig, task=None):
|
||||
"""Build a new model instance."""
|
||||
|
||||
return cls(cfg)
|
||||
|
||||
def apply_mask(
|
||||
self,
|
||||
x,
|
||||
padding_mask,
|
||||
mask_indices=None,
|
||||
mask_channel_indices=None,
|
||||
):
|
||||
B, T, C = x.shape
|
||||
|
||||
if self.mask_channel_prob > 0 and self.mask_channel_before:
|
||||
mask_channel_indices = compute_mask_indices(
|
||||
(B, C),
|
||||
None,
|
||||
self.mask_channel_prob,
|
||||
self.mask_channel_length,
|
||||
self.mask_channel_selection,
|
||||
self.mask_channel_other,
|
||||
no_overlap=self.no_mask_channel_overlap,
|
||||
min_space=self.mask_channel_min_space,
|
||||
)
|
||||
mask_channel_indices = (
|
||||
torch.from_numpy(mask_channel_indices)
|
||||
.to(x.device)
|
||||
.unsqueeze(1)
|
||||
.expand(-1, T, -1)
|
||||
)
|
||||
x[mask_channel_indices] = 0
|
||||
|
||||
if self.mask_prob > 0:
|
||||
if mask_indices is None:
|
||||
mask_indices = compute_mask_indices(
|
||||
(B, T),
|
||||
padding_mask,
|
||||
self.mask_prob,
|
||||
self.mask_length,
|
||||
self.mask_selection,
|
||||
self.mask_other,
|
||||
min_masks=1,
|
||||
no_overlap=self.no_mask_overlap,
|
||||
min_space=self.mask_min_space,
|
||||
require_same_masks=self.cfg.require_same_masks,
|
||||
mask_dropout=self.cfg.mask_dropout,
|
||||
)
|
||||
mask_indices = torch.from_numpy(mask_indices).to(x.device)
|
||||
x = index_put(x, mask_indices, self.mask_emb)
|
||||
else:
|
||||
mask_indices = None
|
||||
|
||||
if self.mask_channel_prob > 0 and not self.mask_channel_before:
|
||||
if mask_channel_indices is None:
|
||||
mask_channel_indices = compute_mask_indices(
|
||||
(B, C),
|
||||
None,
|
||||
self.mask_channel_prob,
|
||||
self.mask_channel_length,
|
||||
self.mask_channel_selection,
|
||||
self.mask_channel_other,
|
||||
no_overlap=self.no_mask_channel_overlap,
|
||||
min_space=self.mask_channel_min_space,
|
||||
)
|
||||
mask_channel_indices = (
|
||||
torch.from_numpy(mask_channel_indices)
|
||||
.to(x.device)
|
||||
.unsqueeze(1)
|
||||
.expand(-1, T, -1)
|
||||
)
|
||||
x = index_put(x, mask_channel_indices, 0)
|
||||
|
||||
return x, mask_indices
|
||||
|
||||
def _get_feat_extract_output_lengths(self, input_lengths: torch.LongTensor):
|
||||
"""
|
||||
Computes the output length of the convolutional layers
|
||||
"""
|
||||
|
||||
def _conv_out_length(input_length, kernel_size, stride):
|
||||
return torch.floor((input_length - kernel_size) / stride + 1)
|
||||
|
||||
conv_cfg_list = eval(self.cfg.conv_feature_layers)
|
||||
|
||||
for i in range(len(conv_cfg_list)):
|
||||
input_lengths = _conv_out_length(
|
||||
input_lengths, conv_cfg_list[i][1], conv_cfg_list[i][2]
|
||||
)
|
||||
|
||||
return input_lengths.to(torch.long)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
source,
|
||||
padding_mask=None,
|
||||
mask=True,
|
||||
features_only=False,
|
||||
layer=None,
|
||||
mask_indices=None,
|
||||
mask_channel_indices=None,
|
||||
padding_count=None,
|
||||
):
|
||||
features = source
|
||||
|
||||
if self.feature_grad_mult > 0:
|
||||
features = self.feature_extractor(features)
|
||||
if self.feature_grad_mult != 1.0:
|
||||
features = GradMultiply.apply(features, self.feature_grad_mult)
|
||||
else:
|
||||
with torch.no_grad():
|
||||
features = self.feature_extractor(features)
|
||||
|
||||
features = features.transpose(1, 2)
|
||||
|
||||
features = self.layer_norm(features)
|
||||
|
||||
orig_padding_mask = padding_mask
|
||||
|
||||
if padding_mask is not None and padding_mask.any():
|
||||
input_lengths = (1 - padding_mask.long()).sum(-1)
|
||||
# apply conv formula to get real output_lengths
|
||||
output_lengths = self._get_feat_extract_output_lengths(input_lengths)
|
||||
|
||||
padding_mask = torch.zeros(
|
||||
features.shape[:2], dtype=features.dtype, device=features.device
|
||||
)
|
||||
|
||||
# these two operations makes sure that all values
|
||||
# before the output lengths indices are attended to
|
||||
padding_mask[
|
||||
(
|
||||
torch.arange(padding_mask.shape[0], device=padding_mask.device),
|
||||
output_lengths - 1,
|
||||
)
|
||||
] = 1
|
||||
padding_mask = (1 - padding_mask.flip([-1]).cumsum(-1).flip([-1])).bool()
|
||||
else:
|
||||
padding_mask = None
|
||||
|
||||
if self.post_extract_proj is not None:
|
||||
features = self.post_extract_proj(features)
|
||||
|
||||
pre_encoder_features = None
|
||||
if self.cfg.ema_transformer_only:
|
||||
pre_encoder_features = features.clone()
|
||||
|
||||
features = self.dropout_input(features)
|
||||
|
||||
if mask:
|
||||
x, mask_indices = self.apply_mask(
|
||||
features,
|
||||
padding_mask,
|
||||
mask_indices=mask_indices,
|
||||
mask_channel_indices=mask_channel_indices,
|
||||
)
|
||||
else:
|
||||
x = features
|
||||
mask_indices = None
|
||||
|
||||
x, layer_results = self.encoder(
|
||||
x,
|
||||
padding_mask=padding_mask,
|
||||
layer=layer,
|
||||
)
|
||||
|
||||
if features_only:
|
||||
return {
|
||||
"x": x,
|
||||
"padding_mask": padding_mask,
|
||||
"layer_results": layer_results,
|
||||
}
|
||||
|
||||
result = {
|
||||
"losses": {},
|
||||
}
|
||||
|
||||
with torch.no_grad():
|
||||
self.ema.model.eval()
|
||||
|
||||
if self.cfg.ema_transformer_only:
|
||||
y, layer_results = self.ema.model.extract_features(
|
||||
pre_encoder_features,
|
||||
padding_mask=padding_mask,
|
||||
min_layer=self.cfg.encoder_layers - self.average_top_k_layers,
|
||||
)
|
||||
y = {
|
||||
"x": y,
|
||||
"padding_mask": padding_mask,
|
||||
"layer_results": layer_results,
|
||||
}
|
||||
else:
|
||||
y = self.ema.model.extract_features(
|
||||
source=source,
|
||||
padding_mask=orig_padding_mask,
|
||||
mask=False,
|
||||
)
|
||||
|
||||
target_layer_results = [l[2] for l in y["layer_results"]]
|
||||
|
||||
permuted = False
|
||||
if self.cfg.instance_norm_target_layer or self.cfg.batch_norm_target_layer:
|
||||
target_layer_results = [
|
||||
tl.permute(1, 2, 0) for tl in target_layer_results # TBC -> BCT
|
||||
]
|
||||
permuted = True
|
||||
|
||||
if self.cfg.batch_norm_target_layer:
|
||||
target_layer_results = [
|
||||
F.batch_norm(
|
||||
tl.float(), running_mean=None, running_var=None, training=True
|
||||
)
|
||||
for tl in target_layer_results
|
||||
]
|
||||
|
||||
if self.cfg.instance_norm_target_layer:
|
||||
target_layer_results = [
|
||||
F.instance_norm(tl.float()) for tl in target_layer_results
|
||||
]
|
||||
|
||||
if permuted:
|
||||
target_layer_results = [
|
||||
tl.transpose(1, 2) for tl in target_layer_results # BCT -> BTC
|
||||
]
|
||||
|
||||
if self.cfg.group_norm_target_layer:
|
||||
target_layer_results = [
|
||||
F.layer_norm(tl.float(), tl.shape[-2:])
|
||||
for tl in target_layer_results
|
||||
]
|
||||
|
||||
if self.cfg.layer_norm_target_layer:
|
||||
target_layer_results = [
|
||||
F.layer_norm(tl.float(), tl.shape[-1:])
|
||||
for tl in target_layer_results
|
||||
]
|
||||
|
||||
y = sum(target_layer_results) / len(target_layer_results)
|
||||
|
||||
if self.cfg.layer_norm_targets:
|
||||
y = F.layer_norm(y.float(), y.shape[-1:])
|
||||
|
||||
if self.cfg.instance_norm_targets:
|
||||
y = F.instance_norm(y.float().transpose(1, 2)).transpose(1, 2)
|
||||
|
||||
if not permuted:
|
||||
y = y.transpose(0, 1)
|
||||
|
||||
y = y[mask_indices]
|
||||
|
||||
x = x[mask_indices]
|
||||
x = self.final_proj(x)
|
||||
|
||||
sz = x.size(-1)
|
||||
|
||||
if self.loss_beta == 0:
|
||||
loss = F.mse_loss(x.float(), y.float(), reduction="none").sum(dim=-1)
|
||||
else:
|
||||
loss = F.smooth_l1_loss(
|
||||
x.float(), y.float(), reduction="none", beta=self.loss_beta
|
||||
).sum(dim=-1)
|
||||
|
||||
if self.loss_scale is not None:
|
||||
scale = self.loss_scale
|
||||
else:
|
||||
scale = 1 / math.sqrt(sz)
|
||||
|
||||
result["losses"]["regression"] = loss.sum() * scale
|
||||
|
||||
if "sample_size" not in result:
|
||||
result["sample_size"] = loss.numel()
|
||||
|
||||
with torch.no_grad():
|
||||
result["target_var"] = self.compute_var(y)
|
||||
result["pred_var"] = self.compute_var(x.float())
|
||||
|
||||
if self.num_updates > 5000 and result["target_var"] < self.cfg.min_target_var:
|
||||
logger.error(
|
||||
f"target var is {result['target_var'].item()} < {self.cfg.min_target_var}, exiting"
|
||||
)
|
||||
raise Exception(
|
||||
f"target var is {result['target_var'].item()} < {self.cfg.min_target_var}, exiting"
|
||||
)
|
||||
if self.num_updates > 5000 and result["pred_var"] < self.cfg.min_pred_var:
|
||||
logger.error(
|
||||
f"pred var is {result['pred_var'].item()} < {self.cfg.min_pred_var}, exiting"
|
||||
)
|
||||
raise Exception(
|
||||
f"pred var is {result['pred_var'].item()} < {self.cfg.min_pred_var}, exiting"
|
||||
)
|
||||
|
||||
if self.ema is not None:
|
||||
result["ema_decay"] = self.ema.get_decay() * 1000
|
||||
|
||||
return result
|
||||
|
||||
@staticmethod
|
||||
def compute_var(y):
|
||||
y = y.view(-1, y.size(-1))
|
||||
if dist.is_initialized():
|
||||
zc = torch.tensor(y.size(0)).cuda()
|
||||
zs = y.sum(dim=0)
|
||||
zss = (y ** 2).sum(dim=0)
|
||||
|
||||
dist.all_reduce(zc)
|
||||
dist.all_reduce(zs)
|
||||
dist.all_reduce(zss)
|
||||
|
||||
var = zss / (zc - 1) - (zs ** 2) / (zc * (zc - 1))
|
||||
return torch.sqrt(var + 1e-6).mean()
|
||||
else:
|
||||
return torch.sqrt(y.var(dim=0) + 1e-6).mean()
|
||||
|
||||
def extract_features(
|
||||
self, source, padding_mask, mask=False, layer=None
|
||||
):
|
||||
res = self.forward(
|
||||
source,
|
||||
padding_mask,
|
||||
mask=mask,
|
||||
features_only=True,
|
||||
layer=layer,
|
||||
)
|
||||
return res
|
||||
|
||||
def remove_pretraining_modules(self, last_layer=None):
|
||||
self.final_proj = None
|
||||
self.ema = None
|
||||
if last_layer is not None:
|
||||
self.encoder.layers = nn.ModuleList(
|
||||
l for i, l in enumerate(self.encoder.layers) if i <= last_layer
|
||||
)
|
||||
@@ -1,87 +0,0 @@
|
||||
# Copyright (c) ByteDance, Inc. and its affiliates.
|
||||
# Copyright (c) Chutong Meng
|
||||
#
|
||||
# This source code is licensed under the MIT license found in the
|
||||
# LICENSE file in the root directory of this source tree.
|
||||
# Based on fairseq (https://github.com/facebookresearch/fairseq)
|
||||
|
||||
import logging
|
||||
|
||||
import torch
|
||||
import torch.nn.functional as F
|
||||
from fairseq import tasks
|
||||
from fairseq.checkpoint_utils import load_checkpoint_to_cpu
|
||||
from fairseq.data.audio.audio_utils import get_features_or_waveform
|
||||
from omegaconf import OmegaConf
|
||||
|
||||
from data2vec_audio import Data2VecAudioModel
|
||||
|
||||
logger = logging.getLogger("dump_feature")
|
||||
|
||||
|
||||
class Data2vecFeatureReader(object):
|
||||
def __init__(self, ckpt_path: str, layer: int, device: str, max_chunk=1600000):
|
||||
state = load_checkpoint_to_cpu(ckpt_path)
|
||||
cfg = state["cfg"]
|
||||
# load task
|
||||
task = tasks.setup_task(cfg.task, from_checkpoint=True)
|
||||
task.load_state_dict(state["task_state"])
|
||||
# load model config
|
||||
if "layer_type" not in cfg.model:
|
||||
# fix a missing key
|
||||
model_config = {k: v for k, v in cfg.model.items()}
|
||||
model_config["layer_type"] = "transformer"
|
||||
model_config = OmegaConf.create(model_config)
|
||||
else:
|
||||
model_config = cfg.model
|
||||
|
||||
# fix param name in the state
|
||||
state["model"]["final_proj.weight"] = state["model"].pop("final_proj.0.weight")
|
||||
state["model"]["final_proj.bias"] = state["model"].pop("final_proj.0.bias")
|
||||
del state["model"]["_ema"]
|
||||
|
||||
# load model
|
||||
model = Data2VecAudioModel.build_model(model_config)
|
||||
model.load_state_dict(
|
||||
state["model"], strict=True, model_cfg=model_config
|
||||
)
|
||||
|
||||
self.device = device
|
||||
logger.info(f"device = {self.device}")
|
||||
|
||||
self.model = model.eval().to(self.device)
|
||||
self.task = task
|
||||
self.layer = layer - 1 # make it 1-based
|
||||
self.max_chunk = max_chunk
|
||||
logger.info(f"TASK CONFIG:\n{self.task.cfg}")
|
||||
logger.info(f" max_chunk = {self.max_chunk}")
|
||||
|
||||
def read_audio(self, path, ref_len=None):
|
||||
wav = get_features_or_waveform(path, need_waveform=True, use_sample_rate=self.task.cfg.sample_rate)
|
||||
if wav.ndim == 2:
|
||||
wav = wav.mean(-1)
|
||||
assert wav.ndim == 1, wav.ndim
|
||||
if ref_len is not None and abs(ref_len - len(wav)) > 160:
|
||||
logger.warning(f"ref {ref_len} != read {len(wav)} ({path})")
|
||||
return wav
|
||||
|
||||
def get_feats(self, path, ref_len=None):
|
||||
x = self.read_audio(path, ref_len=ref_len)
|
||||
with torch.no_grad():
|
||||
x = torch.from_numpy(x).float().to(self.device)
|
||||
if self.task.cfg.normalize:
|
||||
x = F.layer_norm(x, x.shape)
|
||||
x = x.view(1, -1)
|
||||
|
||||
feat = []
|
||||
for start in range(0, x.size(1), self.max_chunk):
|
||||
x_chunk = x[:, start: start + self.max_chunk]
|
||||
res = self.model.extract_features(
|
||||
source=x_chunk,
|
||||
padding_mask=None,
|
||||
mask=False,
|
||||
layer=self.layer,
|
||||
)
|
||||
feat_chunk = res["x"]
|
||||
feat.append(feat_chunk)
|
||||
return torch.cat(feat, 1).squeeze(0)
|
||||
@@ -1,142 +0,0 @@
|
||||
# Copyright (c) ByteDance, Inc. and its affiliates.
|
||||
# Copyright (c) Chutong Meng
|
||||
#
|
||||
# This source code is licensed under the MIT license found in the
|
||||
# LICENSE file in the root directory of this source tree.
|
||||
# Based on fairseq (https://github.com/facebookresearch/fairseq)
|
||||
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
|
||||
from feature_utils import get_path_iterator, dump_feature
|
||||
|
||||
logging.basicConfig(
|
||||
format="%(asctime)s | %(levelname)s | %(name)s | %(message)s",
|
||||
datefmt="%Y-%m-%d %H:%M:%S",
|
||||
level=os.environ.get("LOGLEVEL", "INFO").upper(),
|
||||
stream=sys.stdout,
|
||||
)
|
||||
logger = logging.getLogger("dump_feature")
|
||||
|
||||
|
||||
def main(
|
||||
model_type: str,
|
||||
tsv_path: str,
|
||||
ckpt_path: str,
|
||||
whisper_root: str,
|
||||
whisper_name: str,
|
||||
layer: int,
|
||||
nshard: int,
|
||||
rank: int,
|
||||
feat_dir: str,
|
||||
max_chunk: int,
|
||||
use_cpu: bool = False
|
||||
):
|
||||
device = "cpu" if use_cpu else "cuda"
|
||||
|
||||
# some checks
|
||||
if model_type in ["hubert", "data2vec"]:
|
||||
assert ckpt_path and os.path.exists(ckpt_path)
|
||||
elif model_type in ["whisper"]:
|
||||
assert whisper_name and whisper_root
|
||||
else:
|
||||
raise ValueError(f"Unsupported model type {model_type}")
|
||||
|
||||
reader = None
|
||||
if model_type == "hubert":
|
||||
from hubert_feature_reader import HubertFeatureReader
|
||||
reader = HubertFeatureReader(ckpt_path, layer, device=device, max_chunk=max_chunk)
|
||||
elif model_type == "data2vec":
|
||||
from data2vec_feature_reader import Data2vecFeatureReader
|
||||
reader = Data2vecFeatureReader(ckpt_path, layer, device=device, max_chunk=max_chunk)
|
||||
elif model_type == "whisper":
|
||||
from whisper_feature_reader import WhisperFeatureReader
|
||||
reader = WhisperFeatureReader(whisper_root, whisper_name, layer, device=device)
|
||||
|
||||
assert reader is not None
|
||||
|
||||
generator, num = get_path_iterator(tsv_path, nshard, rank)
|
||||
dump_feature(reader, generator, num, nshard, rank, feat_dir)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument(
|
||||
"--model_type",
|
||||
required=True,
|
||||
type=str,
|
||||
choices=["data2vec", "hubert", "whisper"],
|
||||
help="the type of the speech encoder."
|
||||
)
|
||||
parser.add_argument(
|
||||
"--tsv_path",
|
||||
required=True,
|
||||
type=str,
|
||||
help="the path to the tsv file."
|
||||
)
|
||||
parser.add_argument(
|
||||
"--ckpt_path",
|
||||
required=False,
|
||||
type=str,
|
||||
default=None,
|
||||
help="path to the speech model. must provide for HuBERT and data2vec"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--whisper_root",
|
||||
required=False,
|
||||
type=str,
|
||||
default=None,
|
||||
help="root dir to download/store whisper model. must provide for whisper model."
|
||||
)
|
||||
parser.add_argument(
|
||||
"--whisper_name",
|
||||
required=False,
|
||||
type=str,
|
||||
default=None,
|
||||
help="name of whisper model. e.g., large-v2. must provide for whisper model."
|
||||
)
|
||||
parser.add_argument(
|
||||
"--layer",
|
||||
required=True,
|
||||
type=int,
|
||||
help="which layer of the model. this is 1-based."
|
||||
)
|
||||
parser.add_argument(
|
||||
"--feat_dir",
|
||||
required=True,
|
||||
type=str,
|
||||
help="the output dir to save the representations."
|
||||
)
|
||||
parser.add_argument(
|
||||
"--nshard",
|
||||
required=False,
|
||||
type=int,
|
||||
default=1,
|
||||
help="total number of shards."
|
||||
)
|
||||
parser.add_argument(
|
||||
"--rank",
|
||||
required=False,
|
||||
type=int,
|
||||
default=0,
|
||||
help="shard id of this process."
|
||||
)
|
||||
parser.add_argument(
|
||||
"--max_chunk",
|
||||
type=int,
|
||||
default=1600000,
|
||||
help="max number of frames of each batch."
|
||||
)
|
||||
parser.add_argument(
|
||||
"--use_cpu",
|
||||
default=False,
|
||||
action="store_true",
|
||||
help="whether use cpu instead of gpu."
|
||||
)
|
||||
args = parser.parse_args()
|
||||
logger.info(args)
|
||||
|
||||
main(**vars(args))
|
||||
@@ -1,70 +0,0 @@
|
||||
# Copyright (c) ByteDance, Inc. and its affiliates.
|
||||
# Copyright (c) Chutong Meng
|
||||
#
|
||||
# This source code is licensed under the MIT license found in the
|
||||
# LICENSE file in the root directory of this source tree.
|
||||
# Based on fairseq (https://github.com/facebookresearch/fairseq)
|
||||
|
||||
# ref: https://github.com/facebookresearch/fairseq/blob/main/examples/hubert/simple_kmeans/feature_utils.py
|
||||
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
|
||||
import tqdm
|
||||
from npy_append_array import NpyAppendArray
|
||||
|
||||
|
||||
logging.basicConfig(
|
||||
format="%(asctime)s | %(levelname)s | %(name)s | %(message)s",
|
||||
datefmt="%Y-%m-%d %H:%M:%S",
|
||||
level=os.environ.get("LOGLEVEL", "INFO").upper(),
|
||||
stream=sys.stdout,
|
||||
)
|
||||
logger = logging.getLogger("feature_utils")
|
||||
|
||||
|
||||
def get_shard_range(tot, nshard, rank):
|
||||
assert rank < nshard and rank >= 0, f"invaid rank/nshard {rank}/{nshard}"
|
||||
start = round(tot / nshard * rank)
|
||||
end = round(tot / nshard * (rank + 1))
|
||||
assert start < end, f"start={start}, end={end}"
|
||||
logger.info(
|
||||
f"rank {rank} of {nshard}, process {end-start} "
|
||||
f"({start}-{end}) out of {tot}"
|
||||
)
|
||||
return start, end
|
||||
|
||||
|
||||
def get_path_iterator(tsv, nshard, rank):
|
||||
with open(tsv, "r") as f:
|
||||
root = f.readline().rstrip()
|
||||
lines = [line.rstrip() for line in f]
|
||||
start, end = get_shard_range(len(lines), nshard, rank)
|
||||
lines = lines[start:end]
|
||||
def iterate():
|
||||
for line in lines:
|
||||
subpath, nsample = line.split("\t")
|
||||
yield f"{root}/{subpath}", int(nsample)
|
||||
return iterate, len(lines)
|
||||
|
||||
|
||||
def dump_feature(reader, generator, num, nshard, rank, feat_dir):
|
||||
iterator = generator()
|
||||
|
||||
feat_path = f"{feat_dir}/{rank}_{nshard}.npy"
|
||||
leng_path = f"{feat_dir}/{rank}_{nshard}.len"
|
||||
|
||||
os.makedirs(feat_dir, exist_ok=True)
|
||||
if os.path.exists(feat_path):
|
||||
os.remove(feat_path)
|
||||
|
||||
feat_f = NpyAppendArray(feat_path)
|
||||
with open(leng_path, "w") as leng_f:
|
||||
for path, nsample in tqdm.tqdm(iterator, total=num):
|
||||
feat = reader.get_feats(path, nsample)
|
||||
feat_f.append(feat.cpu().numpy())
|
||||
leng_f.write(f"{len(feat)}\n")
|
||||
logger.info("finished successfully")
|
||||
|
||||
|
||||
@@ -1,64 +0,0 @@
|
||||
# Copyright (c) ByteDance, Inc. and its affiliates.
|
||||
# Copyright (c) Chutong Meng
|
||||
#
|
||||
# This source code is licensed under the MIT license found in the
|
||||
# LICENSE file in the root directory of this source tree.
|
||||
# Based on fairseq (https://github.com/facebookresearch/fairseq)
|
||||
|
||||
import logging
|
||||
|
||||
import fairseq
|
||||
import torch
|
||||
import torch.nn.functional as F
|
||||
|
||||
from fairseq.data.audio.audio_utils import get_features_or_waveform
|
||||
|
||||
logger = logging.getLogger("dump_feature")
|
||||
|
||||
|
||||
class HubertFeatureReader(object):
|
||||
def __init__(self, ckpt_path: str, layer: int, device: str, max_chunk=1600000):
|
||||
(
|
||||
model,
|
||||
cfg,
|
||||
task,
|
||||
) = fairseq.checkpoint_utils.load_model_ensemble_and_task([ckpt_path])
|
||||
|
||||
self.device = device
|
||||
logger.info(f"device = {self.device}")
|
||||
|
||||
self.model = model[0].eval().to(self.device)
|
||||
self.task = task
|
||||
self.layer = layer
|
||||
self.max_chunk = max_chunk
|
||||
logger.info(f"TASK CONFIG:\n{self.task.cfg}")
|
||||
logger.info(f" max_chunk = {self.max_chunk}")
|
||||
|
||||
def read_audio(self, path, ref_len=None):
|
||||
wav = get_features_or_waveform(path, need_waveform=True, use_sample_rate=self.task.cfg.sample_rate)
|
||||
if wav.ndim == 2:
|
||||
wav = wav.mean(-1)
|
||||
assert wav.ndim == 1, wav.ndim
|
||||
if ref_len is not None and abs(ref_len - len(wav)) > 160:
|
||||
logger.warning(f"ref {ref_len} != read {len(wav)} ({path})")
|
||||
return wav
|
||||
|
||||
def get_feats(self, path, ref_len=None):
|
||||
x = self.read_audio(path, ref_len=ref_len)
|
||||
with torch.no_grad():
|
||||
x = torch.from_numpy(x).float().to(self.device)
|
||||
if self.task.cfg.normalize:
|
||||
x = F.layer_norm(x, x.shape)
|
||||
x = x.view(1, -1)
|
||||
|
||||
feat = []
|
||||
for start in range(0, x.size(1), self.max_chunk):
|
||||
x_chunk = x[:, start: start + self.max_chunk]
|
||||
feat_chunk, _ = self.model.extract_features(
|
||||
source=x_chunk,
|
||||
padding_mask=None,
|
||||
mask=False,
|
||||
output_layer=self.layer,
|
||||
)
|
||||
feat.append(feat_chunk)
|
||||
return torch.cat(feat, 1).squeeze(0)
|
||||