4 Commits
Author SHA1 Message Date
smthemex 212634ab5b Merge pull request #35 from smthemex/patch-1
support node2
2026-09-16 17:02:14 +08:00
smthemex 761f73ebd9 support node2 2026-09-16 17:01:08 +08:00
smthemex c6fdeedf51 Merge pull request #34 from smthemex/Pr
init
2026-09-15 11:06:01 +08:00
smthemex 7fa27c6cbc init 2026-09-15 11:04:52 +08:00
241 changed files with 173120 additions and 30795 deletions
-2
View File
@@ -1,2 +0,0 @@
*.bin filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
+1 -166
View File
@@ -1,166 +1 @@
# Byte-compiled / optimized / DLL files
__pycache__/
*.py[cod]
*$py.class
# C extensions
*.so
# Distribution / packaging
.Python
build/
develop-eggs/
dist/
downloads/
eggs/
.eggs/
lib/
lib64/
parts/
sdist/
var/
wheels/
share/python-wheels/
*.egg-info/
.installed.cfg
*.egg
MANIFEST
# PyInstaller
# Usually these files are written by a python script from a template
# before PyInstaller builds the exe, so as to inject date/other infos into it.
*.manifest
*.spec
# Installer logs
pip-log.txt
pip-delete-this-directory.txt
# Unit test / coverage reports
htmlcov/
.tox/
.nox/
.coverage
.coverage.*
.cache
nosetests.xml
coverage.xml
*.cover
*.py,cover
.hypothesis/
.pytest_cache/
cover/
# Translations
*.mo
*.pot
# Django stuff:
*.log
local_settings.py
db.sqlite3
db.sqlite3-journal
# Flask stuff:
instance/
.webassets-cache
# Scrapy stuff:
.scrapy
# Sphinx documentation
docs/_build/
# PyBuilder
.pybuilder/
target/
# Jupyter Notebook
.ipynb_checkpoints
# IPython
profile_default/
ipython_config.py
# pyenv
# For a library or package, you might want to ignore these files since the code is
# intended to run in multiple environments; otherwise, check them in:
# .python-version
# pipenv
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
# However, in case of collaboration, if having platform-specific dependencies or dependencies
# having no cross-platform support, pipenv may install dependencies that don't work, or not
# install all needed dependencies.
#Pipfile.lock
# poetry
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
# This is especially recommended for binary packages to ensure reproducibility, and is more
# commonly ignored for libraries.
# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
#poetry.lock
# pdm
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
#pdm.lock
# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
# in version control.
# https://pdm.fming.dev/latest/usage/project/#working-with-version-control
.pdm.toml
.pdm-python
.pdm-build/
# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
__pypackages__/
# Celery stuff
celerybeat-schedule
celerybeat.pid
# SageMath parsed files
*.sage.py
# Environments
.env
.venv
env/
venv/
ENV/
env.bak/
venv.bak/
# Spyder project settings
.spyderproject
.spyproject
# Rope project settings
.ropeproject
# mkdocs documentation
/site
# mypy
.mypy_cache/
.dmypy.json
dmypy.json
# Pyre type checker
.pyre/
# pytype static type analyzer
.pytype/
# Cython debug symbols
cython_debug/
# PyCharm
# JetBrains specific template is maintained in a separate JetBrains.gitignore that can
# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
# and can be added to the global gitignore or merged into this file. For a more nuclear
# option (not recommended) you can uncomment the following to ignore the entire idea folder.
#.idea/
output
inference/xcodec_mini_infer
inference/run_infer.sh
__pycache__/
-276
View File
@@ -1,276 +0,0 @@
{
"last_node_id": 16,
"last_link_id": 23,
"nodes": [
{
"id": 9,
"type": "YUE_Stage_B_Sampler",
"pos": [
24645.052734375,
-4632.9755859375
],
"size": [
315,
150
],
"flags": {},
"order": 3,
"mode": 0,
"inputs": [
{
"name": "stage1_set",
"type": "STAGE_SET",
"link": 10
},
{
"name": "model",
"type": "MODEL_YUE_B",
"link": 23
}
],
"outputs": [
{
"name": "audio",
"type": "AUDIO",
"links": [
13
],
"slot_index": 0
},
{
"name": "string",
"type": "STRING",
"links": null
}
],
"properties": {
"Node name for S&R": "YUE_Stage_B_Sampler"
},
"widgets_values": [
"decoder_131000.pth",
"decoder_151000.pth",
4
]
},
{
"id": 8,
"type": "YUE_Stage_A_Sampler",
"pos": [
24111.734375,
-4703.33349609375
],
"size": [
449.79949951171875,
795.2402954101562
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL_YUE_A",
"link": 21
}
],
"outputs": [
{
"name": "stage1_set",
"type": "STAGE_SET",
"links": [
10
],
"slot_index": 0
},
{
"name": "info",
"type": "quantization_model",
"links": [
22
],
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "YUE_Stage_A_Sampler"
},
"widgets_values": [
"inspiring female uplifting pop airy vocal electronic bright vocal vocal.",
"[verse]\nStaring at the sunset, colors paint the sky.\nThoughts of you keep swirling, can't deny.\nI know I let you down, I made mistakes.\nBut I'm here to mend the heart I didn't break.\n\n[chorus]\nEvery road you take, I'll be one step behind.\nEvery dream you chase, I'm reaching for the light.\nYou can't fight this feeling now.\nI won't back down.\nYou know you can't deny it now.\n I won't back down \n\n[verse]\nThey might say I'm foolish, chasing after you.\nBut they don't feel this love the way we do.\nMy heart beats only for you, can't you see?\nI won't let you slip away from me. \n\n[chorus]\nEvery road you take, I'll be one step behind.\nEvery dream you chase, I'm reaching for the light.\nYou can't fight this feeling now.\nI won't back down.\nYou know you can't deny it now.\n I won't back down \n\n[bridge]\nNo, I won't back down, won't turn around.\nUntil you're back where you belong.\nI'll cross the oceans wide, stand by your side.\nTogether we are strong. \n\n[outro]\nEvery road you take, I'll be one step behind.\nEvery dream you chase, love's the tie that binds.\nYou can't fight this feeling now.\nI won't back down.",
1698056329,
"randomize",
2,
1.1,
0,
30,
3000,
true,
true,
true,
true
]
},
{
"id": 15,
"type": "YUE_Stage_A_Loader",
"pos": [
23737.6640625,
-4575.09814453125
],
"size": [
315,
202
],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "model",
"type": "MODEL_YUE_A",
"links": [
21
]
}
],
"properties": {
"Node name for S&R": "YUE_Stage_A_Loader"
},
"widgets_values": [
"F:/test/ComfyUI/models/diffusers/Alissonerdx/YuE-s1-7B-anneal-zh-icl-int8",
"ckpt_00360000.pth",
"int8",
false,
16384,
"FP16",
2
]
},
{
"id": 16,
"type": "YUE_Stage_B_Loader",
"pos": [
24641.8046875,
-4368.71630859375
],
"size": [
315,
154
],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [
{
"name": "info",
"type": "quantization_model",
"link": 22
}
],
"outputs": [
{
"name": "model",
"type": "MODEL_YUE_B",
"links": [
23
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "YUE_Stage_B_Loader"
},
"widgets_values": [
"F:/test/ComfyUI/models/diffusers/Alissonerdx/YuE-s2-1B-general-int8",
8192,
2,
"FP16",
false
]
},
{
"id": 5,
"type": "PreviewAudio",
"pos": [
25016.130859375,
-4446.78271484375
],
"size": [
315,
76
],
"flags": {},
"order": 4,
"mode": 0,
"inputs": [
{
"name": "audio",
"type": "AUDIO",
"link": 13
}
],
"outputs": [],
"properties": {
"Node name for S&R": "PreviewAudio"
},
"widgets_values": [
null
]
}
],
"links": [
[
10,
8,
0,
9,
0,
"STAGE_SET"
],
[
13,
9,
0,
5,
0,
"AUDIO"
],
[
21,
15,
0,
8,
0,
"MODEL_YUE_A"
],
[
22,
8,
1,
16,
0,
"quantization_model"
],
[
23,
16,
0,
9,
1,
"MODEL_YUE_B"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 0.8140274938684142,
"offset": [
-23442.787805256274,
4871.586176223436
]
}
},
"version": 0.4
}
+1
View File
@@ -1,3 +1,4 @@
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
+185
View File
@@ -0,0 +1,185 @@
---
license: cc-by-nc-4.0
library_name: transformers
pipeline_tag: feature-extraction
tags:
- audio
- music
- music-understanding
- representation-learning
- mert2
- custom_code
---
<h1 align="center">🤗 MERT-v2-FullSong</h1>
<p align="center"><strong>Music representations with full-song context</strong></p>
<p align="center">24 kHz mono · 632M parameters · 24 layers · 1,024 dimensions · 25 Hz</p>
<p align="center">
<a href="https://map-yue2.github.io/">🎵&nbsp;YuE2&nbsp;project</a>
·
<a href="#quick-start">🚀&nbsp;Quick&nbsp;start</a>
·
<a href="#marble">📊&nbsp;MARBLE</a>
·
<a href="#layer-guide">🎛️&nbsp;Layer&nbsp;guide</a>
·
<a href="#citation">📚&nbsp;Citation</a>
</p>
<p align="center">
<a href="https://huggingface.co/m-a-p/MERT-v2-30s"><img alt="MERT-v2-30s" src="https://img.shields.io/badge/MERT--v2--30s-374151?logo=huggingface&amp;logoColor=FFD21E" height="20" /></a>
&nbsp;
<a href="https://huggingface.co/m-a-p/MERT-v2-FullSong"><img alt="MERT-v2-FullSong" src="https://img.shields.io/badge/MERT--v2--FullSong-374151?logo=huggingface&amp;logoColor=FFD21E" height="20" /></a>
&nbsp;
<a href="https://huggingface.co/m-a-p/YuE2-3B"><img alt="YuE2-3B" src="https://img.shields.io/badge/YuE2--3B-374151?logo=huggingface&amp;logoColor=FFD21E" height="20" /></a>
&nbsp;
<a href="https://huggingface.co/m-a-p/YuE2-Vae"><img alt="🤗 YuE2-Vae" src="https://img.shields.io/badge/YuE2--Vae-374151?logo=huggingface&amp;logoColor=FFD21E" height="20" /></a>
&nbsp;
<a href="https://huggingface.co/m-a-p/YuE2-Vae-legacy"><img alt="🤗 YuE2-Vae-legacy" src="https://img.shields.io/badge/YuE2--Vae--legacy-374151?logo=huggingface&amp;logoColor=FFD21E" height="20" /></a>
&nbsp;
<a href="https://huggingface.co/datasets/m-a-p/WildSongBench"><img alt="🤗 WildSongBench" src="https://img.shields.io/badge/WildSongBench-374151?logo=huggingface&amp;logoColor=FFD21E" height="20" /></a>
&nbsp;
<a href="https://huggingface.co/m-a-p/SheetSage2"><img alt="SheetSage2" src="https://img.shields.io/badge/SheetSage2-374151?logo=huggingface&amp;logoColor=FFD21E" height="20" /></a>
</p>
**MERT-v2-FullSong is a bidirectional music encoder adapted to complete songs lasting 30–360 seconds.** Extract general-purpose music representations at the frame or recording level. Load with standard Hugging Face Transformers, using the familiar [MERT](https://huggingface.co/m-a-p/MERT-v1-330M) workflow.
It continues pretraining from [MERT-v2-30s](https://huggingface.co/m-a-p/MERT-v2-30s) and preserves the same feature interface.
![MERT2 architecture and training](assets/mert2-architecture.png)
*MERT-v2-30s uses the bidirectional backbone in (B); MERT-v2-FullSong continues through the full-song branch in (C). The causal branch is used for YuE2 tokenization.*
<a id="quick-start"></a>
## 🚀 Quick start
Install the matching PyTorch packages:
```bash
python -m pip install torch==2.6.0 torchaudio==2.6.0 transformers==4.53.2 huggingface-hub safetensors soundfile
```
```python
import soundfile as sf
import torch
import torchaudio.functional as AF
from transformers import AutoFeatureExtractor, AutoModel
repo = "m-a-p/MERT-v2-FullSong"
device = "cuda" if torch.cuda.is_available() else "cpu"
processor = AutoFeatureExtractor.from_pretrained(repo, trust_remote_code=True)
model = AutoModel.from_pretrained(repo, trust_remote_code=True).eval().to(device)
audio, sr = sf.read("music.wav", dtype="float32", always_2d=True)
audio = audio[:30 * sr].mean(axis=1) # First 30 seconds, mixed to mono.
waveform = AF.resample(torch.from_numpy(audio), sr, processor.sampling_rate)
inputs = processor(
waveform.numpy(), sampling_rate=processor.sampling_rate, return_tensors="pt",
).to(device)
with torch.inference_mode():
output = model(**inputs, output_hidden_states=True)
frames = output.last_hidden_state # [batch, frames, 1024], 25 Hz
layers = output.hidden_states # 24 tensors, one per block
mask = output.feature_attention_mask[..., None]
embedding = (frames * mask).sum(1) / mask.sum(1).clamp_min(1) # [batch, 1024]
```
`hidden_states[0]` is block 1; `hidden_states[23]` is block 24. Remove the audio slice to process a complete song.
<a id="marble"></a>
## 📊 MARBLE
Reported frozen-encoder results on [MARBLE](https://github.com/a43992899/MARBLE). All scores are multiplied by 100; higher is better. **Bold** marks the best displayed value in each column, including ties.
Baseline scores and parameter counts are reproduced from Tables III and V of [PupuJEPA](https://arxiv.org/html/2606.25713v2); the baselines were not rerun for this release. Evaluation protocols may differ across sources.
**General music understanding.** MTT = MagnaTagATune; key = GiantSteps refined key accuracy; genre and beat = GTZAN; valence and arousal = EmoMusic. ROC = ROC-AUC; AP = average precision.
| Model | Params | MTT ROC | MTT AP | Key acc. | Genre acc. | Beat F1 | Valence R² | Arousal R² |
|---|---:|---:|---:|---:|---:|---:|---:|---:|
| MERT-Large | 330M | 90.6 | 37.9 | 64.1 | 77.6 | 86.8 | 56.7 | 76.1 |
| Dasheng-1.2B | 1.2B | 91.5 | 40.4 | 58.0 | 81.4 | 87.7 | 57.4 | 75.0 |
| MuQ | 310M | 90.5 | 38.5 | 63.2 | 83.8 | 90.1 | 58.3 | 76.4 |
| MusicFM | 330M | 90.9 | 38.3 | 63.0 | 84.1 | 90.2 | 57.2 | 74.4 |
| AudioMAE++ | 307M | 91.2 | 39.5 | 61.7 | 80.3 | 90.0 | 59.0 | 75.7 |
| MATPAC++ | 307M | 90.6 | 38.2 | 63.7 | 81.4 | 90.1 | 57.8 | 74.7 |
| A-JEPA | 307M | 91.0 | 39.2 | 65.0 | 83.8 | 90.0 | 57.4 | 74.8 |
| PupuJEPA-Large | 307M | 91.7 | 40.8 | 66.1 | 86.9 | **91.0** | 62.5 | 76.8 |
| PupuJEPA-Huge | 632M | 91.3 | 39.7 | 64.8 | 85.9 | 90.5 | 62.0 | 78.5 |
| **MERT-v2-30s** | 632M | **91.91** | **41.29** | 66.97 | **91.72** | 90.59 | 63.23 | **80.01** |
| **MERT-v2-FullSong** | 632M | 91.74 | 41.20 | **67.05** | 90.69 | 90.57 | **63.52** | 78.14 |
**MTG-Jamendo tagging.** Mood denotes mood/theme tags.
| Model | Params | Instrument ROC | Instrument AP | Mood ROC | Mood AP | Genre ROC | Genre AP | Top-50 ROC | Top-50 AP |
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|
| MERT-Large | 330M | 75.5 | 18.8 | 75.3 | 13.5 | 86.1 | 18.0 | 82.6 | 29.1 |
| Dasheng-1.2B | 1.2B | 75.0 | 19.0 | 76.1 | 15.5 | 85.5 | 18.8 | 82.4 | 29.6 |
| MuQ | 310M | 74.8 | 19.1 | 73.7 | 13.2 | 85.4 | 19.1 | 83.0 | 30.2 |
| MusicFM | 330M | 74.6 | 18.5 | 74.9 | 14.1 | 85.3 | 19.4 | 81.9 | 29.7 |
| AudioMAE++ | 307M | 77.1 | 19.9 | 75.6 | 14.0 | 86.3 | 18.9 | 83.1 | 31.1 |
| MATPAC++ | 307M | 77.2 | 19.7 | 75.1 | 14.1 | 85.7 | 19.6 | 82.5 | 30.2 |
| A-JEPA | 307M | 76.6 | 19.3 | 74.6 | 14.3 | 85.5 | 19.2 | 82.5 | 29.6 |
| PupuJEPA-Large | 307M | 78.4 | 21.2 | 76.2 | 15.3 | 86.1 | 20.1 | 82.8 | 30.5 |
| PupuJEPA-Huge | 632M | 77.6 | 20.5 | 75.9 | 14.7 | 85.9 | 20.1 | 83.1 | 30.7 |
| **MERT-v2-30s** | 632M | **80.27** | 22.89 | **79.44** | **16.68** | **88.01** | **21.22** | **84.18** | **32.17** |
| **MERT-v2-FullSong** | 632M | **80.27** | **23.51** | 78.74 | 15.74 | 87.98 | 20.66 | 84.13 | 31.62 |
**Supplemental:** Chords1217 frame accuracy is **78.48** for MERT-v2-30s and **77.79** for MERT-v2-FullSong.
<!-- mert2-hf-validation:start -->
**HF reproduction verified:** Both models reproduce all ten MARBLE tasks with the fixed evaluation settings. [Scores and best settings](marble_results.json).
<!-- mert2-hf-validation:end -->
<a id="layer-guide"></a>
## 🎛️ Layer guide
Recommended probe settings with the encoder frozen. L1 = `hidden_states[0]`; All-layer MLP uses all 24 layers.
| Task / dataset | MERT-v2-30s layer | Probe LR | MERT-v2-FullSong layer | Probe LR |
|---|---|---:|---|---:|
| Genre · GTZAN | L23 | 5e-3 | L24 | 5e-4 |
| Beat · GTZAN | L21 | 1e-3 | L23 | 1e-3 |
| Key · GiantSteps | L4 | 1e-3 | L23 | 1e-3 |
| Emotion · EmoMusic | All-layer MLP | 5e-5 | L24 | 5e-4 |
| Chords · Chords1217 | All-layer MLP | 1e-4 | All-layer MLP | 5e-4 |
| Tagging · MagnaTagATune | L22 | 1e-3 | L23 | 1e-3 |
| Instrument · MTG-Jamendo | L14 | 1e-3 | L12 | 1e-3 |
| Mood/theme · MTG-Jamendo | L16 | 1e-3 | L13 | 1e-3 |
| Genre · MTG-Jamendo | L19 | 1e-3 | L16 | 1e-3 |
| Top-50 · MTG-Jamendo | L13 | 1e-3 | L22 | 1e-3 |
<a id="citation"></a>
## 📚 Citation
**Technical report coming soon.** For now, please cite [MERT (ICLR 2024)](https://proceedings.iclr.cc/paper_files/paper/2024/hash/33dffa2e3d2ab74a783d1a8c292f66d9-Abstract-Conference.html) and [MARBLE (NeurIPS 2023)](https://proceedings.neurips.cc/paper_files/paper/2023/hash/7cbeec46f979618beafb4f46d8f39f36-Abstract-Datasets_and_Benchmarks.html) when using MERT-v2 or its benchmark results in your research.
```bibtex
@inproceedings{li2024mert,
title = {MERT: Acoustic Music Understanding Model with Large-Scale Self-supervised Training},
author = {Li, Yizhi and Yuan, Ruibin and Zhang, Ge and Ma, Yinghao and Chen, Xingran and Yin, Hanzhi and Xiao, Chenghao and Lin, Chenghua and Ragni, Anton and Benetos, Emmanouil and Gyenge, Norbert and Dannenberg, Roger and Liu, Ruibo and Chen, Wenhu and Xia, Gus and Shi, Yemin and Huang, Wenhao and Wang, Zili and Guo, Yike and Fu, Jie},
booktitle = {International Conference on Learning Representations},
year = {2024},
url = {https://proceedings.iclr.cc/paper_files/paper/2024/hash/33dffa2e3d2ab74a783d1a8c292f66d9-Abstract-Conference.html}
}
@inproceedings{yuan2023marble,
title = {MARBLE: Music Audio Representation Benchmark for Universal Evaluation},
author = {Yuan, Ruibin and Ma, Yinghao and Li, Yizhi and Zhang, Ge and Chen, Xingran and Yin, Hanzhi and Zhuo, Le and Liu, Yiqi and Huang, Jiawen and Tian, Zeyue and Deng, Binyue and Wang, Ningzhi and Lin, Chenghua and Benetos, Emmanouil and Ragni, Anton and Gyenge, Norbert and Dannenberg, Roger and Chen, Wenhu and Xia, Gus and Xue, Wei and Liu, Si and Wang, Shi and Liu, Ruibo and Guo, Yike and Fu, Jie},
booktitle = {Advances in Neural Information Processing Systems},
volume = {36},
pages = {39626--39647},
year = {2023},
doi = {10.52202/075280-1722},
url = {https://proceedings.neurips.cc/paper_files/paper/2023/hash/7cbeec46f979618beafb4f46d8f39f36-Abstract-Datasets_and_Benchmarks.html}
}
```
Weights: [CC BY-NC 4.0](LICENSE). [Dependency notices](THIRD_PARTY_NOTICES.md).
**YuE2 family:** [Song generation](https://huggingface.co/m-a-p/YuE2-3B) · [Audio decoder](https://huggingface.co/m-a-p/YuE2-Vae) · [Benchmark decoder](https://huggingface.co/m-a-p/YuE2-Vae-legacy).
+11
View File
@@ -0,0 +1,11 @@
# Third-party notices
The MERT2 inference implementation uses separately installed PyTorch,
torchaudio, Hugging Face Transformers, huggingface_hub, and safetensors.
These dependencies retain their respective upstream licenses and notices;
this model repository does not redistribute their source distributions.
The example additionally uses SoundFile, which retains its upstream license.
The MERT2 checkpoint weights are licensed separately under CC BY-NC 4.0;
see [LICENSE](LICENSE). This weight license does not relicense dependencies.
+41
View File
@@ -0,0 +1,41 @@
{
"architectures": [
"MERT2Model"
],
"auto_map": {
"AutoConfig": "configuration_mert2.MERT2Config",
"AutoModel": "modeling_mert2.MERT2Model"
},
"context_seconds": 360.0,
"conv_depthwise_kernel_size": 31,
"frame_rate": 25.0,
"hidden_size": 1024,
"hop_length": 240,
"initializer_range": 0.02,
"inputs_to_logits_ratio": 960,
"intermediate_size": 4096,
"layer_norm_eps": 1e-05,
"minimum_input_samples": 1025,
"model_type": "mert2",
"n_fft": 2048,
"num_attention_heads": 16,
"num_hidden_layers": 24,
"num_mel_bins": 128,
"rotary_embedding_base": 10000,
"sampling_rate": 24000,
"subsampling_channels": [
128,
512,
1024
],
"subsampling_depths": [
3,
4,
5
],
"subsampling_layer_norm_eps": 1e-06,
"torch_dtype": "float32",
"transformers_version": "4.53.2",
"variant": "fs",
"win_length": 2048
}
+76
View File
@@ -0,0 +1,76 @@
"""Configuration for MERT2 music representation models."""
from transformers import PretrainedConfig
class MERT2Config(PretrainedConfig):
"""Architecture shared by the 30-second and full-song MERT2 encoders."""
model_type = "mert2"
def __init__(
self,
hidden_size=1024,
intermediate_size=4096,
num_hidden_layers=24,
num_attention_heads=16,
num_mel_bins=128,
sampling_rate=24000,
n_fft=2048,
win_length=2048,
hop_length=240,
subsampling_channels=None,
subsampling_depths=None,
conv_depthwise_kernel_size=31,
rotary_embedding_base=10000,
layer_norm_eps=1e-5,
subsampling_layer_norm_eps=1e-6,
initializer_range=0.02,
variant="30s",
context_seconds=None,
**kwargs,
):
super().__init__(**kwargs)
self.hidden_size = int(hidden_size)
self.intermediate_size = int(intermediate_size)
self.num_hidden_layers = int(num_hidden_layers)
self.num_attention_heads = int(num_attention_heads)
self.num_mel_bins = int(num_mel_bins)
self.sampling_rate = int(sampling_rate)
self.n_fft = int(n_fft)
self.win_length = int(win_length)
self.hop_length = int(hop_length)
self.subsampling_channels = list(subsampling_channels or [num_mel_bins, 512, hidden_size])
self.subsampling_depths = list(subsampling_depths or [3, 4, 5])
self.conv_depthwise_kernel_size = int(conv_depthwise_kernel_size)
self.rotary_embedding_base = int(rotary_embedding_base)
self.layer_norm_eps = float(layer_norm_eps)
self.subsampling_layer_norm_eps = float(subsampling_layer_norm_eps)
self.initializer_range = float(initializer_range)
self.variant = str(variant)
self.context_seconds = float(context_seconds if context_seconds is not None else (360 if variant == "fs" else 30))
self.inputs_to_logits_ratio = self.hop_length * 4
self.frame_rate = self.sampling_rate / self.inputs_to_logits_ratio
self.minimum_input_samples = self.n_fft // 2 + 1
if min(self.hidden_size, self.intermediate_size, self.num_hidden_layers, self.num_attention_heads) <= 0:
raise ValueError("Encoder dimensions and layer counts must be positive.")
if self.hidden_size % self.num_attention_heads or (self.hidden_size // self.num_attention_heads) % 2:
raise ValueError("hidden_size must divide into an even head dimension.")
if len(self.subsampling_channels) != 3 or len(self.subsampling_depths) != 3:
raise ValueError("The subsampler must contain three channel widths and three depths.")
if self.subsampling_channels[0] != self.num_mel_bins or self.subsampling_channels[-1] != self.hidden_size:
raise ValueError("Subsampling widths must start at num_mel_bins and end at hidden_size.")
if min(*self.subsampling_channels, *self.subsampling_depths, self.num_mel_bins, self.sampling_rate, self.hop_length) <= 0:
raise ValueError("Frontend dimensions and sampling parameters must be positive.")
if self.n_fft < self.win_length or self.win_length <= 0 or self.n_fft % 2:
raise ValueError("n_fft must be even and at least win_length > 0.")
if self.conv_depthwise_kernel_size <= 0 or self.conv_depthwise_kernel_size % 2 != 1:
raise ValueError("conv_depthwise_kernel_size must be positive and odd.")
if self.rotary_embedding_base <= 0 or min(self.layer_norm_eps, self.subsampling_layer_norm_eps) <= 0:
raise ValueError("Rotary base and normalization epsilons must be positive.")
if self.variant not in {"30s", "fs"}:
raise ValueError("variant must be '30s' or 'fs'.")
MERT2Config.register_for_auto_class()
+327
View File
@@ -0,0 +1,327 @@
{
"schema_version": 2,
"benchmark": "MARBLE",
"metric_scale": 1,
"display_scale": 100,
"models": {
"MERT-v2-30s": {
"repo_id": "a43992899/MERT-v2-30s",
"tasks": [
{
"dataset": "GTZAN",
"task": "Genre classification",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 23,
"hidden_state_index": 22
},
"learning_rate": 0.005,
"probe_seed": 1234,
"metrics": {
"file_acc": 0.9172413945198059
}
},
{
"dataset": "GTZAN",
"task": "Beat tracking",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 21,
"hidden_state_index": 20
},
"learning_rate": 0.001,
"probe_seed": 1234,
"metrics": {
"beat_f1": 0.9058816432952881
}
},
{
"dataset": "GiantSteps",
"task": "Key detection",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 4,
"hidden_state_index": 3
},
"learning_rate": 0.001,
"probe_seed": 1234,
"metrics": {
"weighted_score": 0.6697019867549668
}
},
{
"dataset": "EmoMusic",
"task": "Emotion recognition",
"supplemental": false,
"representation": {
"mode": "learned_all_layer_mlp",
"num_layers": 24
},
"learning_rate": 5e-05,
"probe_seed": 1234,
"metrics": {
"valence_r2": 0.6322920322418213,
"arousal_r2": 0.8000704646110535
}
},
{
"dataset": "Chords1217",
"task": "Chord recognition",
"supplemental": true,
"representation": {
"mode": "learned_all_layer_mlp",
"num_layers": 24
},
"learning_rate": 0.0001,
"probe_seed": 1234,
"metrics": {
"acc": 0.7847631573677063
}
},
{
"dataset": "MagnaTagATune",
"task": "Tagging",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 22,
"hidden_state_index": 21
},
"learning_rate": 0.001,
"probe_seed": 1234,
"metrics": {
"auroc": 0.9190762042999268,
"average_precision": 0.4129374325275421
}
},
{
"dataset": "MTG-Jamendo",
"task": "Genre tagging",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 19,
"hidden_state_index": 18
},
"learning_rate": 0.001,
"probe_seed": 1234,
"metrics": {
"auroc": 0.8801141381263733,
"average_precision": 0.21221663057804108
}
},
{
"dataset": "MTG-Jamendo",
"task": "Instrument tagging",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 14,
"hidden_state_index": 13
},
"learning_rate": 0.001,
"probe_seed": 1234,
"metrics": {
"auroc": 0.8027445673942566,
"average_precision": 0.2288619726896286
}
},
{
"dataset": "MTG-Jamendo",
"task": "Mood/theme tagging",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 16,
"hidden_state_index": 15
},
"learning_rate": 0.001,
"probe_seed": 1234,
"metrics": {
"auroc": 0.7944093942642212,
"average_precision": 0.16676680743694305
}
},
{
"dataset": "MTG-Jamendo",
"task": "Top-50 tagging",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 13,
"hidden_state_index": 12
},
"learning_rate": 0.001,
"probe_seed": 1234,
"metrics": {
"auroc": 0.8417603969573975,
"average_precision": 0.3217017948627472
}
}
]
},
"MERT-v2-FullSong": {
"repo_id": "a43992899/MERT-v2-FullSong",
"tasks": [
{
"dataset": "GTZAN",
"task": "Genre classification",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 24,
"hidden_state_index": 23
},
"learning_rate": 0.0005,
"probe_seed": 1234,
"metrics": {
"file_acc": 0.9068965315818787
}
},
{
"dataset": "GTZAN",
"task": "Beat tracking",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 23,
"hidden_state_index": 22
},
"learning_rate": 0.001,
"probe_seed": 1234,
"metrics": {
"beat_f1": 0.9057297706604004
}
},
{
"dataset": "GiantSteps",
"task": "Key detection",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 23,
"hidden_state_index": 22
},
"learning_rate": 0.001,
"probe_seed": 1234,
"metrics": {
"weighted_score": 0.6705298013245033
}
},
{
"dataset": "EmoMusic",
"task": "Emotion recognition",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 24,
"hidden_state_index": 23
},
"learning_rate": 0.0005,
"probe_seed": 1234,
"metrics": {
"valence_r2": 0.6351668834686279,
"arousal_r2": 0.7814431190490723
}
},
{
"dataset": "Chords1217",
"task": "Chord recognition",
"supplemental": true,
"representation": {
"mode": "learned_all_layer_mlp",
"num_layers": 24
},
"learning_rate": 0.0005,
"probe_seed": 1234,
"metrics": {
"acc": 0.7779459357261658
}
},
{
"dataset": "MagnaTagATune",
"task": "Tagging",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 23,
"hidden_state_index": 22
},
"learning_rate": 0.001,
"probe_seed": 1234,
"metrics": {
"auroc": 0.9173608422279358,
"average_precision": 0.4120141565799713
}
},
{
"dataset": "MTG-Jamendo",
"task": "Genre tagging",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 16,
"hidden_state_index": 15
},
"learning_rate": 0.001,
"probe_seed": 1234,
"metrics": {
"auroc": 0.8797574043273926,
"average_precision": 0.20656391978263855
}
},
{
"dataset": "MTG-Jamendo",
"task": "Instrument tagging",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 12,
"hidden_state_index": 11
},
"learning_rate": 0.001,
"probe_seed": 1234,
"metrics": {
"auroc": 0.8026924133300781,
"average_precision": 0.2351299524307251
}
},
{
"dataset": "MTG-Jamendo",
"task": "Mood/theme tagging",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 13,
"hidden_state_index": 12
},
"learning_rate": 0.001,
"probe_seed": 1234,
"metrics": {
"auroc": 0.7874246835708618,
"average_precision": 0.15743640065193176
}
},
{
"dataset": "MTG-Jamendo",
"task": "Top-50 tagging",
"supplemental": false,
"representation": {
"mode": "fixed_layer",
"layer": 22,
"hidden_state_index": 21
},
"learning_rate": 0.001,
"probe_seed": 1234,
"metrics": {
"auroc": 0.8413298726081848,
"average_precision": 0.3162302076816559
}
}
]
}
}
}
+361
View File
@@ -0,0 +1,361 @@
"""Standalone MERT2 waveform-to-representation inference."""
from dataclasses import dataclass
from typing import Optional, Tuple
import torch
from torch import nn
from torch.nn import functional as F
from torchaudio.transforms import AmplitudeToDB, MelScale, Spectrogram
from transformers import PreTrainedModel
from transformers.utils import ModelOutput
from .configuration_mert2 import MERT2Config
@dataclass
class MERT2ModelOutput(ModelOutput):
"""Frame representations and their validity mask.
``hidden_states`` contains one tensor per Conformer block, without an input
embedding. ``feature_attention_mask`` is True at valid output frames.
"""
last_hidden_state: Optional[torch.FloatTensor] = None
hidden_states: Optional[Tuple[torch.FloatTensor, ...]] = None
feature_attention_mask: Optional[torch.BoolTensor] = None
class MERT2MelFrontend(nn.Module):
"""Power log-mel features with fixed, checkpoint-specific normalization."""
def __init__(self, config):
super().__init__()
# Filter construction requires real tensors even during meta loading.
with torch.device("cpu"):
self.register_buffer("mel_mean", torch.zeros(config.num_mel_bins, dtype=torch.float32))
self.register_buffer("mel_std", torch.ones(config.num_mel_bins, dtype=torch.float32))
self.spectrogram = Spectrogram(
n_fft=config.n_fft,
win_length=config.win_length,
hop_length=config.hop_length,
power=2.0,
).float()
self.mel_scale = MelScale(
n_mels=config.num_mel_bins,
sample_rate=config.sampling_rate,
n_stft=config.n_fft // 2 + 1,
).float()
self.amplitude_to_db = AmplitudeToDB(stype="power", top_db=None)
def _apply(self, fn, recurse=True):
# Preserve the original float32 values, including on model.half().
saved = [(module, name, value) for module in self.modules() for name, value in module._buffers.items() if value is not None]
super()._apply(fn, recurse=recurse)
for module, name, original in saved:
current = module._buffers[name]
if original.is_floating_point():
module._buffers[name] = (
current.float() if original.is_meta else original.to(device=current.device, dtype=torch.float32)
)
return self
@torch.no_grad()
def forward(self, waveform):
with torch.autocast(device_type=waveform.device.type, enabled=False):
spectrum = self.spectrogram(waveform.float())
mel = self.amplitude_to_db(self.mel_scale(spectrum))
mel = mel[..., :-1].transpose(-1, -2)
return (mel - self.mel_mean) / self.mel_std.clamp_min(1e-5)
class Transpose(nn.Module):
def forward(self, hidden_states):
return hidden_states.transpose(1, 2)
class GlobalResponseNorm(nn.Module):
def __init__(self, dim):
super().__init__()
self.weight = nn.Parameter(torch.zeros(1, 1, dim))
self.bias = nn.Parameter(torch.zeros(1, 1, dim))
def forward(self, hidden_states):
magnitude = torch.norm(hidden_states, p=2, dim=1, keepdim=True)
normalized = magnitude / (magnitude.mean(dim=-1, keepdim=True) + 1e-6)
return self.weight * (hidden_states * normalized) + self.bias + hidden_states
class ConvNextLayer(nn.Module):
def __init__(self, dim, eps):
super().__init__()
self.depthwise_block = nn.Sequential(
Transpose(), nn.Conv1d(dim, dim, 7, padding=3, groups=dim), Transpose()
)
self.pointwise_block = nn.Sequential(
nn.LayerNorm(dim, eps=eps),
nn.Linear(dim, 4 * dim),
nn.GELU(),
GlobalResponseNorm(4 * dim),
nn.Linear(4 * dim, dim),
)
def forward(self, hidden_states):
return hidden_states + self.pointwise_block(self.depthwise_block(hidden_states))
class ConvNextBlock(nn.Module):
def __init__(self, in_channels, out_channels, stride, depth, eps):
super().__init__()
self.resampling_layer = (
nn.Sequential(
nn.LayerNorm(in_channels, eps=eps),
Transpose(),
nn.Conv1d(in_channels, out_channels, 2, stride=stride),
Transpose(),
)
if in_channels != out_channels or stride > 1 else nn.Identity()
)
self.convnext_layers = nn.Sequential(*[ConvNextLayer(out_channels, eps) for _ in range(depth)])
def forward(self, hidden_states):
return self.convnext_layers(self.resampling_layer(hidden_states))
class RotaryEmbedding(nn.Module):
def __init__(self, config):
super().__init__()
self.head_dim = config.hidden_size // config.num_attention_heads
self.base = config.rotary_embedding_base
with torch.device("cpu"):
inverse_frequency = 1.0 / (
self.base ** (torch.arange(0, self.head_dim, 2, dtype=torch.float32) / self.head_dim)
)
self.register_buffer("inv_freq", inverse_frequency, persistent=False)
self._sequence_length = 0
self._cache_device = None
self._cos = None
self._sin = None
def _apply(self, fn, recurse=True):
original = self.inv_freq
super()._apply(fn, recurse=recurse)
self.inv_freq = original.to(device=self.inv_freq.device, dtype=torch.float32)
return self
def forward(self, hidden_states):
length = hidden_states.shape[1]
if self._cos is None or self._cache_device != hidden_states.device or length > self._sequence_length:
positions = torch.arange(length, device=hidden_states.device, dtype=self.inv_freq.dtype)
frequencies = torch.einsum("i,j->ij", positions, self.inv_freq)
angles = torch.cat((frequencies, frequencies), dim=-1)
self._cos = angles.cos()[:, None, None, :]
self._sin = angles.sin()[:, None, None, :]
self._sequence_length = length
self._cache_device = hidden_states.device
return (
self._cos[:length].to(hidden_states.dtype).permute(1, 0, 2, 3),
self._sin[:length].to(hidden_states.dtype).permute(1, 0, 2, 3),
)
def rotate_half(value):
first, second = value.chunk(2, dim=-1)
return torch.cat((-second, first), dim=-1)
class SelfAttention(nn.Module):
def __init__(self, config):
super().__init__()
self.config = config
self.num_heads = config.num_attention_heads
self.head_dim = config.hidden_size // self.num_heads
self.query_proj = nn.Linear(config.hidden_size, config.hidden_size)
self.key_proj = nn.Linear(config.hidden_size, config.hidden_size)
self.value_proj = nn.Linear(config.hidden_size, config.hidden_size)
self.out_proj = nn.Linear(config.hidden_size, config.hidden_size)
def forward(self, hidden_states, position_embeddings):
batch, time, width = hidden_states.shape
shape = (batch, time, self.num_heads, self.head_dim)
query = self.query_proj(hidden_states).reshape(shape)
key = self.key_proj(hidden_states).reshape(shape)
value = self.value_proj(hidden_states).reshape(shape)
cos, sin = position_embeddings
query = query * cos + rotate_half(query) * sin
key = key * cos + rotate_half(key) * sin
if self.config._attn_implementation == "flash_attention_2":
if query.device.type != "cuda" or query.dtype not in (torch.float16, torch.bfloat16):
raise ValueError("flash_attention_2 requires CUDA and float16/bfloat16 activations; use autocast or load a reduced-precision model.")
try:
from flash_attn import flash_attn_func
except ImportError as error:
raise ImportError("Install flash-attn to use flash_attention_2, or select attn_implementation='sdpa'.") from error
attended = flash_attn_func(query.contiguous(), key.contiguous(), value.contiguous(), dropout_p=0.0, causal=False)
else:
attended = F.scaled_dot_product_attention(
query.transpose(1, 2), key.transpose(1, 2), value.transpose(1, 2), dropout_p=0.0, is_causal=False
).transpose(1, 2)
return self.out_proj(attended.reshape(batch, time, width))
class FeedForward(nn.Module):
def __init__(self, config):
super().__init__()
self.w_1 = nn.Linear(config.hidden_size, config.intermediate_size)
self.w_2 = nn.Linear(config.intermediate_size, config.hidden_size)
def forward(self, hidden_states):
return self.w_2(F.gelu(self.w_1(hidden_states)))
class ConvolutionModule(nn.Module):
def __init__(self, config):
super().__init__()
width = config.hidden_size
kernel = config.conv_depthwise_kernel_size
self.layer_norm = nn.LayerNorm(width, eps=config.layer_norm_eps)
self.conv_block = nn.Sequential(
Transpose(),
nn.Conv1d(width, 2 * width, 1, bias=False),
nn.GLU(dim=1),
nn.Conv1d(width, width, kernel, padding=(kernel - 1) // 2, groups=width, bias=False),
nn.Sequential(Transpose(), nn.LayerNorm(width, eps=config.layer_norm_eps), Transpose()),
nn.GELU(),
nn.Conv1d(width, width, 1, bias=False),
Transpose(),
)
def forward(self, hidden_states):
return self.conv_block(self.layer_norm(hidden_states))
class ConformerBlock(nn.Module):
def __init__(self, config):
super().__init__()
width, eps = config.hidden_size, config.layer_norm_eps
self.ffn1_layer_norm = nn.LayerNorm(width, eps=eps)
self.ffn1 = FeedForward(config)
self.attn_layer_norm = nn.LayerNorm(width, eps=eps)
self.attn = SelfAttention(config)
self.conv_module = ConvolutionModule(config)
self.ffn2_layer_norm = nn.LayerNorm(width, eps=eps)
self.ffn2 = FeedForward(config)
self.final_layer_norm = nn.LayerNorm(width, eps=eps)
def forward(self, hidden_states, position_embeddings):
hidden_states = hidden_states + 0.5 * self.ffn1(self.ffn1_layer_norm(hidden_states))
hidden_states = self.attn(self.attn_layer_norm(hidden_states), position_embeddings) + hidden_states
hidden_states = self.conv_module(hidden_states) + hidden_states
hidden_states = hidden_states + 0.5 * self.ffn2(self.ffn2_layer_norm(hidden_states))
return self.final_layer_norm(hidden_states)
class MERT2Model(PreTrainedModel):
"""Encode mono waveforms into 25 Hz MERT2 frame representations.
Inputs are floating-point mono waveforms sampled at ``config.sampling_rate``.
Do not standardize their amplitude. An optional waveform attention mask must
contain a prefix of ones followed by zeros. Without a mask, every input
sample, including any caller-provided silence or padding, is processed.
"""
config_class = MERT2Config
base_model_prefix = ""
main_input_name = "input_values"
_supports_sdpa = True
_supports_flash_attn_2 = True
_no_split_modules = ["ConvNextBlock", "ConformerBlock"]
def __init__(self, config):
super().__init__(config)
if config._attn_implementation not in {"sdpa", "flash_attention_2"}:
raise ValueError("MERT2 supports attn_implementation='sdpa' or 'flash_attention_2'.")
self.feature_extractor = MERT2MelFrontend(config)
channels = [config.num_mel_bins] + config.subsampling_channels
self.subsampling_module = nn.Sequential(*[
ConvNextBlock(channels[i], channels[i + 1], (1, 2, 2)[i], config.subsampling_depths[i], config.subsampling_layer_norm_eps)
for i in range(3)
])
self.layers = nn.ModuleList([ConformerBlock(config) for _ in range(config.num_hidden_layers)])
self.embed_positions = RotaryEmbedding(config)
self.post_init()
def _init_weights(self, module):
if isinstance(module, (nn.Linear, nn.Conv1d)):
nn.init.normal_(module.weight, mean=0.0, std=self.config.initializer_range)
if module.bias is not None:
nn.init.zeros_(module.bias)
elif isinstance(module, nn.LayerNorm):
nn.init.ones_(module.weight)
nn.init.zeros_(module.bias)
def _get_feat_extract_output_lengths(self, input_lengths):
return input_lengths // self.config.inputs_to_logits_ratio
def _encode(self, input_values, output_hidden_states):
mel = self.feature_extractor(input_values)
input_dtype = self.subsampling_module[0].convnext_layers[0].depthwise_block[1].weight.dtype
hidden = self.subsampling_module(mel.to(dtype=input_dtype))
positions = self.embed_positions(hidden)
states = [] if output_hidden_states else None
for layer in self.layers:
hidden = layer(hidden, positions)
if states is not None:
states.append(hidden)
return hidden, tuple(states) if states is not None else None
def forward(self, input_values, attention_mask=None, output_hidden_states=None, return_dict=None):
"""Return final frames, optionally all block states, and a frame mask.
Different valid waveform lengths are encoded separately so convolution
and global normalization never incorporate another sample's padding.
Outputs are zero-padded to the longest valid feature sequence.
"""
output_hidden_states = self.config.output_hidden_states if output_hidden_states is None else output_hidden_states
return_dict = self.config.use_return_dict if return_dict is None else return_dict
if input_values.ndim != 2 or not input_values.is_floating_point() or input_values.shape[0] == 0:
raise ValueError("input_values must be a nonempty floating-point tensor of shape [batch, samples].")
if input_values.shape[1] < self.config.minimum_input_samples:
raise ValueError(f"Each waveform must contain at least {self.config.minimum_input_samples} samples.")
if not torch.isfinite(input_values).all():
raise ValueError("input_values must contain only finite samples.")
batch, samples = input_values.shape
if attention_mask is None:
hidden, states = self._encode(input_values, output_hidden_states)
feature_mask = torch.ones(hidden.shape[:2], dtype=torch.bool, device=hidden.device)
else:
if attention_mask.shape != input_values.shape:
raise ValueError("attention_mask must have the same shape as input_values.")
attention_mask = attention_mask.to(device=input_values.device)
if not ((attention_mask == 0) | (attention_mask == 1)).all():
raise ValueError("attention_mask must contain only zeros and ones.")
mask = attention_mask.bool()
lengths = mask.sum(dim=1)
expected = torch.arange(samples, device=input_values.device)[None, :] < lengths[:, None]
if not torch.equal(mask, expected):
raise ValueError("attention_mask must be right padded: valid samples followed by padding.")
if (lengths < self.config.minimum_input_samples).any():
raise ValueError(f"Each waveform must contain at least {self.config.minimum_input_samples} valid samples.")
feature_lengths = self._get_feat_extract_output_lengths(lengths)
maximum = int(feature_lengths.max().item())
feature_mask = torch.arange(maximum, device=input_values.device)[None, :] < feature_lengths[:, None]
hidden, state_values = None, None
for length in torch.unique(lengths, sorted=True).tolist():
indices = torch.where(lengths == length)[0]
group_hidden, group_states = self._encode(input_values.index_select(0, indices)[:, :length], output_hidden_states)
padding = (0, 0, 0, maximum - group_hidden.shape[1])
if hidden is None:
hidden = group_hidden.new_zeros(batch, maximum, self.config.hidden_size)
if output_hidden_states:
state_values = [torch.zeros_like(hidden) for _ in self.layers]
hidden = hidden.index_copy(0, indices, F.pad(group_hidden, padding))
if state_values is not None:
for i, value in enumerate(group_states):
state_values[i] = state_values[i].index_copy(0, indices, F.pad(value, padding))
states = tuple(state_values) if state_values is not None else None
output = MERT2ModelOutput(last_hidden_state=hidden, hidden_states=states, feature_attention_mask=feature_mask)
return output if return_dict else output.to_tuple()
MERT2Model.register_for_auto_class("AutoModel")
@@ -0,0 +1,9 @@
{
"feature_extractor_type": "Wav2Vec2FeatureExtractor",
"feature_size": 1,
"sampling_rate": 24000,
"padding_value": 0.0,
"padding_side": "right",
"return_attention_mask": true,
"do_normalize": false
}
+9
View File
@@ -0,0 +1,9 @@
{
"format": "safetensors",
"dtype": "float32",
"parameters": 632295808,
"state_tensors": 876,
"filename": "model.safetensors",
"sha256": "e6dd2ab187d6dd62b6521cd7d8f932e237acf0c5757745a7232082e28391350d",
"bytes": 2529812848
}
+36 -94
View File
@@ -2,9 +2,7 @@
[YuE](https://github.com/multimodal-art-projection/YuE) is a groundbreaking series of open-source foundation models designed for music generation, specifically for transforming lyrics into full songs (lyrics2song). you can use it in comfyUI
# Update
* install quantization moudle (mmgp and exllamav2) if you need ;
* mmgp and exllamav2 的量化库在菜单选中的时候才会需要,避免有些人安装不上;
* update to yue2 / 简单更新到yue2版本;
# 1. Installation
@@ -18,106 +16,50 @@ git clone https://github.com/smthemex/ComfyUI_YuE.git
```
pip install -r requirements.txt
```
* triton is not necessary,I don't test it,you can try
* triton库不是必须的,不过节点里的加速方法用了或许更快,不保真
* use -no-deps if descript-audiotools need high torch 如果descript-audiotools安装需要高版本的torch ,使用pip install -no-deps descript-audiotools
* if use mmgp
```
pip install mmgp
```
* if use exllamav2
```
pip install exllamav2
```
- or
```
git clone https://github.com/turboderp/exllamav2
cd exllamav2
pip install -r requirements.txt
pip install .
```
# 3.models
* 3.1 download from [here](https://huggingface.co/m-a-p/xcodec_mini_infer/tree/main/final_ckpt) and [here](https://huggingface.co/m-a-p/YuE-upsampler/tree/main) path like below,模型路径结构如下
* 3.1 base infer [3B dit](https://huggingface.co/m-a-p/YuE2-3B) and [vae](https://huggingface.co/m-a-p/YuE2-Vae) ,only model.safetensors and vae.safetensors is needed /只需要模型文件;
* 3.1 cover mode [SheetSage2](https://huggingface.co/m-a-p/SheetSage2) and [MERT2](https://huggingface.co/m-a-p/MERT-v2-FullSong) ,only model.safetensors and vae.safetensors is needed /只需要模型文件;
* base infer:
```
-- ComfyUI/models/yue
├── ckpt_00360000.pth
├── decoder_131000.pth
├── decoder_151000.pth
-- ComfyUI/models/diffusion_models
├── yue2_model.safetensors # rename from model.safetensor or not
-- ComfyUI/models/vae
├── yue2_vae.safetensors # rename from model.safetensor or not
```
* 3.2 download from [here](https://huggingface.co/m-a-p/xcodec_mini_infer/tree/main/semantic_ckpts/hf_1_325000), path like below,模型路径结构如下
* if use cover:
```
-- ComfyUI/custom_nodes/ComfyUI_YuE/inference/xcodec_mini_infer/semantic_ckpts/hf_1_325000/
├── pytorch_model.bin
-- ComfyUI/models/clip
├── mert2_model.safetensors # rename from model.safetensor or not
├── sheetsage2.safetensors # rename from model.safetensor or not
```
* 3.3 if your GPU is 4090 or 5090 or best,just use repo below,Of course, you can also just fill out the repo,如果你是4090以上,可以用fp16,只填repo会自动下载,以下是离线版;
* 3.3.1 english [YuE-s1-7B-anneal-en-cot](https://huggingface.co/m-a-p/YuE-s1-7B-anneal-en-cot) or[YuE-s1-7B-anneal-en-icl](https://huggingface.co/m-a-p/YuE-s1-7B-anneal-en-icl) or chinese [YuE-s1-7B-anneal-zh-cot](https://huggingface.co/m-a-p/YuE-s1-7B-anneal-zh-cot),or other ,path like below,模型路径结构如下
```
-- anypath/YuE-s1-7B-anneal-en-icl # 11.5G
├── config.json
├── generation_config.json
├── model.safetensors.index.json
├── tokenizer.model
├── model-00001-of-00003.safetensors
├── model-00002-of-00003.safetensors
├── model-00003-of-00003.safetensors
```
* 3.3.2 just one [YuE-s2-1B-general](https://huggingface.co/m-a-p/YuE-s2-1B-general/tree/main) path like below,模型路径结构如下
```
-- anypath/YuE-s2-1B-general # 3.65G
├── config.json
├── generation_config.json
├── model.safetensors
├── tokenizer.model
```
* 3.4 if your VRAM=<16G,you can use 3.3 origin repo and [int8 or int4](https://github.com/alisson-anjos/YuE-Interface),[exllamav2](https://github.com/sgsdxzy/YuE-exllamav2),[deepbeepmeep](https://github.com/deepbeepmeep/YuEGP) Three quantitative acceleration methods and models,Thanks to these open-source projects ,below is list.如果显存在16G以下,可以用社区的三种量化加速方法和模型,列表如下:
- [exllamav2](https://huggingface.co/Doctor-Shotgun/YuE-s1-7B-anneal-en-cot-exl2) use fp16 Q8,Q6... [@sgsdxzy](https://github.com/sgsdxzy)
- [int8](https://huggingface.co/Alissonerdx/YuE-s1-7B-anneal-en-cot-int8) bitsandbytes [@alisson-anjos](https://github.com/alisson-anjos)
- [exllamav2](https://huggingface.co/collections/Alissonerdx/yue-models-exllamav2-67a539be76b5225ebda95323) use fp16 Q8,Q6...[@alisson-anjos](https://github.com/alisson-anjos)
- [deepbeepmeep](https://github.com/deepbeepmeep/YuEGP) use fp16,int8... [@deepbeepmeep](https://github.com/deepbeepmeep)
3.4.1 path like below,模型路径结构如下
```
-- anypath/YuE-s1-7B-anneal-en-cot-exl2-8.0bpw
├── config.json
....
-- anypath/YuE-s2-1B-general-exl2-8.0bpw
├── config.json
....
```
# 4.Use Tips
* if VRAM>=24G,use origin repo,'quantization_model'choice 'fp16',keep 'use_mmgp' false,prompt_end_time is sampler length,try 30s first.(best quality);
* if VRAM<=16G,use origin repo,'quantization_model'choice 'fp16',keep 'use_mmgp' ture,mmgp_profile choice 2,prompt_end_time is sampler length,try 30s first (best quality,slowly);
* if VRAM<=16G,use int8 repo,'quantization_model'choice 'int8',keep 'use_mmgp' false,prompt_end_time is sampler length,try 30s first.(nromal quality,very slowly 6716s,don't try it)
* if VRAM<=16G,use exllamav2 Q8 repo,'quantization_model'choice 'exllamav2',keep 'use_mmgp' false,exllamav2_cache_mode choice Q8,prompt_end_time is sampler length,try 30s first.(nromal quality,very fast)
* if VRAM>=16G,use origin repo,'quantization_model'choice 'exllamav2',keep 'use_mmgp' false,exllamav2_cache_mode choice fp16,prompt_end_time is sampler length,try 30s first.(best quality, fast and maybe OOM if VRAM<24)
* 显存大于24G,用原生的repo,quantization_model选fp16,关闭use_mmgp,prompt_end_time就是渲染时长先设置为30秒测试(普通玩家的最佳效果)
* 显存小于等于16G,用原生的repo,quantization_model选fp16,开启use_mmgp,prompt_end_time就是渲染时长先设置为30秒测试(效果好,但是慢,需要大内存)
* 显存小于等于16G,用int8的repo,'quantization_model'选int8,关闭use_mmgp,prompt_end_time就是渲染时长先设置为30秒测试(效果还行,速度奇慢6716s,不要尝试)
* 显存小于等于16G,用exllamav2的Q8 repo,'quantization_model'选exllamav2,关闭use_mmgp,exllamav2_cache_mode选择Q8,prompt_end_time就是渲染时长先设置为30秒测试(效果一般,速度非常快)
* 显存大于等于16G,用原生的repo,'quantization_model'选exllamav2,关闭use_mmgp,exllamav2_cache_mode选择fp16,prompt_end_time就是渲染时长先设置为30秒测试(效果和速度未测试)
* if "use_dual_tracks_prompt" set ture ,will Instrumental from file 'pop.00001.Instrumental.mp3' or if set "use_audio_prompt" ture and "use_dual_tracks_prompt" False will Instrumental from file "pop.00001.mp3",you can replace it ,but keep name in same.
* 开启use_dual_tracks_prompt 会借鉴'pop.00001.Instrumental.mp3',开启"use_audio_prompt"并关闭"use_dual_tracks_prompt" 会借鉴 "pop.00001.mp3",你可以用其他歌曲替换掉这两个,但是逐一保持命名不要动,因为在代码里写死了。我迟点再改。
* use_mmgp model have 5 choice(mmgp_profile) ,3-4 will quantization model auto,mmgp模式在mmgp_profile有5个选项,大于等于3会自动量化,数字越大需要内存越大,需要显存越小,但是越慢.
# 4.Example
![](https://github.com/smthemex/ComfyUI_YuE/blob/main/example_workflows/example.png)
# 5.Prompt Engineering Guide 提示词和歌词撰写官方指导
* look [Here](https://github.com/multimodal-art-projection/YuE?tab=readme-ov-file#prompt-engineering-guide) to find how to edit your Genre Tagging Prompt and Lyrics Prompt,链接直达官方的歌词和提示词指导。
* use [top_200_tags.json](https://github.com/smthemex/ComfyUI_YuE/blob/main/top_200_tags.json) to edit your prompts,查看top_200_tags.json来撰写你的歌词提示
# 6.Example
* use fp16 and use_mmgp
![](https://github.com/smthemex/ComfyUI_YuE/blob/main/example.png)
* use int8 bitsandbytes
![](https://github.com/smthemex/ComfyUI_YuE/blob/main/int8_example.png)
# 7.Citation
# 5.Citation
```
@misc{yuan2025yue,
title={YuE: Open Music Foundation Models for Full-Song Generation},
author={Ruibin Yuan and Hanfeng Lin and Shawn Guo and Ge Zhang and Jiahao Pan and Yongyi Zang and Haohe Liu and Xingjian Du and Xeron Du and Zhen Ye and Tianyu Zheng and Yinghao Ma and Minghao Liu and Lijun Yu and Zeyue Tian and Ziya Zhou and Liumeng Xue and Xingwei Qu and Yizhi Li and Tianhao Shen and Ziyang Ma and Shangda Wu and Jun Zhan and Chunhui Wang and Yatian Wang and Xiaohuan Zhou and Xiaowei Chi and Xinyue Zhang and Zhenzhu Yang and Yiming Liang and Xiangzhou Wang and Shansong Liu and Lingrui Mei and Peng Li and Yong Chen and Chenghua Lin and Xie Chen and Gus Xia and Zhaoxiang Zhang and Chao Zhang and Wenhu Chen and Xinyu Zhou and Xipeng Qiu and Roger Dannenberg and Jiaheng Liu and Jian Yang and Stephen Huang and Wei Xue and Xu Tan and Yike Guo},
howpublished={\url{https://github.com/multimodal-art-projection/YuE}},
year={2025},
note={GitHub repository}
@article{li2023mert,
title = {{MERT}: Acoustic Music Understanding Model with Large-Scale Self-supervised Training},
author = {Li, Yizhi and Yuan, Ruibin and Zhang, Ge and Ma, Yinghao and Chen, Xingran and Yin, Hanzhi and Xiao, Chenghao and Lin, Chenghua and Ragni, Anton and Benetos, Emmanouil and Gyenge, Norbert and Dannenberg, Roger and Liu, Ruibo and Chen, Wenhu and Xia, Gus and Shi, Yemin and Huang, Wenhao and Wang, Zili and Guo, Yike and Fu, Jie},
journal = {arXiv preprint arXiv:2306.00107},
year = {2023},
eprint = {2306.00107},
archivePrefix = {arXiv},
url = {https://arxiv.org/abs/2306.00107}
}
@article{yuan2025yue,
title = {{YuE}: Scaling Open Foundation Models for Long-Form Music Generation},
author = {Yuan, Ruibin and Lin, Hanfeng and Guo, Shuyue and Zhang, Ge and Pan, Jiahao and Zang, Yongyi and Liu, Haohe and Liang, Yiming and Ma, Wenye and Du, Xingjian and Du, Xinrun and Ye, Zhen and Zheng, Tianyu and Jiang, Zhengxuan and Ma, Yinghao and Liu, Minghao and Tian, Zeyue and Zhou, Ziya and Xue, Liumeng and Qu, Xingwei and Li, Yizhi and Wu, Shangda and Shen, Tianhao and Ma, Ziyang and Zhan, Jun and Wang, Chunhui and Wang, Yatian and Chi, Xiaowei and Zhang, Xinyue and Yang, Zhenzhu and Wang, Xiangzhou and Liu, Shansong and Mei, Lingrui and Li, Peng and Wang, Junjie and Yu, Jianwei and Pang, Guojian and Li, Xu and Wang, Zihao and Zhou, Xiaohuan and Yu, Lijun and Benetos, Emmanouil and Chen, Yong and Lin, Chenghua and Chen, Xie and Xia, Gus and Zhang, Zhaoxiang and Zhang, Chao and Chen, Wenhu and Zhou, Xinyu and Qiu, Xipeng and Dannenberg, Roger and Liu, Jiaheng and Yang, Jian and Huang, Wenhao and Xue, Wei and Tan, Xu and Guo, Yike},
journal = {arXiv preprint arXiv:2503.08638},
year = {2025},
eprint = {2503.08638},
archivePrefix = {arXiv},
url = {https://arxiv.org/abs/2503.08638}
}
```
+195
View File
@@ -0,0 +1,195 @@
---
license: cc-by-nc-4.0
library_name: transformers
base_model: m-a-p/MERT-v2-FullSong
base_model_relation: adapter
tags:
- audio
- music
- music-transcription
- midi
- abc-notation
- custom_code
---
<h1 align="center">🤗 SheetSage2</h1>
<p align="center"><strong>Music audio to editable scores</strong></p>
<p align="center">Melody · Chords · Beats · Key · Structure</p>
<p align="center">
<a href="https://map-yue2.github.io/">🎵&nbsp;YuE2&nbsp;project</a>
·
<a href="#quick-start">🚀&nbsp;Quick&nbsp;start</a>
·
<a href="#benchmarks">📊&nbsp;Benchmarks</a>
·
<a href="#citation">📚&nbsp;Citation</a>
</p>
<p align="center">
<a href="https://huggingface.co/m-a-p/YuE2-3B"><img alt="🤗 YuE2-3B" src="https://img.shields.io/badge/YuE2--3B-374151?logo=huggingface&amp;logoColor=FFD21E" height="20" /></a>
&nbsp;
<a href="https://huggingface.co/m-a-p/YuE2-Vae"><img alt="🤗 YuE2-Vae" src="https://img.shields.io/badge/YuE2--Vae-374151?logo=huggingface&amp;logoColor=FFD21E" height="20" /></a>
&nbsp;
<a href="https://huggingface.co/m-a-p/YuE2-Vae-legacy"><img alt="🤗 YuE2-Vae-legacy" src="https://img.shields.io/badge/YuE2--Vae--legacy-374151?logo=huggingface&amp;logoColor=FFD21E" height="20" /></a>
&nbsp;
<a href="https://huggingface.co/m-a-p/MERT-v2-30s"><img alt="🤗 MERT-v2-30s" src="https://img.shields.io/badge/MERT--v2--30s-374151?logo=huggingface&amp;logoColor=FFD21E" height="20" /></a>
&nbsp;
<a href="https://huggingface.co/m-a-p/MERT-v2-FullSong"><img alt="🤗 MERT-v2-FullSong" src="https://img.shields.io/badge/MERT--v2--FullSong-374151?logo=huggingface&amp;logoColor=FFD21E" height="20" /></a>
&nbsp;
<a href="https://huggingface.co/datasets/m-a-p/WildSongBench"><img alt="🤗 WildSongBench" src="https://img.shields.io/badge/WildSongBench-374151?logo=huggingface&amp;logoColor=FFD21E" height="20" /></a>
&nbsp;
<a href="https://huggingface.co/m-a-p/SheetSage2"><img alt="SheetSage2" src="https://img.shields.io/badge/SheetSage2-374151?logo=huggingface&amp;logoColor=FFD21E" height="20" /></a>
</p>
**SheetSage2 turns music recordings into lead sheets and timed musical annotations.** Transcribe a complete song, edit its ABC or MIDI, render a piano preview, or extract embeddings and token predictions for your own tools.
Built on [MERT-v2-FullSong](https://huggingface.co/m-a-p/MERT-v2-FullSong), with adapters that merge automatically when you load the model.
![SheetSage2 architecture](assets/architecture.png)
<a id="quick-start"></a>
## 🚀 Quick start
Use Python 3.10 or 3.11 and FFmpeg 6.1 with its shared libraries. Sign in with access to this repository and its MERT-v2 parent:
```bash
python -m pip install huggingface-hub==0.36.0
huggingface-cli download m-a-p/SheetSage2 --local-dir SheetSage2
cd SheetSage2
python -m pip install torch==2.8.0 torchaudio==2.8.0 --index-url https://download.pytorch.org/whl/cu126
python -m pip install -r requirements.txt
```
```python
import torch
from transformers import AutoModel
model = AutoModel.from_pretrained(
"m-a-p/SheetSage2", trust_remote_code=True,
).eval().to("cuda" if torch.cuda.is_available() else "cpu")
result = model.transcribe("song.mp3", output_dir="output")
```
Files are decoded, mixed to mono and resampled automatically. Long songs use overlapping windows.
| Output | File |
|---|---|
| Editable score | `score.abc` |
| Melody and chord accompaniment | `transcription.mid` |
| Separate melodies and chords | `melody_vocal.mid`, `melody_instrumental.mid`, `chords.mid` |
| Timed annotations | `events.json`, `*.lab` |
Omit `output_dir` to keep results in memory. Pass a Tensor or NumPy waveform with its sample rate:
```python
# waveform: [samples] or [channels, samples]
result = model.transcribe(waveform, sampling_rate=24000)
abc = result["abc"] # str
midi = result["midi"] # bytes
events = result["events"] # list of timed events
```
Paths, encoded audio bytes and binary streams are also accepted. `result["midis"]` contains the separate MIDI parts; `result["labs"]` contains annotation text. Request `export_logits=True`, `export_scores=True` or `export_embeddings=True` for CPU tensors in `result["tensors"]`, grouped by window; add `output_hidden_states=True` for all 24 MERT layers. With `output_dir`, these tensors are saved as safetensors instead. Low-level `forward()` returns logits and optional hidden states; `generate()` returns symbolic tokens.
Save a self-contained model for offline use:
```python
model.save_pretrained("sheetsage2-local")
model = AutoModel.from_pretrained(
"sheetsage2-local", trust_remote_code=True, local_files_only=True,
)
```
### Melody-only ABC for covers
Set `melody_only=True` to retain both the `Vocal` and `Ins` melodies while omitting chord symbols from the ABC and chord accompaniment from playback/combined MIDI. Raw predicted annotations remain available; the default full transcription is unchanged.
```python
result = model.transcribe("song.mp3", output_dir="cover-score", melody_only=True)
abc = result["abc"] # Also saved as cover-score/score.abc.
```
```bash
python infer.py song.mp3 --output cover-score --melody-only
```
Review the transcription, then pass `result["abc"]` or the saved `cover-score/score.abc` to [YuE2](https://huggingface.co/m-a-p/YuE2-3B) as `abc`, with `cot="melody"` and your target `style` and `lyrics`. If the requested ABC cannot be produced, Python raises an error (partial results are available as `error.result`) and the CLI exits with a nonzero status.
### Command line and rendering
```bash
python infer.py song.mp3 --output output
# Optional piano audio and printable sheet music:
python setup_render.py
python infer.py song.mp3 --output output --render-audio --render-score
# Render existing results without loading the model:
python render.py --input output --output rendered --audio --score pdf,svg,png
```
On minimal Linux servers, use `python setup_render.py --with-deps`. Audio follows the original MIDI timing. Scores preserve the vocal and instrumental staves. For separate piano previews, use `--render-parts vocal,instrumental,chords` with `infer.py`, or `--parts vocal,instrumental,chords` with `render.py`.
Python also accepts `render_audio=True, render_score="pdf,svg,png"`. In memory mode, `result["rendered"]` contains `audio` (part → WAV bytes) and `score` (format → pages as PDF/PNG bytes or SVG text).
<a id="benchmarks"></a>
## 📊 Benchmarks
Scores (%) on H800 with BF16 inference; higher is better. Use `preset="paper"` or `--preset paper` with the dataset prompts in [scores and settings](benchmark_results.json).
| Task | Dataset | Metric | SheetSage1 | madmom | Specialist | SheetSage2 |
|---|---|---|---:|---:|---:|---:|
| Beat | GTZAN | F1 | 85.79 | 85.79 | **88.75** [Beat This!][beat] | 85.65 |
| Beat | osu2017 | F1 | 91.55 | 91.55 | 88.18 [Beat This!][beat] | **92.29** |
| Downbeat | GTZAN | F1 | 64.33 | 64.33 | 78.28 [Beat This!][beat] | **79.51** |
| Downbeat | osu2017 | F1 | 83.22 | 83.22 | 84.99 [Beat This!][beat] | **91.97** |
| Key | GiantSteps | Weighted | 43.89 | 74.62 | 72.09 [S-KEY][key] | **77.73** |
| Key | GTZAN | Weighted | 54.56 | 72.05 | 74.43 [S-KEY][key] | **75.77** |
| Chord | osu2017 | Maj/min | 79.82 | 77.42 | 84.59 [Jiang et al.][chord] | **90.08** |
| Chord | Chords1217 | Maj/min | 72.98 | 83.52* | **84.09**† [ChordFormer][chordformer] | 83.81 |
| Structure | HarmonixSet | Accuracy | — | — | 80.03 [SongFormer][structure] | **80.51** |
| Structure | HarmonixSet | F1 @ 0.5 s | — | — | **70.63** [SongFormer][structure] | 67.96 |
| Structure | HarmonixSet | F1 @ 3 s | — | — | 79.50 [SongFormer][structure] | **82.86** |
| Melody | RWC-Pop | Vocal F1 | 62.71 | — | 62.71 [SheetSage1][sheetsage] | **82.51** |
| Melody | RWC-Pop | Full F1 | 64.02 | — | 64.02 [SheetSage1][sheetsage] | **75.29** |
Melody F1 uses pitch classes. SheetSage1 beat/downbeat results use madmom. *The madmom chord model includes Chords1217 in training. †ChordFormer uses five-fold cross-validation; SheetSage2 uses one checkpoint across all tracks.
[beat]: https://doi.org/10.5281/zenodo.14877491
[key]: https://doi.org/10.1109/ICASSP49660.2025.10890222
[chord]: https://archives.ismir.net/ismir2019/paper/000078.pdf
[chordformer]: https://arxiv.org/abs/2502.11840
[structure]: https://arxiv.org/abs/2510.02797
[sheetsage]: https://arxiv.org/abs/2212.01884
<a id="citation"></a>
## 📚 Citation
**Technical report coming soon.** For now, please cite [YuE](https://arxiv.org/abs/2503.08638) and [MERT](https://proceedings.iclr.cc/paper_files/paper/2024/hash/33dffa2e3d2ab74a783d1a8c292f66d9-Abstract-Conference.html) when using SheetSage2 in your research.
```bibtex
@article{yuan2025yue,
title = {{YuE}: Scaling Open Foundation Models for Long-Form Music Generation},
author = {Yuan, Ruibin and Lin, Hanfeng and Guo, Shuyue and Zhang, Ge and Pan, Jiahao and Zang, Yongyi and Liu, Haohe and Liang, Yiming and Ma, Wenye and Du, Xingjian and Du, Xinrun and Ye, Zhen and Zheng, Tianyu and Jiang, Zhengxuan and Ma, Yinghao and Liu, Minghao and Tian, Zeyue and Zhou, Ziya and Xue, Liumeng and Qu, Xingwei and Li, Yizhi and Wu, Shangda and Shen, Tianhao and Ma, Ziyang and Zhan, Jun and Wang, Chunhui and Wang, Yatian and Chi, Xiaowei and Zhang, Xinyue and Yang, Zhenzhu and Wang, Xiangzhou and Liu, Shansong and Mei, Lingrui and Li, Peng and Wang, Junjie and Yu, Jianwei and Pang, Guojian and Li, Xu and Wang, Zihao and Zhou, Xiaohuan and Yu, Lijun and Benetos, Emmanouil and Chen, Yong and Lin, Chenghua and Chen, Xie and Xia, Gus and Zhang, Zhaoxiang and Zhang, Chao and Chen, Wenhu and Zhou, Xinyu and Qiu, Xipeng and Dannenberg, Roger and Liu, Jiaheng and Yang, Jian and Huang, Wenhao and Xue, Wei and Tan, Xu and Guo, Yike},
journal = {arXiv preprint arXiv:2503.08638},
year = {2025},
eprint = {2503.08638},
archivePrefix = {arXiv},
url = {https://arxiv.org/abs/2503.08638}
}
@inproceedings{li2024mert,
title = {MERT: Acoustic Music Understanding Model with Large-Scale Self-supervised Training},
author = {Li, Yizhi and Yuan, Ruibin and Zhang, Ge and Ma, Yinghao and Chen, Xingran and Yin, Hanzhi and Xiao, Chenghao and Lin, Chenghua and Ragni, Anton and Benetos, Emmanouil and Gyenge, Norbert and Dannenberg, Roger and Liu, Ruibo and Chen, Wenhu and Xia, Gus and Shi, Yemin and Huang, Wenhao and Wang, Zili and Guo, Yike and Fu, Jie},
booktitle = {International Conference on Learning Representations},
year = {2024},
url = {https://proceedings.iclr.cc/paper_files/paper/2024/hash/33dffa2e3d2ab74a783d1a8c292f66d9-Abstract-Conference.html}
}
```
Weights: [CC BY-NC 4.0](LICENSE). [Third-party notices](THIRD_PARTY_NOTICES.md).
**YuE2 family:** [Song generation](https://huggingface.co/m-a-p/YuE2-3B) · [Music representations](https://huggingface.co/m-a-p/MERT-v2-FullSong) · [Audio decoder](https://huggingface.co/m-a-p/YuE2-Vae).
+8
View File
@@ -0,0 +1,8 @@
# Third-party notices
- MERT-v2-FullSong provides the public music encoder. Its weight terms remain CC BY-NC 4.0.
- The BART decoder uses Hugging Face Transformers (Apache 2.0); PyTorch uses its BSD-style license.
- abcjs 6.6.3 is distributed under MIT; its license accompanies the rendering assets.
- Playwright is distributed under Apache 2.0. Chromium is installed separately with its accompanying notices.
- FluidR3 piano samples by Frank Wen use CC BY 3.0 US. Attribution and source information accompany the samples.
- Other installed dependencies retain their respective licenses.
+1
View File
@@ -0,0 +1 @@
"""SheetSage2 standalone inference."""
+92
View File
@@ -0,0 +1,92 @@
"""File decoding and waveform preparation for music transcription."""
from pathlib import Path
import io
import math
import numbers
import shutil
import subprocess
import numpy as np
import torch
SAMPLE_RATE = 24000
def _encoded_bytes(audio):
if isinstance(audio, (bytes, bytearray, memoryview)):
return bytes(audio)
if callable(getattr(audio, "read", None)):
position = None
try:
position = audio.tell()
audio.seek(0)
except (AttributeError, OSError, ValueError, io.UnsupportedOperation):
position = None
try:
value = audio.read()
finally:
if position is not None:
audio.seek(position)
if not isinstance(value, (bytes, bytearray)):
raise ValueError("Audio streams must return encoded audio bytes")
return bytes(value)
return None
def load_audio(audio, *, sampling_rate=None, max_seconds=None, preset="default"):
if max_seconds is not None and (not math.isfinite(max_seconds) or max_seconds <= 0):
raise ValueError("max_seconds must be finite and positive")
encoded = _encoded_bytes(audio)
if encoded is not None and not encoded:
raise ValueError("Encoded audio must not be empty")
if isinstance(audio, (str, Path)) or encoded is not None:
if preset == "paper":
import torchaudio
source = io.BytesIO(encoded) if encoded is not None else str(audio)
info = torchaudio.info(source, backend="ffmpeg")
if info.num_frames <= 0:
raise ValueError("Cannot determine source audio length")
if encoded is not None:
source.seek(0)
waveform, rate = torchaudio.load(source, frame_offset=0, num_frames=int(info.num_frames),
backend="ffmpeg", channels_first=True)
waveform = waveform.mean(dim=0)
if rate != SAMPLE_RATE:
waveform = torchaudio.functional.resample(waveform, rate, SAMPLE_RATE)
else:
if not shutil.which("ffmpeg"):
raise RuntimeError("FFmpeg is required to read audio files; install it and add it to PATH")
source = "pipe:0" if encoded is not None else str(Path(audio).resolve())
command = ["ffmpeg", "-v", "error", "-nostdin", "-i", source, "-vn"]
if max_seconds is not None:
command += ["-t", str(float(max_seconds))]
command += ["-ac", "1", "-ar", str(SAMPLE_RATE), "-f", "f32le", "pipe:1"]
result = subprocess.run(command, input=encoded, capture_output=True, timeout=600, check=False)
if result.returncode:
raise ValueError("Cannot decode audio: " + result.stderr.decode(errors="replace")[-1200:])
waveform = torch.from_numpy(np.frombuffer(result.stdout, dtype="<f4").copy())
else:
if not isinstance(sampling_rate, numbers.Real) or not math.isfinite(sampling_rate) or sampling_rate <= 0 or sampling_rate != int(sampling_rate):
raise ValueError("Provide sampling_rate for an array or tensor waveform")
waveform = torch.as_tensor(audio, dtype=torch.float32, device="cpu")
if waveform.ndim == 2:
if waveform.shape[0] > 32:
raise ValueError("Multichannel audio must have shape [channels, samples]")
waveform = waveform.mean(dim=0)
if waveform.ndim != 1:
raise ValueError("Audio must have shape [samples] or [channels, samples]")
if sampling_rate != SAMPLE_RATE:
import torchaudio
waveform = torchaudio.functional.resample(waveform, int(sampling_rate), SAMPLE_RATE)
waveform = waveform.float().contiguous()
if max_seconds is not None:
waveform = waveform[:round(max_seconds * SAMPLE_RATE)]
if waveform.numel() < 1025 or not torch.isfinite(waveform).all():
raise ValueError("Audio must contain at least 1025 finite samples at 24 kHz")
return waveform
def slice_audio(audio, start, seconds):
offset, count = round(start * SAMPLE_RATE), round(seconds * SAMPLE_RATE)
result = audio[offset:offset + count]
return torch.nn.functional.pad(result, (0, count - len(result)))
+351
View File
@@ -0,0 +1,351 @@
{
"scores": [
{
"task": "beat",
"dataset": "GTZAN",
"metric": "F1",
"score": 85.65,
"settings": {
"preset": "paper",
"sample_rate": 24000,
"window_seconds": 300,
"overlap_seconds": 100,
"lookahead_seconds": 0,
"max_length": 5120,
"decoding": "greedy",
"attention": "sdpa",
"weight_dtype": "float32",
"autocast_dtype": "bfloat16",
"batch_size": 1,
"prompts": [
"timestamp",
"downbeat_meter",
"key"
],
"device": "NVIDIA H800",
"cuda_version": "12.6"
}
},
{
"task": "beat",
"dataset": "osu2017",
"metric": "F1",
"score": 92.29,
"settings": {
"preset": "paper",
"sample_rate": 24000,
"window_seconds": 300,
"overlap_seconds": 100,
"lookahead_seconds": 0,
"max_length": 5120,
"decoding": "greedy",
"attention": "sdpa",
"weight_dtype": "float32",
"autocast_dtype": "bfloat16",
"batch_size": 1,
"prompts": [
"timestamp",
"downbeat_meter",
"key",
"chord_full"
],
"device": "NVIDIA H800",
"cuda_version": "12.6"
}
},
{
"task": "downbeat",
"dataset": "GTZAN",
"metric": "F1",
"score": 79.51,
"settings": {
"preset": "paper",
"sample_rate": 24000,
"window_seconds": 300,
"overlap_seconds": 100,
"lookahead_seconds": 0,
"max_length": 5120,
"decoding": "greedy",
"attention": "sdpa",
"weight_dtype": "float32",
"autocast_dtype": "bfloat16",
"batch_size": 1,
"prompts": [
"timestamp",
"downbeat_meter",
"key"
],
"device": "NVIDIA H800",
"cuda_version": "12.6"
}
},
{
"task": "downbeat",
"dataset": "osu2017",
"metric": "F1",
"score": 91.97,
"settings": {
"preset": "paper",
"sample_rate": 24000,
"window_seconds": 300,
"overlap_seconds": 100,
"lookahead_seconds": 0,
"max_length": 5120,
"decoding": "greedy",
"attention": "sdpa",
"weight_dtype": "float32",
"autocast_dtype": "bfloat16",
"batch_size": 1,
"prompts": [
"timestamp",
"downbeat_meter",
"key",
"chord_full"
],
"device": "NVIDIA H800",
"cuda_version": "12.6"
}
},
{
"task": "key",
"dataset": "GiantSteps",
"metric": "weighted_accuracy",
"score": 77.73,
"settings": {
"preset": "paper",
"sample_rate": 24000,
"window_seconds": 300,
"overlap_seconds": 100,
"lookahead_seconds": 0,
"max_length": 5120,
"decoding": "greedy",
"attention": "sdpa",
"weight_dtype": "float32",
"autocast_dtype": "bfloat16",
"batch_size": 1,
"prompts": [
"timestamp",
"downbeat_meter",
"key"
],
"device": "NVIDIA H800",
"cuda_version": "12.6"
}
},
{
"task": "key",
"dataset": "GTZAN",
"metric": "weighted_accuracy",
"score": 75.77,
"settings": {
"preset": "paper",
"sample_rate": 24000,
"window_seconds": 300,
"overlap_seconds": 100,
"lookahead_seconds": 0,
"max_length": 5120,
"decoding": "greedy",
"attention": "sdpa",
"weight_dtype": "float32",
"autocast_dtype": "bfloat16",
"batch_size": 1,
"prompts": [
"timestamp",
"downbeat_meter",
"key"
],
"device": "NVIDIA H800",
"cuda_version": "12.6"
}
},
{
"task": "chord",
"dataset": "osu2017",
"metric": "majmin",
"score": 90.08,
"settings": {
"preset": "paper",
"sample_rate": 24000,
"window_seconds": 300,
"overlap_seconds": 100,
"lookahead_seconds": 0,
"max_length": 5120,
"decoding": "greedy",
"attention": "sdpa",
"weight_dtype": "float32",
"autocast_dtype": "bfloat16",
"batch_size": 1,
"prompts": [
"timestamp",
"downbeat_meter",
"key",
"chord_full"
],
"device": "NVIDIA H800",
"cuda_version": "12.6"
}
},
{
"task": "chord",
"dataset": "Chords1217",
"metric": "majmin",
"score": 83.81,
"settings": {
"preset": "paper",
"sample_rate": 24000,
"window_seconds": 300,
"overlap_seconds": 100,
"lookahead_seconds": 0,
"max_length": 5120,
"decoding": "greedy",
"attention": "sdpa",
"weight_dtype": "float32",
"autocast_dtype": "bfloat16",
"batch_size": 1,
"prompts": [
"timestamp",
"downbeat_meter",
"chord_full"
],
"device": "NVIDIA H800",
"cuda_version": "12.6"
}
},
{
"task": "structure",
"dataset": "HarmonixSet",
"metric": "accuracy",
"score": 80.51,
"settings": {
"preset": "paper",
"sample_rate": 24000,
"window_seconds": 300,
"overlap_seconds": 100,
"lookahead_seconds": 0,
"max_length": 5120,
"decoding": "greedy",
"attention": "sdpa",
"weight_dtype": "float32",
"autocast_dtype": "bfloat16",
"batch_size": 1,
"prompts": [
"timestamp",
"downbeat_meter",
"structure"
],
"device": "NVIDIA H800",
"cuda_version": "12.6"
}
},
{
"task": "structure",
"dataset": "HarmonixSet",
"metric": "F1@0.5s",
"score": 67.96,
"settings": {
"preset": "paper",
"sample_rate": 24000,
"window_seconds": 300,
"overlap_seconds": 100,
"lookahead_seconds": 0,
"max_length": 5120,
"decoding": "greedy",
"attention": "sdpa",
"weight_dtype": "float32",
"autocast_dtype": "bfloat16",
"batch_size": 1,
"prompts": [
"timestamp",
"downbeat_meter",
"structure"
],
"device": "NVIDIA H800",
"cuda_version": "12.6"
}
},
{
"task": "structure",
"dataset": "HarmonixSet",
"metric": "F1@3s",
"score": 82.86,
"settings": {
"preset": "paper",
"sample_rate": 24000,
"window_seconds": 300,
"overlap_seconds": 100,
"lookahead_seconds": 0,
"max_length": 5120,
"decoding": "greedy",
"attention": "sdpa",
"weight_dtype": "float32",
"autocast_dtype": "bfloat16",
"batch_size": 1,
"prompts": [
"timestamp",
"downbeat_meter",
"structure"
],
"device": "NVIDIA H800",
"cuda_version": "12.6"
}
},
{
"task": "melody",
"dataset": "RWC-Pop",
"metric": "vocal_pitch_class_F1",
"score": 82.51,
"settings": {
"preset": "paper",
"sample_rate": 24000,
"window_seconds": 300,
"overlap_seconds": 100,
"lookahead_seconds": 0,
"max_length": 5120,
"decoding": "greedy",
"attention": "sdpa",
"weight_dtype": "float32",
"autocast_dtype": "bfloat16",
"batch_size": 1,
"prompts": [
"timestamp",
"downbeat_meter",
"structure",
"key",
"chord_full",
"melody_full"
],
"device": "NVIDIA H800",
"cuda_version": "12.6"
}
},
{
"task": "melody",
"dataset": "RWC-Pop",
"metric": "full_pitch_class_F1",
"score": 75.29,
"settings": {
"preset": "paper",
"sample_rate": 24000,
"window_seconds": 300,
"overlap_seconds": 100,
"lookahead_seconds": 0,
"max_length": 5120,
"decoding": "greedy",
"attention": "sdpa",
"weight_dtype": "float32",
"autocast_dtype": "bfloat16",
"batch_size": 1,
"prompts": [
"timestamp",
"downbeat_meter",
"structure",
"key",
"chord_full",
"melody_full"
],
"device": "NVIDIA H800",
"cuda_version": "12.6"
}
}
]
}
+75
View File
@@ -0,0 +1,75 @@
{
"architectures": [
"SheetSage2Model"
],
"auto_map": {
"AutoConfig": "configuration_sheetsage2.SheetSage2Config",
"AutoModel": "modeling_sheetsage2.SheetSage2Model",
"AutoModelForSeq2SeqLM": "modeling_sheetsage2.SheetSage2Model",
"AutoProcessor": "processing_sheetsage2.SheetSage2Processor"
},
"backbone_config": {
"architectures": [
"MERT2Model"
],
"context_seconds": 360.0,
"conv_depthwise_kernel_size": 31,
"frame_rate": 25.0,
"hidden_size": 1024,
"hop_length": 240,
"initializer_range": 0.02,
"inputs_to_logits_ratio": 960,
"intermediate_size": 4096,
"layer_norm_eps": 1e-05,
"minimum_input_samples": 1025,
"model_type": "mert2",
"n_fft": 2048,
"num_attention_heads": 16,
"num_hidden_layers": 24,
"num_mel_bins": 128,
"rotary_embedding_base": 10000,
"sampling_rate": 24000,
"subsampling_channels": [
128,
512,
1024
],
"subsampling_depths": [
3,
4,
5
],
"subsampling_layer_norm_eps": 1e-06,
"torch_dtype": "float32",
"transformers_version": "4.53.2",
"variant": "fs",
"win_length": 2048
},
"base_model_name_or_path": "m-a-p/MERT-v2-FullSong",
"base_model_revision": "d8ba1c745e733b3908ce6ad16ebeb17ac7600a42",
"base_model_sha256": "e6dd2ab187d6dd62b6521cd7d8f932e237acf0c5757745a7232082e28391350d",
"bos_token_id": 1,
"decoder_dropout": 0.1,
"decoder_layers": 6,
"decoder_start_token_id": 1,
"encoder_attn_implementation": "sdpa",
"eos_token_id": 2,
"hidden_size": 512,
"input_audio_length": 300.0,
"intermediate_size": 2048,
"is_encoder_decoder": true,
"lora_alpha": 128.0,
"lora_rank": 64,
"max_output_seq_len": 5120,
"model_type": "sheetsage2",
"num_attention_heads": 8,
"pad_token_id": 0,
"sampling_rate": 24000,
"time_hz": 100,
"tokenizer_fingerprint": "5ba3325af0344c7f",
"tokenizer_schema_version": "v1",
"transformers_version": "4.45.2",
"use_cache": true,
"vocab_size": 31678,
"weights_format": "adapter"
}
+76
View File
@@ -0,0 +1,76 @@
"""Configuration for MERT2 music representation models."""
from transformers import PretrainedConfig
class MERT2Config(PretrainedConfig):
"""Architecture shared by the 30-second and full-song MERT2 encoders."""
model_type = "mert2"
def __init__(
self,
hidden_size=1024,
intermediate_size=4096,
num_hidden_layers=24,
num_attention_heads=16,
num_mel_bins=128,
sampling_rate=24000,
n_fft=2048,
win_length=2048,
hop_length=240,
subsampling_channels=None,
subsampling_depths=None,
conv_depthwise_kernel_size=31,
rotary_embedding_base=10000,
layer_norm_eps=1e-5,
subsampling_layer_norm_eps=1e-6,
initializer_range=0.02,
variant="30s",
context_seconds=None,
**kwargs,
):
super().__init__(**kwargs)
self.hidden_size = int(hidden_size)
self.intermediate_size = int(intermediate_size)
self.num_hidden_layers = int(num_hidden_layers)
self.num_attention_heads = int(num_attention_heads)
self.num_mel_bins = int(num_mel_bins)
self.sampling_rate = int(sampling_rate)
self.n_fft = int(n_fft)
self.win_length = int(win_length)
self.hop_length = int(hop_length)
self.subsampling_channels = list(subsampling_channels or [num_mel_bins, 512, hidden_size])
self.subsampling_depths = list(subsampling_depths or [3, 4, 5])
self.conv_depthwise_kernel_size = int(conv_depthwise_kernel_size)
self.rotary_embedding_base = int(rotary_embedding_base)
self.layer_norm_eps = float(layer_norm_eps)
self.subsampling_layer_norm_eps = float(subsampling_layer_norm_eps)
self.initializer_range = float(initializer_range)
self.variant = str(variant)
self.context_seconds = float(context_seconds if context_seconds is not None else (360 if variant == "fs" else 30))
self.inputs_to_logits_ratio = self.hop_length * 4
self.frame_rate = self.sampling_rate / self.inputs_to_logits_ratio
self.minimum_input_samples = self.n_fft // 2 + 1
if min(self.hidden_size, self.intermediate_size, self.num_hidden_layers, self.num_attention_heads) <= 0:
raise ValueError("Encoder dimensions and layer counts must be positive.")
if self.hidden_size % self.num_attention_heads or (self.hidden_size // self.num_attention_heads) % 2:
raise ValueError("hidden_size must divide into an even head dimension.")
if len(self.subsampling_channels) != 3 or len(self.subsampling_depths) != 3:
raise ValueError("The subsampler must contain three channel widths and three depths.")
if self.subsampling_channels[0] != self.num_mel_bins or self.subsampling_channels[-1] != self.hidden_size:
raise ValueError("Subsampling widths must start at num_mel_bins and end at hidden_size.")
if min(*self.subsampling_channels, *self.subsampling_depths, self.num_mel_bins, self.sampling_rate, self.hop_length) <= 0:
raise ValueError("Frontend dimensions and sampling parameters must be positive.")
if self.n_fft < self.win_length or self.win_length <= 0 or self.n_fft % 2:
raise ValueError("n_fft must be even and at least win_length > 0.")
if self.conv_depthwise_kernel_size <= 0 or self.conv_depthwise_kernel_size % 2 != 1:
raise ValueError("conv_depthwise_kernel_size must be positive and odd.")
if self.rotary_embedding_base <= 0 or min(self.layer_norm_eps, self.subsampling_layer_norm_eps) <= 0:
raise ValueError("Rotary base and normalization epsilons must be positive.")
if self.variant not in {"30s", "fs"}:
raise ValueError("variant must be '30s' or 'fs'.")
MERT2Config.register_for_auto_class()
+88
View File
@@ -0,0 +1,88 @@
"""Configuration for SheetSage2 audio-to-symbolic transcription."""
from transformers import PretrainedConfig
from .configuration_mert2 import MERT2Config
class SheetSage2Config(PretrainedConfig):
model_type = "sheetsage2"
def __init__(
self,
vocab_size=31678,
hidden_size=512,
decoder_layers=6,
num_attention_heads=8,
intermediate_size=2048,
decoder_dropout=0.1,
input_audio_length=300.0,
max_output_seq_len=5120,
time_hz=100,
sampling_rate=24000,
lora_rank=64,
lora_alpha=128,
weights_format="adapter",
backbone_config=None,
encoder_attn_implementation="sdpa",
base_model_name_or_path="m-a-p/MERT-v2-FullSong",
base_model_revision="d8ba1c745e733b3908ce6ad16ebeb17ac7600a42",
base_model_sha256="e6dd2ab187d6dd62b6521cd7d8f932e237acf0c5757745a7232082e28391350d",
tokenizer_schema_version="v1",
tokenizer_fingerprint="5ba3325af0344c7f",
**kwargs,
):
for name, default in (
("is_encoder_decoder", True), ("tie_word_embeddings", True),
("pad_token_id", 0), ("bos_token_id", 1), ("eos_token_id", 2),
("decoder_start_token_id", 1),
("use_cache", True),
):
kwargs.setdefault(name, default)
super().__init__(**kwargs)
self.vocab_size = int(vocab_size)
self.hidden_size = int(hidden_size)
self.decoder_layers = int(decoder_layers)
self.num_attention_heads = int(num_attention_heads)
self.intermediate_size = int(intermediate_size)
self.decoder_dropout = float(decoder_dropout)
self.input_audio_length = float(input_audio_length)
self.max_output_seq_len = int(max_output_seq_len)
self.time_hz = int(time_hz)
self.sampling_rate = int(sampling_rate)
self.lora_rank = int(lora_rank)
self.lora_alpha = float(lora_alpha)
self.weights_format = str(weights_format)
self.backbone_config = dict(backbone_config or MERT2Config(variant="fs").to_dict())
self.backbone_config.pop("_name_or_path", None)
self.backbone_config.pop("auto_map", None)
self.encoder_attn_implementation = str(encoder_attn_implementation)
self.base_model_name_or_path = str(base_model_name_or_path)
self.base_model_revision = str(base_model_revision)
self.base_model_sha256 = str(base_model_sha256)
self.tokenizer_schema_version = str(tokenizer_schema_version)
self.tokenizer_fingerprint = str(tokenizer_fingerprint)
self.architectures = ["SheetSage2Model"]
self.auto_map = {
"AutoConfig": "configuration_sheetsage2.SheetSage2Config",
"AutoModel": "modeling_sheetsage2.SheetSage2Model",
"AutoModelForSeq2SeqLM": "modeling_sheetsage2.SheetSage2Model",
"AutoProcessor": "processing_sheetsage2.SheetSage2Processor",
}
if self.weights_format not in {"adapter", "merged"}:
raise ValueError("weights_format must be 'adapter' or 'merged'.")
if self.encoder_attn_implementation not in {"sdpa", "flash_attention_2"}:
raise ValueError("Select encoder attention 'sdpa' or 'flash_attention_2'.")
if min(self.vocab_size, self.hidden_size, self.decoder_layers, self.num_attention_heads,
self.intermediate_size, self.input_audio_length, self.max_output_seq_len,
self.time_hz, self.sampling_rate, self.lora_rank, self.lora_alpha) <= 0:
raise ValueError("Model dimensions and timing parameters must be positive.")
if self.hidden_size % self.num_attention_heads:
raise ValueError("hidden_size must be divisible by num_attention_heads.")
if not 0 <= self.decoder_dropout < 1:
raise ValueError("decoder_dropout must lie in [0, 1).")
if self.sampling_rate != self.backbone_config["sampling_rate"]:
raise ValueError("Processor and encoder sampling rates must match.")
SheetSage2Config.register_for_auto_class()
+14
View File
@@ -0,0 +1,14 @@
import numpy as np
DURATION_TEMPLATES = np.array(
[
1, 2, 3, 4, 6, 8, 12, 16, 24, 32, 48, 64,
96, 128, 192, 256, 384, 512, 768, 1024,
1536, 2048, 3072, 4096,
],
dtype=np.int32,
)
duration_boundaries = (DURATION_TEMPLATES[:-1] + DURATION_TEMPLATES[1:]) / 2
+200
View File
@@ -0,0 +1,200 @@
"""Lossless decoded events/LAB/MIDI, then independently validated ABC."""
from pathlib import Path
import json
from fractions import Fraction
import numpy as np
import pretty_midi
from .notation_sheetsage2 import generate_abc_from_data
from .io_sheetsage2 import atomic_write_text
from .generation_sheetsage2 import field_text
from .midi_sheetsage2 import build_playback, midi_bytes
def rows_text(rows):
return "".join("\t".join(str(x) for x in row) + "\n" for row in rows)
def write_rows(path, rows):
atomic_write_text(path, rows_text(rows))
def interval_rows(events, field, duration):
rows = [[e["time"], 0, e["values"][field]] for e in events if field in e["values"]]
for i, row in enumerate(rows):
row[1] = rows[i + 1][0] if i + 1 < len(rows) else duration
return [r for r in rows if r[1] > r[0]]
def rhythm_rows(events):
rows, meter = [], None
for event in events:
rhythm = event["values"].get("rhythm", {})
meter = rhythm.get("meter", meter)
eighth = rhythm.get("eighth_position")
if eighth is not None and meter is not None:
# The model stores eighth-note position, not denominator-beat index.
position = Fraction(int(eighth) * int(meter[1]), 8)
if position.denominator != 1:
raise ValueError(f"Eighth position {eighth} is off the {meter[0]}/{meter[1]} beat grid")
if not 0 <= position < meter[0]:
raise ValueError(f"Eighth position {eighth} is outside meter {meter}")
rows.append([float(event["time"]), int(position) + 1, int(meter[0]), int(meter[1])])
return rows
def _midi(notes, paper=False):
midi = pretty_midi.PrettyMIDI(resolution=220 if paper else 960)
names = ((0, "vocal_melody"), (1, "instrumental_melody")) if paper else ((0, "Vocal"), (1, "Ins"))
for track, name in names:
instrument = pretty_midi.Instrument(program=0, name=name)
for start, end, pitch, source in notes:
if track == source and end > start:
instrument.notes.append(pretty_midi.Note(100, pitch, start, end))
midi.instruments.append(instrument)
return midi
def notation_notes(notes):
"""A monophonic notation view, retaining the exact raw prediction separately."""
result, diagnostics = [], []
for track in (0, 1):
ordered = sorted((list(n) for n in notes if n[3] == track), key=lambda n: (n[0], n[2], n[1]))
for i, note in enumerate(ordered):
if i + 1 < len(ordered) and note[1] > ordered[i + 1][0] + 1e-6:
note[1] = ordered[i + 1][0]
diagnostics.append(f"notation only: clipped track {track} note at {note[0]:.3f} to next onset")
if note[1] > note[0] + 1e-6:
result.append(note)
return sorted(result), diagnostics
def export_result(decoded, tokenizer, output_dir=None, duration=None, paper=False, *, melody_only=False):
"""Build ABC, MIDI and LAB in memory, optionally writing the same bytes.
Statistics retain their file-interface names. ``payload`` contains the
in-memory ABC text, MIDI bytes, decoded event list, LAB texts and playback.
melody_only removes chords from notation and playback without changing
decoded events or raw LAB annotations.
"""
if not isinstance(melody_only, bool):
raise ValueError("melody_only must be True or False")
if duration is None:
raise TypeError("duration is required")
texts, midis, notation_midis, labs = {}, {}, {}, {}
def keep_rows(name, rows):
text = rows_text(rows)
texts[name] = text
if name.endswith(".lab"):
labs[name[:-4]] = text
events = decoded["events"]
notes = []
for event in events:
start = float(event["time"])
for note in event["values"].get("melody", ()):
end = max(start + 0.04, float(note["end_time"])) if paper else min(duration, float(note["end_time"]))
if end > start:
notes.append([start, end, int(note["pitch"]), int(note["track"])])
if not paper:
notes.sort()
texts["events.json"] = json.dumps(decoded, indent=2)
keep_rows("events.tsv", [["time", "global_subbeat", "fields"]] +
[[e["time"], e["global_subbeat"], field_text(e, tokenizer)] for e in events])
midis["melody"] = midi_bytes(_midi(notes, paper=paper))
lab_notes = [[f"{a:.6f}", f"{b:.6f}", p, t] for a, b, p, t in notes] if paper else notes
keep_rows("melody_full.lab", lab_notes)
for track, name in ((0, "vocal"), (1, "instrumental")):
selected = [n for n in notes if n[3] == track]
keep_rows(f"melody_{name}.lab", [n[:3] for n in lab_notes if n[3] == track])
midis[f"melody_{name}"] = midi_bytes(_midi(selected, paper=paper))
intervals = {}
for field in ("chord", "key", "structure"):
rows = interval_rows(events, field, duration)
if paper:
rows = [[float(e["time"]), None, e["values"][field]] for e in events if field in e["values"]]
for i, row in enumerate(rows):
row[1] = max(row[0], rows[i + 1][0]) if i + 1 < len(rows) else float(duration)
intervals[field] = rows
keep_rows(f"{field}.lab", rows)
if paper:
rows = []
for event in events:
rhythm, stamp = event["values"].get("rhythm"), event["values"].get("timestamp")
if rhythm is None and stamp is None:
continue
meter = rhythm.get("meter") if isinstance(rhythm, dict) else None
eighth = rhythm.get("eighth_position") if isinstance(rhythm, dict) else None
rows.append([f"{event['time']:.6f}", "" if stamp is None else f"{float(stamp):.6f}",
"" if eighth is None else int(eighth), "" if meter is None else f"{meter[0]}/{meter[1]}"])
keep_rows("beat_meter.lab", rows)
raw_rhythm = [[e["time"], json.dumps(e["values"].get("rhythm", {}))]
for e in events if "rhythm" in e["values"] or "timestamp" in e["values"]]
keep_rows("rhythm_events.lab", raw_rhythm)
diagnostics, abc_error, score, text = [], None, None, None
try:
beats = rhythm_rows(events)
keep_rows("beat.lab", beats)
keep_rows("downbeat.lab", [[r[0]] for r in beats if r[1] == 1])
if len(beats) < 2:
raise ValueError("At least two decoded beats are required for ABC")
# canonical notation's last beat is a boundary, so append it explicitly. Continue the
# final tempo only as far as the audio/last note and let it pad the bar.
abc_beats = [list(b) for b in beats]
period = float(np.median(np.diff([b[0] for b in beats[-9:]])))
if period <= 0:
raise ValueError("Decoded beats must increase in time")
end = max(duration, max((n[1] for n in notes), default=0))
while abc_beats[-1][0] < end - 1e-6:
prev = abc_beats[-1]
abc_beats.append([prev[0] + period, prev[1] % prev[2] + 1, prev[2], prev[3]])
keep_rows("notation/song_beats.txt", abc_beats)
notation_intervals = {}
for field, suffix in (("chord", "chords"), ("key", "keys"), ("structure", "structures")):
rows = interval_rows(events, field, duration)
# Clip to the actual beat domain: events beyond it have no ABC bin.
rows = [[max(abc_beats[0][0], a), min(abc_beats[-1][0], b), v]
for a, b, v in rows if b > abc_beats[0][0] and a < abc_beats[-1][0]]
notation_intervals[field] = rows
keep_rows(f"notation/song_{suffix}.txt", rows)
if not interval_rows(events, "key", duration):
raise ValueError("No key was decoded; cannot construct a keyed ABC score")
clean, adjustments = notation_notes(notes)
diagnostics.extend(adjustments)
notation_midis["notation/song_melody.mid"] = midi_bytes(_midi(clean))
text, score = generate_abc_from_data(
notation_midis["notation/song_melody.mid"], abc_beats,
notation_intervals["chord"], notation_intervals["key"], notation_intervals["structure"],
melody_only=melody_only,
)
diagnostics.extend(score.diagnostics)
texts["score.abc"] = text
abc_measures = len(score.measures)
except (ValueError, FileNotFoundError) as exc:
abc_error = str(exc)
abc_measures = 0
playback, playback_midis = build_playback(
midis["melody"], [] if melody_only else intervals["chord"], score, duration)
midis.update(playback_midis)
texts["playback.json"] = json.dumps(playback, indent=2)
diagnostics.extend(playback["warnings"])
if output_dir is not None:
output_dir = Path(output_dir)
output_dir.mkdir(parents=True, exist_ok=True)
for name, content in texts.items():
atomic_write_text(output_dir / name, content)
for name, content in {**{f"{name}.mid": content for name, content in midis.items()}, **notation_midis}.items():
path = output_dir / name
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(content)
if abc_error is not None:
(output_dir / "score.abc").unlink(missing_ok=True)
else:
(output_dir / "score.browser.abc").unlink(missing_ok=True)
payload = dict(abc=text, midi=midis["transcription"], midis=midis,
events=events, labs=labs, playback=playback)
return dict(melody_notes=len(notes), vocal_notes=sum(n[3] == 0 for n in notes),
instrumental_notes=sum(n[3] == 1 for n in notes), events=len(events),
abc_measures=abc_measures, abc_error=abc_error, diagnostics=diagnostics, payload=payload)
+634
View File
@@ -0,0 +1,634 @@
"""Prompt grammar, cached generation and overlap context from the evaluated model."""
import copy
from contextlib import nullcontext
from pathlib import Path
import numpy as np
import torch
FULL_TASK_PROMPTS = ("timestamp", "downbeat_meter", "structure", "key", "chord_full", "melody_full")
FIELD_TO_INDEX = {"timestamp": 0, "rhythm": 1, "structure": 2, "key": 3, "chord": 4, "melody": 5}
class PromptGrammarState:
def __init__(self, tokenizer):
self.tokenizer = tokenizer
self.generated_events = 0
self.in_shift = True
self.shift_run = 0
self.payload_count = 0
self.last_field_index = -1
self.incomplete = None
def _allow_field_starts(self, allowed):
tokenizer = self.tokenizer
if self.last_field_index < FIELD_TO_INDEX["timestamp"]:
allowed[tokenizer.time_token_start : tokenizer.time_token_end] = True
if self.last_field_index < FIELD_TO_INDEX["rhythm"]:
allowed[tokenizer.meter_token_start : tokenizer.meter_token_end] = True
allowed[
tokenizer.eighth_position_token_start : tokenizer.eighth_position_token_end
] = True
if self.last_field_index < FIELD_TO_INDEX["structure"]:
allowed[
tokenizer.structure_token_start : tokenizer.structure_token_end
] = True
if self.last_field_index < FIELD_TO_INDEX["key"]:
allowed[tokenizer.key_token_start : tokenizer.key_token_end] = True
if self.last_field_index < FIELD_TO_INDEX["chord"]:
allowed[
tokenizer.full_chord_token_start : tokenizer.full_chord_token_end
] = True
if self.last_field_index <= FIELD_TO_INDEX["melody"]:
allowed[tokenizer.pitch_token_start : tokenizer.pitch_token_end] = True
def allowed(self, device):
tokenizer = self.tokenizer
allowed = torch.zeros(tokenizer.n_tokens, dtype=torch.bool, device=device)
can_end = self.payload_count > 0
if can_end:
allowed[tokenizer.eos_token] = True
if self.payload_count > 0 or self.in_shift:
if self.shift_run < 4:
allowed[
tokenizer.subbeat_shift_token_start : tokenizer.subbeat_shift_token_end
] = True
if self.incomplete == "rhythm_after_meter":
allowed[
tokenizer.eighth_position_token_start : tokenizer.eighth_position_token_end
] = True
return allowed
if self.incomplete == "melody_after_pitch":
allowed[tokenizer.duration_token_start : tokenizer.duration_token_end] = True
allowed[tokenizer.pitch_token_start : tokenizer.pitch_token_end] = True
return allowed
self._allow_field_starts(allowed)
return allowed
def update(self, token):
tokenizer = self.tokenizer
token = int(token)
token_type = tokenizer.token_type(token)
if token == tokenizer.eos_token:
return True
if token_type == "subbeat_shift":
if not self.in_shift and self.payload_count > 0:
self.generated_events += 1
self.payload_count = 0
self.last_field_index = -1
self.incomplete = None
self.in_shift = True
self.shift_run += 1
return False
self.in_shift = False
self.shift_run = 0
self.payload_count += 1
if token_type == "time":
self.last_field_index = FIELD_TO_INDEX["timestamp"]
self.incomplete = None
elif token_type == "meter":
self.last_field_index = FIELD_TO_INDEX["rhythm"]
self.incomplete = "rhythm_after_meter"
elif token_type == "eighth_position":
self.last_field_index = FIELD_TO_INDEX["rhythm"]
self.incomplete = None
elif token_type == "structure":
self.last_field_index = FIELD_TO_INDEX["structure"]
self.incomplete = None
elif token_type == "key":
self.last_field_index = FIELD_TO_INDEX["key"]
self.incomplete = None
elif token_type == "chord_full":
self.last_field_index = FIELD_TO_INDEX["chord"]
self.incomplete = None
elif token_type == "pitch":
self.last_field_index = FIELD_TO_INDEX["melody"]
self.incomplete = "melody_after_pitch"
elif token_type == "duration":
self.last_field_index = FIELD_TO_INDEX["melody"]
self.incomplete = None
else:
raise RuntimeError(f"Unexpected prompt token type {token_type!r}")
return False
def inference_autocast(device, dtype=torch.bfloat16):
if device.type != "cuda" or dtype is None:
return nullcontext()
return torch.autocast(device_type="cuda", dtype=dtype)
def select_cache_batch(cache, indices, batch_size):
if cache is None:
return None
batch_select = getattr(cache, "batch_select_indices", None)
if callable(batch_select):
batch_select(indices)
return cache
if torch.is_tensor(cache):
if cache.ndim > 0 and cache.shape[0] == batch_size:
return cache.index_select(0, indices)
return cache
if isinstance(cache, tuple):
return tuple(select_cache_batch(item, indices, batch_size) for item in cache)
if isinstance(cache, list):
return [select_cache_batch(item, indices, batch_size) for item in cache]
if isinstance(cache, dict):
return {
key: select_cache_batch(value, indices, batch_size)
for key, value in cache.items()
}
return cache
@torch.inference_mode()
def constrained_prompt_generate_batch(
model,
audio,
prompts,
max_sequence_length,
prefix_tokens=None,
autocast_dtype=torch.bfloat16,
stop_time_seconds=None,
progress_callback=None,
memory=None,
step_callback=None,
):
tokenizer = model.tokenizer
prefix = list(prefix_tokens) if prefix_tokens is not None else tokenizer.prompt_prefix(prompts)
if not prefix or prefix[0] != tokenizer.sos_token:
raise ValueError("generation prefix must begin with <|sos|>")
if prefix[-1] == tokenizer.eos_token:
prefix = prefix[:-1]
if audio.ndim != 2 or audio.shape[0] < 1:
raise ValueError(f"audio must have shape [batch, samples], got {tuple(audio.shape)}")
states = [PromptGrammarState(tokenizer) for _ in range(audio.shape[0])]
stop_times = [
None if stop_time_seconds is None else float(stop_time_seconds)
for _ in states
]
out_index = prefix.index(tokenizer.out_token)
for state in states:
for token in prefix[out_index + 1 :]:
state.update(token)
output_tokens = [list(prefix) for _ in states]
active_sample_ids = list(range(audio.shape[0]))
decoder_input = torch.tensor(
[prefix] * audio.shape[0],
dtype=torch.long,
device=audio.device,
)
past_key_values = None
if memory is None:
with inference_autocast(audio.device, autocast_dtype):
memory = model.encode(audio)
current_length = len(prefix)
while active_sample_ids and current_length < int(max_sequence_length):
with inference_autocast(audio.device, autocast_dtype):
logits, past_key_values = model.decode(
memory,
decoder_input,
use_cache=True,
past_key_values=past_key_values,
)
next_logits = logits[:, -1].float()
allowed = torch.stack(
[state.allowed(next_logits.device) for state in states],
dim=0,
)
masked_logits = next_logits.masked_fill(~allowed, float("-inf"))
next_tokens = masked_logits.argmax(dim=-1)
if step_callback is not None:
step_callback(current_length, active_sample_ids, next_logits, masked_logits)
next_token_values = next_tokens.tolist()
keep_positions = []
for position, (sample_id, state, stop_time, token) in enumerate(
zip(active_sample_ids, states, stop_times, next_token_values)
):
output_tokens[sample_id].append(int(token))
finished = state.update(token)
if (
not finished
and stop_time is not None
and tokenizer.time_token_start <= int(token) < tokenizer.time_token_end
and tokenizer.token_to_time_id(token) / tokenizer.time_hz >= stop_time
):
output_tokens[sample_id].append(tokenizer.eos_token)
finished = True
if not finished:
keep_positions.append(position)
current_length += 1
if progress_callback is not None and current_length % 64 == 0:
progress_callback(current_length)
if not keep_positions:
break
old_batch_size = len(active_sample_ids)
decoder_input = next_tokens[:, None]
if len(keep_positions) == old_batch_size:
continue
keep = torch.tensor(keep_positions, dtype=torch.long, device=audio.device)
active_sample_ids = [active_sample_ids[position] for position in keep_positions]
states = [states[position] for position in keep_positions]
stop_times = [stop_times[position] for position in keep_positions]
memory = memory.index_select(0, keep)
past_key_values = select_cache_batch(past_key_values, keep, old_batch_size)
decoder_input = decoder_input.index_select(0, keep)
for tokens in output_tokens:
if tokens[-1] != tokenizer.eos_token:
tokens.append(tokenizer.eos_token)
return [torch.tensor(tokens, dtype=torch.long) for tokens in output_tokens]
@torch.inference_mode()
def constrained_prompt_generate(
model,
audio,
prompts,
max_sequence_length,
prefix_tokens=None,
autocast_dtype=torch.bfloat16,
stop_time_seconds=None,
progress_callback=None,
memory=None,
step_callback=None,
):
if audio.shape[0] != 1:
raise ValueError(
"constrained_prompt_generate expects batch size 1; "
"use constrained_prompt_generate_batch for batched inference"
)
return constrained_prompt_generate_batch(
model,
audio,
prompts,
max_sequence_length,
prefix_tokens=prefix_tokens,
autocast_dtype=autocast_dtype,
stop_time_seconds=stop_time_seconds,
progress_callback=progress_callback,
memory=memory,
step_callback=step_callback,
)[0]
def decode_generated_tokens(tokenizer, tokens, song_id, window_index=None):
try:
return tokenizer.decode_sequence(tokens, strict=True), None
except ValueError as exc:
recoverable_errors = (
"empty event at subbeat",
"belongs to inactive output field",
)
if not any(message in str(exc) for message in recoverable_errors):
raise
decoded = tokenizer.decode_sequence(tokens, strict=False)
warning = {
"song_id": song_id,
"warning": "strict decode failed; recovered in non-strict decode",
"error": str(exc),
}
if window_index is not None:
warning["window_index"] = int(window_index)
return decoded, warning
def event_time_map(decoded, target_seconds):
anchors = []
for event in decoded["events"]:
value = event["values"].get("timestamp")
if value is not None:
anchors.append((int(event["subbeat"]), float(value)))
if not anchors:
return lambda step: min(float(target_seconds), max(0.0, float(step) * 0.125))
anchors = sorted(dict(anchors).items())
steps = np.asarray([item[0] for item in anchors], dtype=np.float64)
times = np.asarray([item[1] for item in anchors], dtype=np.float64)
if len(anchors) >= 2:
step_seconds = float(np.median(np.diff(times) / np.maximum(np.diff(steps), 1)))
if not np.isfinite(step_seconds) or step_seconds <= 0:
step_seconds = 0.125
else:
step_seconds = 0.125
def lookup(step):
step = float(step)
if step <= steps[0]:
return float(np.clip(times[0] + (step - steps[0]) * step_seconds, 0, target_seconds))
if step >= steps[-1]:
return float(np.clip(times[-1] + (step - steps[-1]) * step_seconds, 0, target_seconds))
return float(np.interp(step, steps, times))
return lookup
def field_text(event, tokenizer):
parts = []
for field in tokenizer.event_field_order:
value = event["values"].get(field)
if value is None:
continue
if field == "melody":
note_parts = []
for note in value:
duration = note["duration_bin"]
note_parts.append(
f"pitch={note['pitch']}:track={note['track']}:dur_bin={duration}:dur_steps={note['duration_steps']}"
)
parts.append("melody=[" + ",".join(note_parts) + "]")
elif isinstance(value, dict):
parts.append(field + "=" + ",".join(f"{k}:{v}" for k, v in value.items()))
else:
parts.append(f"{field}={value}")
return "; ".join(parts)
def event_start_time(event, time_lookup):
if "time" in event:
return float(event["time"])
return float(time_lookup(event["subbeat"]))
def write_events_tsv(decoded, tokenizer, time_lookup, path):
with Path(path).open("w", encoding="utf-8") as f:
f.write("event_index\tsubbeat\ttime\tfields\ttokens\n")
for index, event in enumerate(decoded["events"]):
token_text = " ".join(
tokenizer.describe(token)
for field in tokenizer.event_field_order
for token in event["tokens_by_field"].get(field, ())
)
f.write(
f"{index}\t{event['subbeat']}\t{event_start_time(event, time_lookup):.6f}\t"
f"{field_text(event, tokenizer)}\t{token_text}\n"
)
def clone_decoded_event(event):
return {
"subbeat": int(event["subbeat"]),
"tokens_by_field": {
field: [int(token) for token in tokens]
for field, tokens in event["tokens_by_field"].items()
},
"values": copy.deepcopy(event["values"]),
}
def refresh_event_values(event, tokenizer, prompts):
event["values"] = {
field: tokenizer._decode_field(field, tokens, prompts)
for field, tokens in event["tokens_by_field"].items()
if tokens
}
def set_event_local_timestamp(event, tokenizer, local_time):
time_id = int(round(float(local_time) * tokenizer.time_hz))
time_id = max(0, min(time_id, tokenizer.n_time_tokens - 1))
event["tokens_by_field"]["timestamp"] = [tokenizer.time_id_to_token(time_id)]
def active_context_before(events, tokenizer, time_abs):
state = {}
for event in events:
event_time = event.get("time")
if event_time is None or float(event_time) > float(time_abs) + 1e-6:
continue
for field in ("structure", "key", "chord"):
tokens = event["tokens_by_field"].get(field)
if tokens:
state[field] = [int(token) for token in tokens]
rhythm_tokens = event["tokens_by_field"].get("rhythm", ())
meter_tokens = [
int(token)
for token in rhythm_tokens
if tokenizer.token_type(token) == "meter"
]
if meter_tokens:
state["meter"] = meter_tokens[:1]
return state
def apply_prefix_context(event, context, tokenizer, prompts):
for field in ("structure", "key", "chord"):
if field not in event["tokens_by_field"] and field in context:
event["tokens_by_field"][field] = list(context[field])
rhythm_tokens = list(event["tokens_by_field"].get("rhythm", ()))
has_meter = any(tokenizer.token_type(token) == "meter" for token in rhythm_tokens)
has_eighth = any(
tokenizer.token_type(token) == "eighth_position" for token in rhythm_tokens
)
if has_eighth and not has_meter and "meter" in context:
event["tokens_by_field"]["rhythm"] = list(context["meter"]) + rhythm_tokens
refresh_event_values(event, tokenizer, prompts)
def build_overlap_prefix_tokens(
stitched_events,
tokenizer,
prompts,
window_start,
prefix_end,
):
eps = 1e-4
source_events = [
event
for event in stitched_events
if float(window_start) - eps <= float(event.get("time", -1.0)) < float(prefix_end) - eps
]
source_events.sort(
key=lambda event: (
int(
event.get(
"global_subbeat",
event.get("source_subbeat", event["subbeat"]),
)
),
float(event.get("time", 0.0)),
)
)
first_beat_index = next(
(
index
for index, event in enumerate(source_events)
if "timestamp" in event["values"] or "rhythm" in event["values"]
),
None,
)
if first_beat_index is None:
return None, None, None
source_events = source_events[first_beat_index:]
base_subbeat = int(
source_events[0].get(
"global_subbeat",
source_events[0].get("source_subbeat", source_events[0]["subbeat"]),
)
)
context = active_context_before(
stitched_events,
tokenizer,
source_events[0]["time"],
)
prefix_events = []
for source in source_events:
event = clone_decoded_event(source)
source_subbeat = int(
source.get(
"global_subbeat",
source.get("source_subbeat", source["subbeat"]),
)
)
event["subbeat"] = max(0, source_subbeat - base_subbeat)
if "timestamp" in event["tokens_by_field"]:
set_event_local_timestamp(
event,
tokenizer,
float(source["time"]) - float(window_start),
)
refresh_event_values(event, tokenizer, prompts)
prefix_events.append(event)
apply_prefix_context(prefix_events[0], context, tokenizer, prompts)
prefix_decoded = {
"schema_version": tokenizer.schema_version,
"prompts": prompts,
"events": prefix_events,
"has_eos": False,
}
return (
prefix_decoded,
tokenizer.encode_decoded_sequence(prefix_decoded),
base_subbeat,
)
def stitched_window_events(
decoded,
time_lookup,
window_start,
accept_start,
accept_end,
song_duration,
window_index,
global_subbeat_base=0,
):
accepted = []
eps = 1e-4
for event in decoded["events"]:
local_time = float(time_lookup(event["subbeat"]))
abs_time = float(window_start) + local_time
if abs_time < float(accept_start) - eps:
continue
if abs_time >= float(accept_end) - eps or abs_time >= float(song_duration) - eps:
continue
output = clone_decoded_event(event)
output["time"] = float(np.clip(abs_time, 0.0, song_duration))
output["window_index"] = int(window_index)
output["window_start"] = float(window_start)
output["source_subbeat"] = int(event["subbeat"])
output["global_subbeat"] = int(global_subbeat_base) + int(event["subbeat"])
if "timestamp" in output["values"]:
output["values"]["timestamp"] = output["time"]
notes = output["values"].get("melody")
if notes is not None:
fixed_notes = []
for note in notes:
fixed_note = copy.deepcopy(note)
duration_steps = int(fixed_note["duration_steps"])
local_end = float(time_lookup(int(event["subbeat"]) + duration_steps))
end_time = float(window_start) + local_end
end_time = min(
float(song_duration),
max(output["time"] + 0.04, end_time),
)
fixed_note["end_time"] = end_time
fixed_notes.append(fixed_note)
output["values"]["melody"] = fixed_notes
accepted.append(output)
return accepted
def write_window_tokens(path, windows, tokenizer):
with Path(path).open("w", encoding="utf-8") as f:
for window in windows:
f.write(
f"# window_index={window['window_index']} "
f"start={window['start']:.6f} end={window['end']:.6f} "
f"prefix_end={window['prefix_end']:.6f} "
f"accept=[{window['accept_start']:.6f},{window['accept_end']:.6f}) "
f"generation_stop={window['generation_stop']} "
f"prefix_tokens={window['prefix_tokens']} "
f"tokens={int(window['tokens'].numel())}\n"
)
for index, token in enumerate(window["tokens"].tolist()):
f.write(f"{index}\t{token}\t{tokenizer.describe(token)}\n")
f.write("\n")
@torch.inference_mode()
def generate(model, input_values, prompts=None, *, attention_mask=None, max_length=None,
max_new_tokens=None, prefix_tokens=None,
autocast_dtype=torch.bfloat16, stop_time_seconds=None,
return_dict_in_generate=False, output_logits=False, output_scores=False,
progress_callback=None):
"""Generate symbolic token sequences with the model's event grammar.
Inputs are 24 kHz waveforms [batch, samples]. Scores/logits are per generated
step, after the prompt prefix, with completed batch items filled by NaNs.
"""
from transformers.utils import ModelOutput
if input_values.ndim == 1:
input_values = input_values.unsqueeze(0)
if attention_mask is not None and attention_mask.ndim == 1:
attention_mask = attention_mask.unsqueeze(0)
input_values = input_values.to(next(model.parameters()).device)
prompts = model.tokenizer.normalize_prompts(prompts or FULL_TASK_PROMPTS)
prefix = list(prefix_tokens) if prefix_tokens is not None else model.tokenizer.prompt_prefix(prompts)
prefix_length = len(prefix) - (1 if prefix and prefix[-1] == model.tokenizer.eos_token else 0)
if max_new_tokens is not None:
if max_length is not None:
raise ValueError("Specify max_new_tokens or max_length, not both.")
if not isinstance(max_new_tokens, int) or isinstance(max_new_tokens, bool) or max_new_tokens < 1:
raise ValueError("max_new_tokens must be a positive integer.")
max_length = prefix_length + max_new_tokens
limit = model.max_output_seq_len if max_length is None else max_length
if not isinstance(limit, int) or isinstance(limit, bool) or not prefix_length < limit <= model.max_output_seq_len:
raise ValueError("The token limit must exceed the prefix length and fit the decoder context.")
memory = None
if attention_mask is not None:
with inference_autocast(input_values.device, autocast_dtype):
memory = model.get_audio_features(input_values, attention_mask=attention_mask).encoder_last_hidden_state
raw, scores, positions = [], [], []
batch_size = input_values.shape[0]
def capture(position, ids, logits, masked):
positions.append(position)
for enabled, values, output in ((output_logits, logits, raw), (output_scores, masked, scores)):
if enabled:
row = torch.full((batch_size, values.shape[-1]), float('nan'), dtype=torch.float32)
row[ids] = values.detach().cpu()
output.append(row)
tokens = constrained_prompt_generate_batch(
model, input_values, prompts, limit,
prefix_tokens=prefix_tokens, autocast_dtype=autocast_dtype,
stop_time_seconds=stop_time_seconds, progress_callback=progress_callback,
step_callback=capture if output_logits or output_scores else None,
memory=memory,
)
sequences = torch.nn.utils.rnn.pad_sequence(tokens, batch_first=True, padding_value=model.tokenizer.pad_token)
if return_dict_in_generate or output_logits or output_scores:
return ModelOutput(sequences=sequences, logits=tuple(raw) if output_logits else None,
scores=tuple(scores) if output_scores else None,
token_positions=torch.tensor(positions, dtype=torch.long))
return sequences
+78
View File
@@ -0,0 +1,78 @@
"""Transcribe a music recording with SheetSage2."""
import argparse
from pathlib import Path
import sys
SCRIPT_DIR = Path(__file__).absolute().parent
sys.path.insert(0, str(SCRIPT_DIR))
def main():
parser = argparse.ArgumentParser(description="SheetSage2: music audio to ABC, MIDI, annotations and features")
parser.add_argument("audio", type=Path)
parser.add_argument("--output", type=Path, default=Path("output"))
local = SCRIPT_DIR
parser.add_argument("--model", default=str(local) if (local / "config.json").exists() else "m-a-p/SheetSage2")
parser.add_argument("--revision", help="Model and code commit on Hugging Face")
parser.add_argument("--device", default="auto")
parser.add_argument("--dtype", choices=("bf16", "fp32"), default="bf16")
parser.add_argument("--preset", choices=("default", "paper"), default="default")
parser.add_argument("--max-seconds", type=float)
parser.add_argument("--overlap", type=float)
parser.add_argument("--lookahead", type=float)
parser.add_argument("--prompts", nargs="+")
parser.add_argument("--melody-only", action="store_true",
help="Keep vocal and instrumental melodies; omit chords from ABC and playback")
parser.add_argument("--export-logits", action="store_true")
parser.add_argument("--export-scores", action="store_true")
parser.add_argument("--export-embeddings", action="store_true")
parser.add_argument("--all-layers", action="store_true", help="Also export all 24 MERT block features")
parser.add_argument("--render-audio", action="store_true")
parser.add_argument("--render-score", nargs="?", const="pdf", default=False, help="pdf,svg,png (default: pdf)")
parser.add_argument("--render-parts", default="mix", help="mix,melody,vocal,instrumental,chords,all")
parser.add_argument("--local-files-only", action="store_true")
args = parser.parse_args()
if args.render_audio or args.render_score:
from rendering_sheetsage2 import validate_render_options
try:
validate_render_options(audio=args.render_audio, score=args.render_score, parts=args.render_parts)
except ValueError as exc:
parser.error(str(exc))
import torch
from transformers import AutoModel
torch.set_num_threads(min(4, torch.get_num_threads()))
device = "cuda" if torch.cuda.is_available() else "cpu" if args.device == "auto" else args.device
if args.device != "auto":
device = args.device
model = AutoModel.from_pretrained(
args.model, revision=args.revision, code_revision=args.revision,
local_files_only=args.local_files_only, trust_remote_code=True,
).eval().to(device)
def progress(value):
if value["stage"] == "encoding":
print(f"Window {value['window']}/{value['windows']}", flush=True)
options = dict(dtype=args.dtype, preset=args.preset, max_seconds=args.max_seconds,
overlap_seconds=args.overlap, lookahead_seconds=args.lookahead,
export_logits=args.export_logits, export_scores=args.export_scores,
export_embeddings=args.export_embeddings, output_hidden_states=args.all_layers,
render_audio=args.render_audio, render_score=args.render_score,
render_parts=tuple(args.render_parts.split(",")), progress=progress)
if args.prompts:
options["prompts"] = args.prompts
if args.melody_only:
options["melody_only"] = True
try:
result = model.transcribe(args.audio, output_dir=args.output, **options)
except (ValueError, RuntimeError, FileNotFoundError) as exc:
parser.exit(1, f"SheetSage2: {exc}\n")
if args.melody_only and (result.get("abc_error") or not result.get("abc")):
reason = result.get("abc_error") or "no ABC score was produced"
parser.exit(1, f"SheetSage2: melody-only ABC unavailable: {reason}. "
f"Transcription outputs were saved to {args.output.resolve()}.\n")
print(f"Saved transcription to {args.output.resolve()}")
if result.get("abc_error"):
print(f"ABC unavailable: {result['abc_error']}. MIDI and annotations were saved.")
if __name__ == "__main__":
main()
+67
View File
@@ -0,0 +1,67 @@
import os
import stat
import tempfile
from contextlib import contextmanager
from pathlib import Path
def _target_path(path) -> Path:
target = Path(path)
target.parent.mkdir(parents=True, exist_ok=True)
return target
def _flush_path(path: Path) -> None:
with path.open("rb+") as handle:
os.fsync(handle.fileno())
def _replacement_mode(target: Path) -> int:
try:
return stat.S_IMODE(target.stat().st_mode)
except FileNotFoundError:
mask = os.umask(0)
os.umask(mask)
return 0o666 & ~mask
@contextmanager
def atomic_output_path(path):
target = _target_path(path)
mode = _replacement_mode(target)
fd, tmp_name = tempfile.mkstemp(
# Keep the temporary basename independent of the final basename. Some
# annotation IDs are already close to NAME_MAX; repeating the target
# name here made an otherwise valid final path fail with ENAMETOOLONG.
prefix=".tmp_",
suffix=".tmp",
dir=str(target.parent),
)
os.close(fd)
tmp_path = Path(tmp_name)
try:
yield str(tmp_path)
os.chmod(tmp_path, mode)
_flush_path(tmp_path)
os.replace(tmp_path, target)
except Exception:
try:
tmp_path.unlink()
except FileNotFoundError:
pass
raise
def atomic_write_text(path, text: str, encoding: str = "utf-8", newline: str = "\n") -> str:
with atomic_output_path(path) as tmp_path:
with open(tmp_path, "w", encoding=encoding, newline=newline) as handle:
handle.write(text)
handle.flush()
os.fsync(handle.fileno())
return str(path)
def atomic_write_pretty_midi(midi, path) -> str:
with atomic_output_path(path) as tmp_path:
midi.write(tmp_path)
return str(path)
+29
View File
@@ -0,0 +1,29 @@
STRUCTURE_LABELS = ['silence', 'intro', 'outro', 'verse', 'chorus', 'bridge', 'pre-chorus', 'post-chorus',
'interlude', 'fade-out', 'loop', 'rap', 'preshot', 'irregular']
STRUCTURE_LABELS = ['silence', 'intro', 'outro', 'verse', 'chorus', 'bridge', 'pre-chorus', 'post-chorus',
'interlude', 'fade-out', 'loop', 'rap', 'preshot', 'irregular', 'instrumental',
'intro and verse', 'pre-chorus and chorus', 'verse and pre-chorus', 'solo', 'theme', 'development', 'variation', 'pre-outro']
MIREX_STRUCTURE_LABEL_REDUCTION = {
'silence': 'silence',
'intro': 'intro',
'outro': 'outro',
'verse': 'verse',
'chorus': 'chorus',
'bridge': 'bridge',
'pre-chorus': 'verse',
'post-chorus': 'verse',
'interlude': 'inst',
'inst': 'inst',
'fade-out': 'outro',
'loop': 'chorus',
'rap': 'verse',
'preshot': 'inst',
'irregular': 'verse',
}
+92
View File
@@ -0,0 +1,92 @@
"""Original-time MIDI playback and a separate score-to-audio measure map."""
import json
from io import BytesIO
from pathlib import Path
import mir_eval.chord
import numpy as np
import pretty_midi
from .io_sheetsage2 import atomic_write_text
def chord_pitches(label):
if label in {"N", "X", "?"}:
return []
root, bitmap, bass = mir_eval.chord.encode(label, reduce_extended_chords=True)
if root < 0:
return []
# Keep inversion bass below the chord; no ABC chord-name reinterpretation.
upper = [48 + root + int(interval) for interval in np.flatnonzero(bitmap > 0)]
return sorted(set([36 + (root + bass) % 12] + upper))
def measure_map(score, duration):
rows, position = [], 0.0
for measure in score.measures:
length = measure.abc_numerator / measure.abc_denominator
actual = measure.numerator / measure.denominator
start = float(score.beats[measure.start_beat].time)
end = float(score.beats[measure.end_beat].time)
rows.append(dict(index=measure.index, start=start, end=min(end, duration),
score_start=position, score_end=position + length,
leading_rest=length - actual if measure.pad_before else 0.0,
trailing_rest=length - actual if not measure.pad_before else 0.0))
position += length
return rows
def midi_bytes(midi):
"""Serialize a PrettyMIDI object without touching the filesystem."""
stream = BytesIO()
midi.write(stream)
return stream.getvalue()
def build_playback(melody_midi, chord_rows, score, duration):
"""Return playback metadata and MIDI bytes, preserving MIDI tick rounding."""
if isinstance(melody_midi, pretty_midi.PrettyMIDI):
melody_midi = midi_bytes(melody_midi)
midi = pretty_midi.PrettyMIDI(BytesIO(melody_midi))
chord_track = pretty_midi.Instrument(0, name="Chords")
warnings = []
for start, end, label in chord_rows:
start, end = max(0.0, float(start)), min(duration, float(end))
try:
pitches = chord_pitches(label)
except mir_eval.chord.InvalidChordException as exc:
warnings.append(f"Chord playback skipped {label}: {exc}")
continue
# Rearticulate long chord spans at downbeats, including repeated bars.
downbeats = [b.time for b in score.beats if b.beat_id == 1] if score is not None else []
cuts = [start] + [time for time in downbeats if start < time < end] + [end]
for a, b in zip(cuts, cuts[1:]):
if b <= a:
continue
for pitch in pitches:
chord_track.notes.append(pretty_midi.Note(48, pitch, a, b))
midi.instruments.append(chord_track)
transcription = midi_bytes(midi)
chords = pretty_midi.PrettyMIDI(resolution=midi.resolution)
chords.instruments.append(chord_track)
midis = {"transcription": transcription, "chords": midi_bytes(chords)}
# Decode the serialized bytes, so playback follows the returned MIDI exactly.
midi = pretty_midi.PrettyMIDI(BytesIO(transcription))
tracks = [dict(name=instrument.name, program=int(instrument.program),
notes=[dict(pitch=int(n.pitch), start=float(n.start), end=float(n.end), velocity=int(n.velocity))
for n in instrument.notes]) for instrument in midi.instruments]
data = dict(version=1, duration=duration, midi="transcription.mid", tracks=tracks,
measures=measure_map(score, duration) if score is not None else [], warnings=warnings)
return data, midis
def export_playback(directory, score, duration):
"""Write playback files for an existing directory of exported annotations."""
directory = Path(directory)
chord_path = directory / "chord.lab"
rows = [line.split(maxsplit=2) for line in chord_path.read_text(encoding="utf-8").splitlines()] if chord_path.exists() else []
data, midis = build_playback((directory / "melody.mid").read_bytes(), rows, score, duration)
for name, content in midis.items():
(directory / f"{name}.mid").write_bytes(content)
atomic_write_text(directory / "playback.json", json.dumps(data, indent=2))
return data
+361
View File
@@ -0,0 +1,361 @@
"""Standalone MERT2 waveform-to-representation inference."""
from dataclasses import dataclass
from typing import Optional, Tuple
import torch
from torch import nn
from torch.nn import functional as F
from torchaudio.transforms import AmplitudeToDB, MelScale, Spectrogram
from transformers import PreTrainedModel
from transformers.utils import ModelOutput
from .configuration_mert2 import MERT2Config
@dataclass
class MERT2ModelOutput(ModelOutput):
"""Frame representations and their validity mask.
``hidden_states`` contains one tensor per Conformer block, without an input
embedding. ``feature_attention_mask`` is True at valid output frames.
"""
last_hidden_state: Optional[torch.FloatTensor] = None
hidden_states: Optional[Tuple[torch.FloatTensor, ...]] = None
feature_attention_mask: Optional[torch.BoolTensor] = None
class MERT2MelFrontend(nn.Module):
"""Power log-mel features with fixed, checkpoint-specific normalization."""
def __init__(self, config):
super().__init__()
# Filter construction requires real tensors even during meta loading.
with torch.device("cpu"):
self.register_buffer("mel_mean", torch.zeros(config.num_mel_bins, dtype=torch.float32))
self.register_buffer("mel_std", torch.ones(config.num_mel_bins, dtype=torch.float32))
self.spectrogram = Spectrogram(
n_fft=config.n_fft,
win_length=config.win_length,
hop_length=config.hop_length,
power=2.0,
).float()
self.mel_scale = MelScale(
n_mels=config.num_mel_bins,
sample_rate=config.sampling_rate,
n_stft=config.n_fft // 2 + 1,
).float()
self.amplitude_to_db = AmplitudeToDB(stype="power", top_db=None)
def _apply(self, fn, recurse=True):
# Preserve the original float32 values, including on model.half().
saved = [(module, name, value) for module in self.modules() for name, value in module._buffers.items() if value is not None]
super()._apply(fn, recurse=recurse)
for module, name, original in saved:
current = module._buffers[name]
if original.is_floating_point():
module._buffers[name] = (
current.float() if original.is_meta else original.to(device=current.device, dtype=torch.float32)
)
return self
@torch.no_grad()
def forward(self, waveform):
with torch.autocast(device_type=waveform.device.type, enabled=False):
spectrum = self.spectrogram(waveform.float())
mel = self.amplitude_to_db(self.mel_scale(spectrum))
mel = mel[..., :-1].transpose(-1, -2)
return (mel - self.mel_mean) / self.mel_std.clamp_min(1e-5)
class Transpose(nn.Module):
def forward(self, hidden_states):
return hidden_states.transpose(1, 2)
class GlobalResponseNorm(nn.Module):
def __init__(self, dim):
super().__init__()
self.weight = nn.Parameter(torch.zeros(1, 1, dim))
self.bias = nn.Parameter(torch.zeros(1, 1, dim))
def forward(self, hidden_states):
magnitude = torch.norm(hidden_states, p=2, dim=1, keepdim=True)
normalized = magnitude / (magnitude.mean(dim=-1, keepdim=True) + 1e-6)
return self.weight * (hidden_states * normalized) + self.bias + hidden_states
class ConvNextLayer(nn.Module):
def __init__(self, dim, eps):
super().__init__()
self.depthwise_block = nn.Sequential(
Transpose(), nn.Conv1d(dim, dim, 7, padding=3, groups=dim), Transpose()
)
self.pointwise_block = nn.Sequential(
nn.LayerNorm(dim, eps=eps),
nn.Linear(dim, 4 * dim),
nn.GELU(),
GlobalResponseNorm(4 * dim),
nn.Linear(4 * dim, dim),
)
def forward(self, hidden_states):
return hidden_states + self.pointwise_block(self.depthwise_block(hidden_states))
class ConvNextBlock(nn.Module):
def __init__(self, in_channels, out_channels, stride, depth, eps):
super().__init__()
self.resampling_layer = (
nn.Sequential(
nn.LayerNorm(in_channels, eps=eps),
Transpose(),
nn.Conv1d(in_channels, out_channels, 2, stride=stride),
Transpose(),
)
if in_channels != out_channels or stride > 1 else nn.Identity()
)
self.convnext_layers = nn.Sequential(*[ConvNextLayer(out_channels, eps) for _ in range(depth)])
def forward(self, hidden_states):
return self.convnext_layers(self.resampling_layer(hidden_states))
class RotaryEmbedding(nn.Module):
def __init__(self, config):
super().__init__()
self.head_dim = config.hidden_size // config.num_attention_heads
self.base = config.rotary_embedding_base
with torch.device("cpu"):
inverse_frequency = 1.0 / (
self.base ** (torch.arange(0, self.head_dim, 2, dtype=torch.float32) / self.head_dim)
)
self.register_buffer("inv_freq", inverse_frequency, persistent=False)
self._sequence_length = 0
self._cache_device = None
self._cos = None
self._sin = None
def _apply(self, fn, recurse=True):
original = self.inv_freq
super()._apply(fn, recurse=recurse)
self.inv_freq = original.to(device=self.inv_freq.device, dtype=torch.float32)
return self
def forward(self, hidden_states):
length = hidden_states.shape[1]
if self._cos is None or self._cache_device != hidden_states.device or length > self._sequence_length:
positions = torch.arange(length, device=hidden_states.device, dtype=self.inv_freq.dtype)
frequencies = torch.einsum("i,j->ij", positions, self.inv_freq)
angles = torch.cat((frequencies, frequencies), dim=-1)
self._cos = angles.cos()[:, None, None, :]
self._sin = angles.sin()[:, None, None, :]
self._sequence_length = length
self._cache_device = hidden_states.device
return (
self._cos[:length].to(hidden_states.dtype).permute(1, 0, 2, 3),
self._sin[:length].to(hidden_states.dtype).permute(1, 0, 2, 3),
)
def rotate_half(value):
first, second = value.chunk(2, dim=-1)
return torch.cat((-second, first), dim=-1)
class SelfAttention(nn.Module):
def __init__(self, config):
super().__init__()
self.config = config
self.num_heads = config.num_attention_heads
self.head_dim = config.hidden_size // self.num_heads
self.query_proj = nn.Linear(config.hidden_size, config.hidden_size)
self.key_proj = nn.Linear(config.hidden_size, config.hidden_size)
self.value_proj = nn.Linear(config.hidden_size, config.hidden_size)
self.out_proj = nn.Linear(config.hidden_size, config.hidden_size)
def forward(self, hidden_states, position_embeddings):
batch, time, width = hidden_states.shape
shape = (batch, time, self.num_heads, self.head_dim)
query = self.query_proj(hidden_states).reshape(shape)
key = self.key_proj(hidden_states).reshape(shape)
value = self.value_proj(hidden_states).reshape(shape)
cos, sin = position_embeddings
query = query * cos + rotate_half(query) * sin
key = key * cos + rotate_half(key) * sin
if self.config._attn_implementation == "flash_attention_2":
if query.device.type != "cuda" or query.dtype not in (torch.float16, torch.bfloat16):
raise ValueError("flash_attention_2 requires CUDA and float16/bfloat16 activations; use autocast or load a reduced-precision model.")
try:
from flash_attn import flash_attn_func
except ImportError as error:
raise ImportError("Install flash-attn to use flash_attention_2, or select attn_implementation='sdpa'.") from error
attended = flash_attn_func(query.contiguous(), key.contiguous(), value.contiguous(), dropout_p=0.0, causal=False)
else:
attended = F.scaled_dot_product_attention(
query.transpose(1, 2), key.transpose(1, 2), value.transpose(1, 2), dropout_p=0.0, is_causal=False
).transpose(1, 2)
return self.out_proj(attended.reshape(batch, time, width))
class FeedForward(nn.Module):
def __init__(self, config):
super().__init__()
self.w_1 = nn.Linear(config.hidden_size, config.intermediate_size)
self.w_2 = nn.Linear(config.intermediate_size, config.hidden_size)
def forward(self, hidden_states):
return self.w_2(F.gelu(self.w_1(hidden_states)))
class ConvolutionModule(nn.Module):
def __init__(self, config):
super().__init__()
width = config.hidden_size
kernel = config.conv_depthwise_kernel_size
self.layer_norm = nn.LayerNorm(width, eps=config.layer_norm_eps)
self.conv_block = nn.Sequential(
Transpose(),
nn.Conv1d(width, 2 * width, 1, bias=False),
nn.GLU(dim=1),
nn.Conv1d(width, width, kernel, padding=(kernel - 1) // 2, groups=width, bias=False),
nn.Sequential(Transpose(), nn.LayerNorm(width, eps=config.layer_norm_eps), Transpose()),
nn.GELU(),
nn.Conv1d(width, width, 1, bias=False),
Transpose(),
)
def forward(self, hidden_states):
return self.conv_block(self.layer_norm(hidden_states))
class ConformerBlock(nn.Module):
def __init__(self, config):
super().__init__()
width, eps = config.hidden_size, config.layer_norm_eps
self.ffn1_layer_norm = nn.LayerNorm(width, eps=eps)
self.ffn1 = FeedForward(config)
self.attn_layer_norm = nn.LayerNorm(width, eps=eps)
self.attn = SelfAttention(config)
self.conv_module = ConvolutionModule(config)
self.ffn2_layer_norm = nn.LayerNorm(width, eps=eps)
self.ffn2 = FeedForward(config)
self.final_layer_norm = nn.LayerNorm(width, eps=eps)
def forward(self, hidden_states, position_embeddings):
hidden_states = hidden_states + 0.5 * self.ffn1(self.ffn1_layer_norm(hidden_states))
hidden_states = self.attn(self.attn_layer_norm(hidden_states), position_embeddings) + hidden_states
hidden_states = self.conv_module(hidden_states) + hidden_states
hidden_states = hidden_states + 0.5 * self.ffn2(self.ffn2_layer_norm(hidden_states))
return self.final_layer_norm(hidden_states)
class MERT2Model(PreTrainedModel):
"""Encode mono waveforms into 25 Hz MERT2 frame representations.
Inputs are floating-point mono waveforms sampled at ``config.sampling_rate``.
Do not standardize their amplitude. An optional waveform attention mask must
contain a prefix of ones followed by zeros. Without a mask, every input
sample, including any caller-provided silence or padding, is processed.
"""
config_class = MERT2Config
base_model_prefix = ""
main_input_name = "input_values"
_supports_sdpa = True
_supports_flash_attn_2 = True
_no_split_modules = ["ConvNextBlock", "ConformerBlock"]
def __init__(self, config):
super().__init__(config)
if config._attn_implementation not in {"sdpa", "flash_attention_2"}:
raise ValueError("MERT2 supports attn_implementation='sdpa' or 'flash_attention_2'.")
self.feature_extractor = MERT2MelFrontend(config)
channels = [config.num_mel_bins] + config.subsampling_channels
self.subsampling_module = nn.Sequential(*[
ConvNextBlock(channels[i], channels[i + 1], (1, 2, 2)[i], config.subsampling_depths[i], config.subsampling_layer_norm_eps)
for i in range(3)
])
self.layers = nn.ModuleList([ConformerBlock(config) for _ in range(config.num_hidden_layers)])
self.embed_positions = RotaryEmbedding(config)
self.post_init()
def _init_weights(self, module):
if isinstance(module, (nn.Linear, nn.Conv1d)):
nn.init.normal_(module.weight, mean=0.0, std=self.config.initializer_range)
if module.bias is not None:
nn.init.zeros_(module.bias)
elif isinstance(module, nn.LayerNorm):
nn.init.ones_(module.weight)
nn.init.zeros_(module.bias)
def _get_feat_extract_output_lengths(self, input_lengths):
return input_lengths // self.config.inputs_to_logits_ratio
def _encode(self, input_values, output_hidden_states):
mel = self.feature_extractor(input_values)
input_dtype = self.subsampling_module[0].convnext_layers[0].depthwise_block[1].weight.dtype
hidden = self.subsampling_module(mel.to(dtype=input_dtype))
positions = self.embed_positions(hidden)
states = [] if output_hidden_states else None
for layer in self.layers:
hidden = layer(hidden, positions)
if states is not None:
states.append(hidden)
return hidden, tuple(states) if states is not None else None
def forward(self, input_values, attention_mask=None, output_hidden_states=None, return_dict=None):
"""Return final frames, optionally all block states, and a frame mask.
Different valid waveform lengths are encoded separately so convolution
and global normalization never incorporate another sample's padding.
Outputs are zero-padded to the longest valid feature sequence.
"""
output_hidden_states = self.config.output_hidden_states if output_hidden_states is None else output_hidden_states
return_dict = self.config.use_return_dict if return_dict is None else return_dict
if input_values.ndim != 2 or not input_values.is_floating_point() or input_values.shape[0] == 0:
raise ValueError("input_values must be a nonempty floating-point tensor of shape [batch, samples].")
if input_values.shape[1] < self.config.minimum_input_samples:
raise ValueError(f"Each waveform must contain at least {self.config.minimum_input_samples} samples.")
if not torch.isfinite(input_values).all():
raise ValueError("input_values must contain only finite samples.")
batch, samples = input_values.shape
if attention_mask is None:
hidden, states = self._encode(input_values, output_hidden_states)
feature_mask = torch.ones(hidden.shape[:2], dtype=torch.bool, device=hidden.device)
else:
if attention_mask.shape != input_values.shape:
raise ValueError("attention_mask must have the same shape as input_values.")
attention_mask = attention_mask.to(device=input_values.device)
if not ((attention_mask == 0) | (attention_mask == 1)).all():
raise ValueError("attention_mask must contain only zeros and ones.")
mask = attention_mask.bool()
lengths = mask.sum(dim=1)
expected = torch.arange(samples, device=input_values.device)[None, :] < lengths[:, None]
if not torch.equal(mask, expected):
raise ValueError("attention_mask must be right padded: valid samples followed by padding.")
if (lengths < self.config.minimum_input_samples).any():
raise ValueError(f"Each waveform must contain at least {self.config.minimum_input_samples} valid samples.")
feature_lengths = self._get_feat_extract_output_lengths(lengths)
maximum = int(feature_lengths.max().item())
feature_mask = torch.arange(maximum, device=input_values.device)[None, :] < feature_lengths[:, None]
hidden, state_values = None, None
for length in torch.unique(lengths, sorted=True).tolist():
indices = torch.where(lengths == length)[0]
group_hidden, group_states = self._encode(input_values.index_select(0, indices)[:, :length], output_hidden_states)
padding = (0, 0, 0, maximum - group_hidden.shape[1])
if hidden is None:
hidden = group_hidden.new_zeros(batch, maximum, self.config.hidden_size)
if output_hidden_states:
state_values = [torch.zeros_like(hidden) for _ in self.layers]
hidden = hidden.index_copy(0, indices, F.pad(group_hidden, padding))
if state_values is not None:
for i, value in enumerate(group_states):
state_values[i] = state_values[i].index_copy(0, indices, F.pad(value, padding))
states = tuple(state_values) if state_values is not None else None
output = MERT2ModelOutput(last_hidden_state=hidden, hidden_states=states, feature_attention_mask=feature_mask)
return output if return_dict else output.to_tuple()
MERT2Model.register_for_auto_class("AutoModel")
+448
View File
@@ -0,0 +1,448 @@
"""Hugging Face SheetSage2 model with a shared MERT-v2 encoder."""
from dataclasses import dataclass
import copy
import hashlib
from pathlib import Path
import shutil
from types import SimpleNamespace
from typing import Optional, Tuple
import torch
from torch import nn
from torch.nn import functional as F
from transformers import AutoModel, BartConfig, PreTrainedModel
from transformers.models.bart.modeling_bart import BartDecoder
from transformers.utils import ModelOutput
from transformers.utils.hub import cached_file
from .configuration_mert2 import MERT2Config
from .modeling_mert2 import MERT2Model
from .configuration_sheetsage2 import SheetSage2Config
from .tokenization_sheetsage2 import SheetSage2Tokenizer
PROJECTIONS = ("query_proj", "key_proj", "value_proj", "out_proj")
BASE_CODE_HASHES = {
"configuration_mert2.py": "77b53ec9d7ee31a599d744fb006e812c7eeaf7390deb46e2f460cf8c17b00bd6",
"modeling_mert2.py": "b1a3174e5649c4b26b0c90d8626f0adacfbbba111a58ed3bb72ad651945a2f5c",
}
def _hf_relative_dependencies():
# Transformers 4.45 copies direct imports into a fresh local module cache.
from .audio_sheetsage2 import load_audio
from .durations_sheetsage2 import DURATION_TEMPLATES
from .exports_sheetsage2 import export_result
from .io_sheetsage2 import atomic_write_text
from .labels_sheetsage2 import STRUCTURE_LABELS
from .midi_sheetsage2 import export_playback
from .notation_sheetsage2 import generate_abc_from_exports
from .rendering_sheetsage2 import render_outputs
from .schema_sheetsage2 import get_prompt_multitask_schema
from .tensors_sheetsage2 import WindowTensorWriter
def _sha256(path):
digest = hashlib.sha256()
with Path(path).open("rb") as stream:
for block in iter(lambda: stream.read(8 << 20), b""):
digest.update(block)
return digest.hexdigest()
@dataclass
class SheetSage2EncoderOutput(ModelOutput):
"""Frame features. Block states exclude the separately returned input state."""
encoder_last_hidden_state: Optional[torch.FloatTensor] = None
backbone_last_hidden_state: Optional[torch.FloatTensor] = None
mixed_hidden_state: Optional[torch.FloatTensor] = None
input_hidden_state: Optional[torch.FloatTensor] = None
backbone_hidden_states: Optional[Tuple[torch.FloatTensor, ...]] = None
feature_attention_mask: Optional[torch.BoolTensor] = None
@property
def last_hidden_state(self):
return self.encoder_last_hidden_state
@dataclass
class SheetSage2Output(ModelOutput):
logits: Optional[torch.FloatTensor] = None
past_key_values: Optional[Tuple] = None
decoder_hidden_states: Optional[Tuple[torch.FloatTensor, ...]] = None
encoder_last_hidden_state: Optional[torch.FloatTensor] = None
backbone_last_hidden_state: Optional[torch.FloatTensor] = None
mixed_hidden_state: Optional[torch.FloatTensor] = None
input_hidden_state: Optional[torch.FloatTensor] = None
backbone_hidden_states: Optional[Tuple[torch.FloatTensor, ...]] = None
feature_attention_mask: Optional[torch.BoolTensor] = None
class AttentionAdapter(nn.Module):
def __init__(self, width, rank):
super().__init__()
self.lora_A = nn.Linear(width, rank, bias=False)
self.lora_B = nn.Linear(rank, width, bias=False)
class EncoderAdapters(nn.Module):
def __init__(self, config):
super().__init__()
self.layers = nn.ModuleList()
for _ in range(config.backbone_config["num_hidden_layers"]):
layer = nn.Module()
layer.attn = nn.Module()
for name in PROJECTIONS:
setattr(layer.attn, name, AttentionAdapter(config.backbone_config["hidden_size"], config.lora_rank))
self.layers.append(layer)
class SheetSage2Model(PreTrainedModel):
"""Audio-to-symbolic model, with optional frame and decoder representations.
``from_pretrained`` loads the pinned MERT-v2 parent and merges attention
adapters in float32. ``save_pretrained`` writes an independent merged model.
``forward`` returns raw vocabulary logits; grammar-masked generation scores
are available separately through ``generate(output_scores=True)``.
"""
config_class = SheetSage2Config
base_model_prefix = ""
main_input_name = "input_values"
_supports_sdpa = True
_tied_weights_keys = ["decoder.embed_tokens.weight", "output_projection.weight"]
_no_split_modules = ["ConformerBlock", "BartDecoderLayer"]
def __init__(self, config):
super().__init__(config)
self.hparams = SimpleNamespace(input_audio_length=config.input_audio_length, time_hz=config.time_hz)
self.max_output_seq_len = config.max_output_seq_len
self.tokenizer = SheetSage2Tokenizer(
config.input_audio_length, config.time_hz, config.tokenizer_schema_version,
expected_fingerprint=config.tokenizer_fingerprint,
)
if self.tokenizer.n_tokens != config.vocab_size:
raise ValueError("Tokenizer vocabulary size does not match the model.")
if config.weights_format == "adapter":
self.encoder = None
self.adapter = EncoderAdapters(config)
else:
ec = MERT2Config(**config.backbone_config)
ec._attn_implementation = config.encoder_attn_implementation
self.encoder = MERT2Model(ec)
self.adapter = None
self.layer_weight = nn.Parameter(torch.zeros(config.backbone_config["num_hidden_layers"] + 1))
self.encoder_projection = nn.Linear(config.backbone_config["hidden_size"], config.hidden_size)
dc = BartConfig(
vocab_size=config.vocab_size, d_model=config.hidden_size,
decoder_layers=config.decoder_layers, decoder_attention_heads=config.num_attention_heads,
decoder_ffn_dim=config.intermediate_size, max_position_embeddings=config.max_output_seq_len,
dropout=config.decoder_dropout, attention_dropout=config.decoder_dropout,
activation_dropout=config.decoder_dropout, activation_function="gelu",
pad_token_id=config.pad_token_id, bos_token_id=config.bos_token_id,
eos_token_id=config.eos_token_id, is_encoder_decoder=True, use_cache=True,
)
dc._attn_implementation = "sdpa"
self.token_embedding = nn.Embedding(config.vocab_size, config.hidden_size, padding_idx=config.pad_token_id)
self.decoder = BartDecoder(dc, embed_tokens=self.token_embedding)
self.decoder.gradient_checkpointing_disable()
self.output_projection = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
self.output_projection.weight = self.token_embedding.weight
self.lora_merged = config.weights_format == "merged"
self.post_init()
def get_input_embeddings(self):
return self.token_embedding
def set_input_embeddings(self, value):
self.token_embedding = value
self.decoder.embed_tokens.weight = value.weight
def tie_weights(self):
super().tie_weights()
if hasattr(self, "decoder") and hasattr(self, "token_embedding"):
self.decoder.embed_tokens.weight = self.token_embedding.weight
def get_output_embeddings(self):
return self.output_projection
def set_output_embeddings(self, value):
self.output_projection = value
def _init_weights(self, module):
if isinstance(module, (nn.Linear, nn.Embedding)):
nn.init.normal_(module.weight, mean=0.0, std=0.02)
if getattr(module, "bias", None) is not None:
nn.init.zeros_(module.bias)
if isinstance(module, nn.Embedding) and module.padding_idx is not None:
module.weight.data[module.padding_idx].zero_()
elif isinstance(module, nn.LayerNorm):
nn.init.ones_(module.weight)
nn.init.zeros_(module.bias)
@torch.no_grad()
def merge_lora(self):
if self.lora_merged:
return self
if self.encoder is None or self.adapter is None:
raise RuntimeError("Load both the MERT-v2 parent and adapters before merging.")
scale = self.config.lora_alpha / self.config.lora_rank
with torch.autocast("cpu", enabled=False):
for layer, adapter in zip(self.encoder.layers, self.adapter.layers):
for name in PROJECTIONS:
projection = getattr(layer.attn, name)
update = getattr(adapter.attn, name)
values = (projection.weight, update.lora_A.weight, update.lora_B.weight)
if any(value.device.type != "cpu" or value.dtype != torch.float32 for value in values):
raise ValueError("Merge adapters on CPU with float32 parameters.")
projection.weight.add_((update.lora_B.weight @ update.lora_A.weight) * scale)
self.adapter = None
self.lora_merged = True
self.config.weights_format = "merged"
return self
@classmethod
def from_pretrained(cls, pretrained_model_name_or_path, *model_args, **kwargs):
base_path = kwargs.pop("base_model_path", None)
requested_dtype = kwargs.pop("torch_dtype", torch.float32)
target_device = kwargs.pop("device_map", None)
backend = kwargs.pop("attn_implementation", None)
return_loading_info = kwargs.pop("output_loading_info", False)
config = kwargs.pop("config", None)
hub_keys = ("cache_dir", "force_download", "local_files_only", "token", "revision", "subfolder")
hub_args = {name: kwargs[name] for name in hub_keys if name in kwargs}
if config is None:
config = cls.config_class.from_pretrained(pretrained_model_name_or_path, **hub_args)
if backend is not None:
config.encoder_attn_implementation = backend
if config.encoder_attn_implementation not in {"sdpa", "flash_attention_2"}:
raise ValueError("Select attn_implementation='sdpa' or 'flash_attention_2'.")
if isinstance(target_device, dict):
if set(target_device) != {""}:
raise ValueError("Use a single device for SheetSage2: device_map={'': 'cuda:0'}.")
target_device = target_device[""]
if target_device == "auto":
target_device = "cuda" if torch.cuda.is_available() else "cpu"
if requested_dtype == "auto":
requested_dtype = torch.float32
if isinstance(requested_dtype, str):
requested_dtype = getattr(torch, requested_dtype, None)
if requested_dtype not in {None, torch.float32, torch.bfloat16, torch.float16}:
raise ValueError("Use torch_dtype float32, bfloat16, float16, or 'auto'.")
# Adapters must be loaded and merged before any reduced-precision cast.
model, loading_info = super().from_pretrained(
pretrained_model_name_or_path, *model_args, config=config,
torch_dtype=torch.float32, attn_implementation="sdpa", output_loading_info=True, **kwargs,
)
if any(loading_info.get(name) for name in ("missing_keys", "unexpected_keys", "mismatched_keys", "error_msgs")):
raise ValueError(f"Incomplete or incompatible SheetSage2 weights: {loading_info}")
if config.weights_format == "adapter":
parent = str(base_path or config.base_model_name_or_path)
parent_hub_args = {name: hub_args[name] for name in ("cache_dir", "force_download", "local_files_only", "token") if name in hub_args}
parent_hub_args["revision"] = config.base_model_revision
expected_files = dict(BASE_CODE_HASHES, **{"model.safetensors": config.base_model_sha256})
for filename, expected in expected_files.items():
resolved = cached_file(parent, filename, **parent_hub_args)
if _sha256(resolved) != expected:
raise ValueError(f"MERT-v2 parent integrity check failed: {filename}")
model.encoder = AutoModel.from_pretrained(
parent, trust_remote_code=True, code_revision=config.base_model_revision,
torch_dtype=torch.float32, attn_implementation=config.encoder_attn_implementation,
**parent_hub_args,
)
for name, expected in config.backbone_config.items():
if name in ("hidden_size", "intermediate_size", "num_hidden_layers", "num_attention_heads",
"sampling_rate", "hop_length", "n_fft", "win_length", "num_mel_bins",
"conv_depthwise_kernel_size", "rotary_embedding_base", "subsampling_channels",
"subsampling_depths", "layer_norm_eps", "subsampling_layer_norm_eps", "variant"):
if getattr(model.encoder.config, name) != expected:
raise ValueError(f"MERT-v2 parent architecture mismatch: {name}")
model.merge_lora()
if requested_dtype is not None:
model.to(dtype=requested_dtype)
if target_device is not None:
model.to(target_device)
# Resource lookup follows the caller's cache and offline settings. These
# transient options are never written into config or model weights.
model._hub_resource_options = {name: hub_args[name] for name in ("cache_dir", "local_files_only", "token") if name in hub_args}
resource_args = dict(model._hub_resource_options, local_files_only=True,
revision=config._commit_hash or hub_args.get("revision"),
subfolder=hub_args.get("subfolder", ""),
_raise_exceptions_for_missing_entries=False)
resource_config = cached_file(str(pretrained_model_name_or_path), "config.json", **resource_args)
model._source_snapshot = Path(resource_config).parent if resource_config else Path(pretrained_model_name_or_path)
model.eval().requires_grad_(False)
return (model, loading_info) if return_loading_info else model
def save_pretrained(self, save_directory, *args, **kwargs):
if not self.lora_merged or self.encoder is None:
raise ValueError("Load and merge the model before saving a standalone snapshot.")
original_config = self.config
source = original_config._name_or_path
self.config = copy.deepcopy(original_config)
self.config.weights_format = "merged"
self.config._name_or_path = ""
self.config.backbone_config.pop("_name_or_path", None)
try:
result = super().save_pretrained(save_directory, *args, **kwargs)
from .processing_sheetsage2 import SheetSage2Processor
SheetSage2Processor.from_model_config(self.config).save_pretrained(save_directory)
finally:
self.config = original_config
source_dir = getattr(self, "_source_snapshot", Path(source))
if source and not source_dir.is_dir():
from huggingface_hub import snapshot_download
try:
source_dir = Path(snapshot_download(source, revision=original_config._commit_hash,
cache_dir=getattr(self, "_hub_resource_options", {}).get("cache_dir"),
local_files_only=True))
except (OSError, ValueError):
source_dir = None
if source and source_dir is not None and source_dir.is_dir():
destination = Path(save_directory)
for name in ("infer.py", "render.py", "setup_render.py", "requirements.txt", "requirements-render.txt",
"LICENSE", "THIRD_PARTY_NOTICES.md"):
if (source_dir / name).is_file() and (source_dir / name).resolve() != (destination / name).resolve():
shutil.copy2(source_dir / name, destination / name)
assets = source_dir / "render_assets"
if (assets / "manifest.json").is_file() and assets.resolve() != (destination / "render_assets").resolve():
from .rendering_sheetsage2 import _verify_assets
try:
_verify_assets(assets, audio=True)
except (FileNotFoundError, ValueError):
pass
else:
shutil.copytree(assets, destination / "render_assets", dirs_exist_ok=True)
return result
def _prepare_audio(self, input_values, attention_mask=None):
if self.encoder is None or not self.lora_merged:
raise RuntimeError("Load the model with from_pretrained before inference.")
if input_values.ndim != 2 or input_values.shape[0] < 1 or not input_values.is_floating_point():
raise ValueError("input_values must be floating-point [batch, samples].")
if not torch.isfinite(input_values).all():
raise ValueError("Audio must contain only finite samples.")
batch, samples = input_values.shape
minimum = self.encoder.config.minimum_input_samples
window = round(self.config.input_audio_length * self.config.sampling_rate)
if samples < minimum:
raise ValueError(f"Each waveform must contain at least {minimum} samples.")
if samples > window:
raise ValueError("Audio exceeds one model window; use transcribe for whole songs.")
if attention_mask is None:
lengths = torch.full((batch,), samples, dtype=torch.long, device=input_values.device)
else:
if attention_mask.shape != input_values.shape:
raise ValueError("attention_mask must have the same shape as input_values.")
mask = attention_mask.to(device=input_values.device)
if not ((mask == 0) | (mask == 1)).all():
raise ValueError("attention_mask must contain zeros and ones.")
mask = mask.bool()
lengths = mask.sum(1)
expected = torch.arange(samples, device=input_values.device)[None] < lengths[:, None]
if not torch.equal(mask, expected) or (lengths < minimum).any():
raise ValueError("attention_mask must identify a nonempty right-padded waveform.")
input_values = input_values.masked_fill(~mask, 0)
# The encoder attends to the complete fixed window, including this silence.
input_values = F.pad(input_values.float(), (0, window - samples))
stride = self.encoder.config.inputs_to_logits_ratio
if window % stride:
input_values = F.pad(input_values, (0, stride - window % stride))
return input_values, lengths
def get_audio_features(self, input_values, attention_mask=None, output_hidden_states=None, return_dict=True):
output_hidden_states = self.config.output_hidden_states if output_hidden_states is None else output_hidden_states
waveform, lengths = self._prepare_audio(input_values, attention_mask)
mel = self.encoder.feature_extractor(waveform)
weight_dtype = self.encoder.subsampling_module[0].convnext_layers[0].depthwise_block[1].weight.dtype
hidden = self.encoder.subsampling_module(mel.to(dtype=weight_dtype))
input_hidden = hidden if output_hidden_states else None
weights = torch.softmax(self.layer_weight, dim=0)
mixed = hidden * weights[0]
positions = self.encoder.embed_positions(hidden)
states = [] if output_hidden_states else None
for weight, layer in zip(weights[1:], self.encoder.layers):
hidden = layer(hidden, positions)
mixed = mixed + hidden * weight
if states is not None:
states.append(hidden)
memory = self.encoder_projection(mixed)
stride = self.encoder.config.inputs_to_logits_ratio
frame_mask = torch.arange(memory.shape[1], device=memory.device)[None] < ((lengths + stride - 1) // stride)[:, None]
output = SheetSage2EncoderOutput(
encoder_last_hidden_state=memory, backbone_last_hidden_state=hidden,
mixed_hidden_state=mixed, input_hidden_state=input_hidden,
backbone_hidden_states=tuple(states) if states is not None else None,
feature_attention_mask=frame_mask,
)
return output if return_dict else output.to_tuple()
def encode(self, audio):
return self.get_audio_features(audio).last_hidden_state
def _decode(self, memory, decoder_input_ids, use_cache=False, past_key_values=None, output_hidden_states=False):
attention_mask = None if past_key_values is not None else decoder_input_ids != self.tokenizer.pad_token
output = self.decoder(
input_ids=decoder_input_ids, attention_mask=attention_mask,
encoder_hidden_states=memory, encoder_attention_mask=None,
past_key_values=past_key_values, use_cache=use_cache,
output_hidden_states=output_hidden_states, return_dict=True,
)
return self.output_projection(output.last_hidden_state), output
def decode(self, memory, decoder_input_ids, use_cache=False, past_key_values=None):
logits, output = self._decode(memory, decoder_input_ids, use_cache, past_key_values)
return logits, output.past_key_values
def forward(self, input_values=None, decoder_input_ids=None, attention_mask=None,
encoder_outputs=None, past_key_values=None, use_cache=None,
output_hidden_states=None, return_dict=None):
output_hidden_states = self.config.output_hidden_states if output_hidden_states is None else output_hidden_states
use_cache = self.config.use_cache if use_cache is None else use_cache
if decoder_input_ids is None:
raise ValueError("decoder_input_ids is required for logits; use generate to transcribe audio.")
if encoder_outputs is None:
if input_values is None:
raise ValueError("Provide input_values or encoder_outputs.")
encoder_outputs = self.get_audio_features(input_values, attention_mask, output_hidden_states)
if torch.is_tensor(encoder_outputs):
encoder_outputs = SheetSage2EncoderOutput(encoder_last_hidden_state=encoder_outputs)
logits, decoded = self._decode(encoder_outputs.last_hidden_state, decoder_input_ids,
use_cache, past_key_values, output_hidden_states)
output = SheetSage2Output(
logits=logits, past_key_values=decoded.past_key_values,
decoder_hidden_states=decoded.hidden_states,
encoder_last_hidden_state=encoder_outputs.last_hidden_state,
backbone_last_hidden_state=encoder_outputs.backbone_last_hidden_state if output_hidden_states else None,
mixed_hidden_state=encoder_outputs.mixed_hidden_state if output_hidden_states else None,
input_hidden_state=encoder_outputs.input_hidden_state if output_hidden_states else None,
backbone_hidden_states=encoder_outputs.backbone_hidden_states if output_hidden_states else None,
feature_attention_mask=encoder_outputs.feature_attention_mask,
)
return_dict = self.config.use_return_dict if return_dict is None else return_dict
return output if return_dict else output.to_tuple()
def generate(self, input_values, **kwargs):
"""Generate grammar-constrained symbolic tokens with autoregressive caching."""
from .generation_sheetsage2 import generate
return generate(self, input_values, **kwargs)
def transcribe(self, audio, output_dir=None, *, melody_only=False, **kwargs):
"""Return transcription in memory; set output_dir to also save files.
Accepts a path, encoded audio bytes, binary stream, or waveform with
sampling_rate. Returns ABC text, MIDI bytes, timed events, and optional
per-window CPU tensors. With output_dir, optional tensors are saved
instead of retained in memory. Set melody_only=True to retain both vocal
and instrumental melodies while omitting chords from ABC and playback;
raw predicted annotations remain available. The default keeps full
transcription. See the model card for rendering options.
"""
from .pipeline_sheetsage2 import transcribe
return transcribe(self, audio, output_dir=output_dir, melody_only=melody_only, **kwargs)
SheetSage2ForConditionalGeneration = SheetSage2Model
SheetSage2Model.register_for_auto_class("AutoModel")
File diff suppressed because it is too large Load Diff
+259
View File
@@ -0,0 +1,259 @@
"""Whole-song inference with cached overlap prefixes and right-hand audio context."""
from pathlib import Path
import json
import re
import time
import torch
from .io_sheetsage2 import atomic_write_text
from .audio_sheetsage2 import SAMPLE_RATE, load_audio, slice_audio
from .generation_sheetsage2 import (
FULL_TASK_PROMPTS, build_overlap_prefix_tokens, constrained_prompt_generate,
decode_generated_tokens, event_time_map, stitched_window_events, write_window_tokens,
)
from .exports_sheetsage2 import export_result
def _tensor_files(output_dir):
"""Inventory only files owned by the optional tensor exporter."""
files = set()
tensor_dir = output_dir / "tensors"
if tensor_dir.is_symlink():
return files
for directory in tensor_dir.glob("window-*"):
if re.fullmatch(r"window-\d{4,}", directory.name) and directory.is_dir() and not directory.is_symlink():
for path in directory.iterdir():
if path.is_file() and re.fullmatch(r"(?:index\.json|audio\.safetensors|(?:tokens|decoder)-\d{4,}\.safetensors)", path.name):
files.add(path)
return files
def sliding_window_plan(duration, window_seconds=300.0, overlap_seconds=200.0, lookahead_seconds=100.0):
if not duration > 0 or not window_seconds > 0:
raise ValueError("Duration and window length must be positive")
if not 0 <= lookahead_seconds <= overlap_seconds < window_seconds:
raise ValueError("Require 0 <= lookahead <= overlap < window length")
hop = window_seconds - overlap_seconds
start, accepted = 0.0, 0.0
result = []
while True:
last = start + window_seconds >= duration - 1e-6
accept_end = duration if last else start + window_seconds - lookahead_seconds
result.append(dict(start=start, end=min(duration, start + window_seconds),
accept_start=accepted, accept_end=accept_end, prefix_end=accepted,
generation_stop=None if last else window_seconds - lookahead_seconds))
if last:
return result
accepted = accept_end
start = min(start + hop, duration - window_seconds)
class Transcriber:
def __init__(self, model, dtype="bf16"):
self.model = model
self.device = next(model.parameters()).device
self.dtype = {"bf16": torch.bfloat16, "fp32": None}[dtype]
@torch.inference_mode()
def analyze(self, audio_path, output_dir=None, *, prompts=FULL_TASK_PROMPTS,
sampling_rate=None, max_seconds=None, preset="default",
overlap_seconds=None, lookahead_seconds=None, progress=None,
export_logits=False, export_scores=False, export_embeddings=False,
output_hidden_states=False, melody_only=False):
def report(stage, **fields):
if progress:
progress(dict(stage=stage, **fields))
if not isinstance(melody_only, bool):
raise ValueError("melody_only must be True or False")
if preset not in ("default", "paper"):
raise ValueError("preset must be default or paper")
defaults = (200.0, 100.0) if preset == "default" else (100.0, 0.0)
overlap_seconds = defaults[0] if overlap_seconds is None else overlap_seconds
lookahead_seconds = defaults[1] if lookahead_seconds is None else lookahead_seconds
if preset == "paper" and (overlap_seconds, lookahead_seconds) != defaults:
raise ValueError("paper preset fixes overlap=100 and lookahead=0")
started = time.monotonic()
output_dir = Path(output_dir) if output_dir is not None else None
if output_dir is not None:
output_dir.mkdir(parents=True, exist_ok=True)
previous_tensors = _tensor_files(output_dir) if output_dir is not None else set()
current_tensors = set()
tensor_results = []
prompts = self.model.tokenizer.normalize_prompts(prompts)
if "timestamp" not in prompts:
raise ValueError("timestamp is required to export timed annotations")
report("audio")
audio = load_audio(audio_path, sampling_rate=sampling_rate, max_seconds=max_seconds, preset=preset)
duration = len(audio) / SAMPLE_RATE
window_length = float(self.model.hparams.input_audio_length)
plan = sliding_window_plan(duration, window_length, overlap_seconds, lookahead_seconds)
if self.device.type == "cuda":
torch.cuda.reset_peak_memory_stats(self.device)
stitched, records, warnings = [], [], []
for index, window in enumerate(plan):
report("encoding", window=index + 1, windows=len(plan), start=window["start"])
segment = slice_audio(audio, window["start"], window_length)[None].to(self.device)
prefix, base = None, 0
if index:
_, prefix, base = build_overlap_prefix_tokens(
stitched, self.model.tokenizer, prompts, window["start"], window["prefix_end"])
if prefix is not None:
if preset == "paper" and len(prefix) >= self.model.max_output_seq_len - 16:
warnings.append(f"Window {index + 1}: overlap prefix exceeded the context")
prefix = None
elif preset != "paper" and len(prefix) >= self.model.max_output_seq_len - 128:
raise ValueError("Overlap prefix fills the context; reduce overlap_seconds")
tick = time.monotonic()
memory = None
tensor_record = None
if export_embeddings or output_hidden_states or export_logits or export_scores:
from .tensors_sheetsage2 import WindowTensorWriter
tensor_record = WindowTensorWriter(output_dir, index, window, export_logits, export_scores)
if export_embeddings or output_hidden_states:
memory = tensor_record.audio_features(self.model, segment, self.dtype, output_hidden_states)
tokens = constrained_prompt_generate(
self.model, segment, prompts, self.model.max_output_seq_len,
prefix_tokens=prefix, autocast_dtype=self.dtype,
stop_time_seconds=(None if preset == "paper" else
window["generation_stop"] if window["generation_stop"] is not None else
min(duration - window["start"], window_length)),
memory=memory,
step_callback=tensor_record.capture if tensor_record is not None and (export_logits or export_scores) else None,
progress_callback=lambda n: report("decoding", window=index + 1, windows=len(plan), tokens=n),
)
if tensor_record is not None:
if export_embeddings:
tensor_record.decoder_features(self.model, segment, tokens, self.dtype, memory)
tensor_data = tensor_record.finish()
if tensor_data is not None:
tensor_results.append(tensor_data)
if tensor_record.directory is not None:
current_tensors.update(tensor_record.directory / name for name in tensor_record.files)
current_tensors.add(tensor_record.directory / "index.json")
if len(tokens) > self.model.max_output_seq_len:
warnings.append(f"Window {index + 1} reached the token limit; inspect its token coverage")
decoded, warning = decode_generated_tokens(self.model.tokenizer, tokens, Path(audio_path).stem if isinstance(audio_path, (str, Path)) else "audio", index)
if warning:
warnings.append(warning["error"])
lookup = event_time_map(decoded, window_length)
accepted = stitched_window_events(
decoded, lookup, window["start"], window["accept_start"], window["accept_end"],
duration, index, global_subbeat_base=base or 0,
)
stitched.extend(accepted)
if preset == "paper":
stitched.sort(key=lambda e: (float(e.get("time", 0)), int(e.get("window_index", 0)),
int(e.get("source_subbeat", e["subbeat"]))))
record = dict(window, window_index=index, prefix_tokens=0 if prefix is None else len(prefix),
tokens=tokens, events=len(decoded["events"]), accepted_events=len(accepted),
elapsed_seconds=time.monotonic() - tick)
records.append(record)
report("window_complete", window=index + 1, windows=len(plan), tokens=len(tokens))
if preset != "paper":
stitched.sort(key=lambda e: (e["time"], e["global_subbeat"]))
decoded = dict(schema_version=self.model.tokenizer.schema_version,
prompts=list(prompts), events=stitched, has_eos=True)
report("notation")
exported = export_result(decoded, self.model.tokenizer, output_dir, duration,
paper=preset == "paper", melody_only=melody_only)
payload = exported.pop("payload")
if output_dir is not None:
write_window_tokens(output_dir / "tokens.txt", records, self.model.tokenizer)
atomic_write_text(output_dir / "tokens.json", json.dumps([
dict(r, tokens=r["tokens"].tolist()) for r in records
]))
result = dict(
audio=Path(audio_path).name if isinstance(audio_path, (str, Path)) else "audio",
duration_seconds=duration, prompts=list(prompts), preset=preset, melody_only=melody_only,
dtype="bf16" if self.dtype and self.device.type == "cuda" else "fp32",
window_seconds=window_length, overlap_seconds=overlap_seconds, lookahead_seconds=lookahead_seconds,
windows=[dict(r, tokens=len(r["tokens"])) for r in records],
elapsed_seconds=time.monotonic() - started, warnings=warnings,
peak_gpu_mib=torch.cuda.max_memory_allocated(self.device) / 1024**2 if self.device.type == "cuda" else 0,
**exported,
)
if output_dir is not None:
atomic_write_text(output_dir / "result.json", json.dumps(result, indent=2))
for path in previous_tensors - current_tensors:
path.unlink(missing_ok=True)
for directory in {path.parent for path in previous_tensors - current_tensors}:
if directory.is_dir() and not any(directory.iterdir()):
directory.rmdir()
report("complete", **exported)
return dict(result, num_events=result["events"], **payload,
tokens=[r["tokens"] for r in records], tensors=tensor_results,
_metadata=result)
@torch.inference_mode()
def transcribe(model, audio, output_dir=None, *, render_audio=False, render_score=False,
render_parts=("mix",), dtype="bf16", melody_only=False, **kwargs):
"""Transcribe a path, encoded audio bytes, binary stream, or waveform.
Arrays use channels-first layout and require sampling_rate. With
output_dir=None, all audio/results stay in memory: ABC is text, MIDI is
bytes, events are a list, and optional features are CPU tensors grouped by
window. A directory saves the standard output files and writes optional
tensors to disk instead of keeping them in memory. Optional
rendering also returns bytes/text when no output directory is given.
melody_only=True omits chords from ABC and MIDI playback, retaining both
melody voices and the original decoded events/LAB annotations. If the
requested melody-only ABC cannot be built, raises RuntimeError with the
completed transcription available as error.result.
"""
if not isinstance(melody_only, bool):
raise ValueError("melody_only must be True or False")
output_dir = Path(output_dir) if output_dir is not None else None
if render_audio or render_score:
from .rendering_sheetsage2 import render_outputs, render_memory, validate_render_options
formats, render_parts = validate_render_options(audio=render_audio, score=render_score, parts=render_parts)
result = Transcriber(model, dtype=dtype).analyze(audio, output_dir, melody_only=melody_only, **kwargs)
metadata = result.pop("_metadata", None)
if metadata is None:
metadata = dict(result)
if render_audio or render_score:
try:
assets = Path(__file__).with_name("render_assets")
if not assets.is_dir():
source = getattr(model, "_source_snapshot", Path(model.config._name_or_path))
if (source / "render_assets").is_dir():
assets = source / "render_assets"
else:
from huggingface_hub import snapshot_download
assets = Path(snapshot_download(model.config._name_or_path,
revision=model.config._commit_hash, allow_patterns=["render_assets/**"],
**getattr(model, "_hub_resource_options", {}))) / "render_assets"
# A missing score does not prevent a requested MIDI piano preview.
rendered = None
if render_audio or not result.get("abc_error"):
options = dict(audio=render_audio, score=() if result.get("abc_error") else formats,
parts=render_parts, assets_dir=assets)
if output_dir is None:
rendered = render_memory(midi=result["midi"], abc=result["abc"],
duration=result["duration_seconds"], **options)
else:
rendered = render_outputs(input_dir=output_dir, output_dir=output_dir, **options)
result["rendered"] = rendered
if output_dir is not None:
metadata["rendered"] = rendered
if formats and result.get("abc_error"):
raise ValueError(f"ABC unavailable: {result['abc_error']}")
except Exception as exc:
result["render_error"] = str(exc)
metadata["render_error"] = str(exc)
if output_dir is not None:
atomic_write_text(output_dir / "result.json", json.dumps(metadata, indent=2))
location = f"saved to {output_dir.resolve()}" if output_dir is not None else "completed in memory"
error = RuntimeError(f"Transcription {location}; rendering failed: {exc}")
error.result = result
raise error from exc
if output_dir is not None:
atomic_write_text(output_dir / "result.json", json.dumps(metadata, indent=2))
if melody_only and not result.get("abc"):
reason = result.get("abc_error") or "No ABC score was produced"
error = RuntimeError(f"Melody-only ABC unavailable: {reason}")
error.result = result
raise error
return result
+109
View File
@@ -0,0 +1,109 @@
"""Waveform preparation and symbolic decoding for SheetSage2."""
import numbers
import torch
import torchaudio
from transformers import ProcessorMixin
from transformers.feature_extraction_utils import BatchFeature
from .tokenization_sheetsage2 import SheetSage2Tokenizer
def _hf_relative_dependencies():
# Keep standalone AutoProcessor loading compatible with Transformers 4.45.
from .durations_sheetsage2 import DURATION_TEMPLATES
from .labels_sheetsage2 import STRUCTURE_LABELS
from .schema_sheetsage2 import get_prompt_multitask_schema
class SheetSage2Processor(ProcessorMixin):
"""Prepare mono waveforms without amplitude normalization.
Pass one mono waveform, a batch tensor, or a list of mono waveforms. Use
``model.transcribe`` for files and audio longer than one model window.
"""
attributes = []
valid_kwargs = ["sampling_rate", "window_seconds", "time_hz", "schema_version", "tokenizer_fingerprint"]
def __init__(self, sampling_rate=24000, window_seconds=300.0, time_hz=100,
schema_version="v1", tokenizer_fingerprint="5ba3325af0344c7f", **kwargs):
super().__init__(**{key: value for key, value in kwargs.items() if key == "chat_template"})
self.sampling_rate = int(sampling_rate)
self.window_seconds = float(window_seconds)
self.time_hz = int(time_hz)
self.schema_version = str(schema_version)
self.tokenizer_fingerprint = str(tokenizer_fingerprint)
self.tokenizer = SheetSage2Tokenizer(
self.window_seconds, self.time_hz, self.schema_version,
expected_fingerprint=self.tokenizer_fingerprint,
)
@property
def model_input_names(self):
return ["input_values", "attention_mask"]
@classmethod
def from_model_config(cls, config):
return cls(config.sampling_rate, config.input_audio_length, config.time_hz,
config.tokenizer_schema_version, config.tokenizer_fingerprint)
def __call__(self, audio, sampling_rate=None, padding=True, return_tensors="pt",
return_attention_mask=True):
source_rate = self.sampling_rate if sampling_rate is None else int(sampling_rate)
if source_rate <= 0:
raise ValueError("sampling_rate must be positive.")
if isinstance(audio, (str, bytes)):
raise ValueError("Pass waveform samples here; use model.transcribe for audio files.")
if isinstance(audio, (list, tuple)) and audio and not isinstance(audio[0], numbers.Number):
waveforms = [torch.as_tensor(value) for value in audio]
else:
value = torch.as_tensor(audio)
waveforms = list(value) if value.ndim == 2 else [value]
if not waveforms:
raise ValueError("Provide at least one waveform.")
prepared = []
maximum = round(self.window_seconds * self.sampling_rate)
for waveform in waveforms:
if waveform.ndim != 1 or not waveform.is_floating_point() or not torch.isfinite(waveform).all():
raise ValueError("Each waveform must be a one-dimensional finite floating-point array.")
waveform = waveform.to(dtype=torch.float32)
if source_rate != self.sampling_rate:
waveform = torchaudio.functional.resample(waveform, source_rate, self.sampling_rate)
if waveform.numel() < 1025:
raise ValueError("Each waveform must contain at least 1025 samples at 24 kHz.")
if waveform.numel() > maximum:
raise ValueError("Audio exceeds one model window; use model.transcribe for whole songs.")
prepared.append(waveform)
if len({str(value.device) for value in prepared}) != 1:
raise ValueError("All waveforms in a batch must be on the same device.")
if padding is False:
length = max(value.numel() for value in prepared)
if any(value.numel() != length for value in prepared):
raise ValueError("Use padding=True for waveforms of different lengths.")
elif padding is True or padding == "max_length":
length = maximum
elif padding == "longest":
length = max(value.numel() for value in prepared)
else:
raise ValueError("padding must be True, False, 'max_length', or 'longest'.")
lengths = torch.tensor([value.numel() for value in prepared], device=prepared[0].device)
values = torch.stack([torch.nn.functional.pad(value, (0, length - value.numel())) for value in prepared])
data = {"input_values": values}
if return_attention_mask:
data["attention_mask"] = (torch.arange(length, device=values.device)[None] < lengths[:, None]).long()
if return_tensors == "np":
data = {key: value.cpu().numpy() for key, value in data.items()}
elif return_tensors not in {None, "pt"}:
raise ValueError("return_tensors must be 'pt', 'np', or None.")
return BatchFeature(data=data)
def decode(self, token_ids, strict=True):
return self.tokenizer.decode_sequence(token_ids, strict=strict)
def batch_decode(self, sequences, strict=True):
return [self.decode(tokens, strict=strict) for tokens in sequences]
SheetSage2Processor.register_for_auto_class()
+11
View File
@@ -0,0 +1,11 @@
{
"processor_class": "SheetSage2Processor",
"sampling_rate": 24000,
"window_seconds": 300.0,
"time_hz": 100,
"schema_version": "v1",
"tokenizer_fingerprint": "5ba3325af0344c7f",
"auto_map": {
"AutoProcessor": "processing_sheetsage2.SheetSage2Processor"
}
}
+45
View File
@@ -0,0 +1,45 @@
"""Render existing SheetSage2 MIDI and ABC outputs without loading a model."""
import argparse
import json
from pathlib import Path
import sys
SCRIPT_DIR = Path(__file__).absolute().parent
sys.path.insert(0, str(SCRIPT_DIR))
from rendering_sheetsage2 import render_outputs
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--input", help="Transcription output directory.")
parser.add_argument("--midi", help="MIDI file with original note timing.")
parser.add_argument("--abc", help="ABC score file.")
parser.add_argument("--output", help="Output directory; defaults to --input or rendered/.")
parser.add_argument("--audio", action="store_true", help="Render piano WAV.")
parser.add_argument("--score", default="", help="Comma-separated pdf,svg,png.")
parser.add_argument("--parts", default="mix", help="Comma-separated mix,melody,vocal,instrumental,chords, or all (audio only).")
parser.add_argument("--duration", type=float, help="Minimum audio duration in seconds; preserves trailing silence.")
args = parser.parse_args()
audio = args.audio
score = args.score
if not audio and not score:
if args.input:
audio, score = True, "pdf"
else:
audio, score = bool(args.midi), "pdf" if args.abc else ""
if not args.input and not args.midi and not args.abc:
parser.error("Provide --input, --midi, or --abc.")
try:
result = render_outputs(args.input, midi=args.midi, abc=args.abc,
output_dir=args.output, audio=audio, score=score,
parts=args.parts, duration=args.duration)
except (ValueError, OSError, RuntimeError) as exc:
print(f"Rendering failed: {exc}", file=sys.stderr)
return 1
print(json.dumps(result, indent=2))
return 0
if __name__ == "__main__":
raise SystemExit(main())
+306
View File
@@ -0,0 +1,306 @@
"""Optional offline piano and staff rendering. No model or GPU is loaded."""
from __future__ import annotations
import base64
import hashlib
import io
import json
import math
import mimetypes
from pathlib import Path
import re
import wave
PARTS = ("mix", "melody", "vocal", "instrumental", "chords")
FORMATS = ("pdf", "svg", "png")
def _options(value, allowed, label):
values = value.split(",") if isinstance(value, str) else list(value)
values = tuple(dict.fromkeys(x.strip().lower() for x in values if x.strip()))
unknown = set(values) - set(allowed)
if unknown:
raise ValueError(f"Unknown {label}: {', '.join(sorted(unknown))}. Choose from {', '.join(allowed)}.")
return values
def validate_render_options(*, audio=False, score=(), parts=("mix",)):
"""Validate output choices without importing a model or browser runtime.
Return normalized ``(score_formats, audio_parts)`` tuples. ``score=True``
selects PDF; no requested output is valid for an inference-only call.
"""
score = ("pdf",) if score is True else () if score is False or score is None else score
formats = _options(score, FORMATS, "score format")
requested = _options(parts, PARTS + ("all",), "audio part")
if "all" in requested:
requested = PARTS
if audio and not requested:
raise ValueError("Choose at least one audio part.")
return formats, requested
def _tracks(midi):
try:
import pretty_midi
except ImportError as exc:
raise RuntimeError("Install rendering support first: python setup_render.py") from exc
try:
source = io.BytesIO(bytes(midi)) if isinstance(midi, (bytes, bytearray, memoryview)) else str(midi)
parsed = pretty_midi.PrettyMIDI(source)
except Exception as exc:
name = midi.name if isinstance(midi, Path) else "in-memory MIDI"
raise ValueError(f"Cannot read MIDI: {name}: {exc}") from exc
result = []
for instrument in parsed.instruments:
if instrument.is_drum:
raise ValueError("Piano rendering does not support drum tracks.")
notes = []
for note in instrument.notes:
if not all(math.isfinite(x) for x in (note.start, note.end)) or note.start < 0 or note.end <= note.start:
raise ValueError("MIDI note times must be finite, nonnegative, and increasing.")
if not 21 <= note.pitch <= 109:
raise ValueError(f"Piano samples cover MIDI pitches 21–109; received {note.pitch}.")
notes.append(dict(pitch=int(note.pitch), start=float(note.start), end=float(note.end), velocity=int(note.velocity)))
result.append(dict(name=instrument.name, notes=notes))
return result, float(parsed.get_end_time())
def _select(tracks, part):
def role(track):
name = track["name"].strip().lower()
if "chord" in name:
return "chords"
if "vocal" in name:
return "vocal"
if name in {"ins", "instrument", "instrumental"} or "instrumental" in name:
return "instrumental"
return "melody"
if part == "mix":
return tracks
if part == "melody":
return [track for track in tracks if role(track) != "chords"]
return [track for track in tracks if role(track) == part]
def _verify_assets(directory, *, audio):
manifest = directory / "manifest.json"
if not manifest.is_file():
raise FileNotFoundError("Rendering assets are missing. Download render_assets/ from the model repository.")
inventory = json.loads(manifest.read_text(encoding="utf-8"))["files"]
for relative, digest in inventory.items():
if not audio and relative.startswith("soundfonts/"):
continue
path = directory / relative
if not path.is_file() or hashlib.sha256(path.read_bytes()).hexdigest() != digest:
raise ValueError(f"Rendering asset is missing or corrupt: {relative}")
return set(inventory)
def _metadata_duration(directory):
for name, key in (("result.json", "duration_seconds"), ("playback.json", "duration")):
path = directory / name
if path.is_file():
value = json.loads(path.read_text(encoding="utf-8")).get(key)
if value is not None:
return float(value)
return None
def render_outputs(input_dir=None, *, midi=None, abc=None, output_dir=None,
audio=False, score=(), parts=("mix",), duration=None,
assets_dir=None):
"""Render existing outputs, preserving MIDI note times and canonical ABC.
``audio=True`` writes piano WAV; ``score`` selects pdf/svg/png. ``parts``
selects audio stems; ``all`` requests every available stem. SVG and PNG
contain one A4 page per file; PDF contains every page. No network request
is permitted during rendering. Assets and the browser must be installed.
"""
formats, requested = validate_render_options(audio=audio, score=score, parts=parts)
if not audio and not formats:
raise ValueError("Request audio or at least one score format.")
directory = Path(input_dir).expanduser().resolve() if input_dir else None
if directory is not None and not directory.is_dir():
raise FileNotFoundError(f"Input directory does not exist: {directory}")
destination = Path(output_dir or directory or "rendered").expanduser().resolve()
if midi is None and directory:
midi = directory / ("transcription.mid" if (directory / "transcription.mid").exists() else "melody.mid")
if abc is None and directory:
abc = directory / "score.abc"
if duration is None and directory:
duration = _metadata_duration(directory)
if duration is not None and (not math.isfinite(duration) or duration < 0):
raise ValueError("Duration must be finite and nonnegative.")
tracks, seconds = [], 0.0
if audio:
if midi is None or not Path(midi).is_file():
raise FileNotFoundError("Audio rendering needs an existing MIDI file (--midi or --input).")
tracks, end = _tracks(Path(midi))
seconds = max(end, duration or 0)
if seconds <= 0:
raise ValueError("An empty MIDI needs a positive --duration.")
abc_text = None
if formats:
if abc is None or not Path(abc).is_file():
raise FileNotFoundError("Score rendering needs an existing ABC file (--abc or --input).")
abc_text = Path(abc).read_text(encoding="utf-8")
if not abc_text.strip():
raise ValueError("ABC input is empty.")
return _render(tracks, seconds, abc_text, audio=audio, formats=formats,
requested=requested, destination=destination, assets_dir=assets_dir)
def render_memory(*, midi=None, abc=None, audio=False, score=(), parts=("mix",),
duration=None, assets_dir=None):
"""Render MIDI bytes and ABC text without writing input or result files.
Return WAV bytes in ``audio[part]`` and page lists in ``score[format]``:
PDF and PNG values are bytes, SVG values are strings. A PDF list contains
one complete document. The optional browser runtime may use its own profile.
"""
formats, requested = validate_render_options(audio=audio, score=score, parts=parts)
if not audio and not formats:
raise ValueError("Request audio or at least one score format.")
if duration is not None and (not math.isfinite(duration) or duration < 0):
raise ValueError("Duration must be finite and nonnegative.")
tracks, seconds = [], 0.0
if audio:
if not isinstance(midi, (bytes, bytearray, memoryview)):
raise ValueError("Audio rendering needs MIDI bytes.")
tracks, end = _tracks(midi)
seconds = max(end, duration or 0)
if seconds <= 0:
raise ValueError("An empty MIDI needs a positive duration.")
if formats and (not isinstance(abc, str) or not abc.strip()):
raise ValueError("Score rendering needs nonempty ABC text.")
return _render(tracks, seconds, abc, audio=audio, formats=formats,
requested=requested, assets_dir=assets_dir)
def _render(tracks, seconds, abc_text, *, audio, formats, requested,
destination=None, assets_dir=None):
assets = Path(assets_dir) if assets_dir else Path(__file__).with_name("render_assets")
assets = assets.resolve()
asset_files = _verify_assets(assets, audio=audio)
try:
from playwright.sync_api import sync_playwright, Error as BrowserError
except ImportError as exc:
raise RuntimeError("Install rendering support first: python setup_render.py") from exc
output = {"audio": {}, "score": {}, "warnings": []}
blocked = []
previous_pages = []
if destination is not None:
destination.mkdir(parents=True, exist_ok=True)
for path in destination.iterdir():
match = re.fullmatch(r"score_(\d{3,})\.(svg|png)", path.name)
if match and match[2] in formats and path.is_file() and not path.is_symlink():
previous_pages.append((int(match[1]), path))
with sync_playwright() as playwright:
try:
browser = playwright.chromium.launch(headless=True, args=["--autoplay-policy=no-user-gesture-required", "--disable-gpu"])
except Exception as exc:
raise RuntimeError("Could not start the renderer. Run python setup_render.py; on Linux with missing system libraries, use --with-deps.") from exc
try:
context = browser.new_context(viewport={"width": 794, "height": 1123}, device_scale_factor=2, service_workers="block", offline=True)
def route_handler(route):
from urllib.parse import unquote, urlsplit
url = urlsplit(route.request.url)
if url.scheme != "https" or url.netloc != "render.invalid":
blocked.append(route.request.url)
route.abort()
return
if url.path == "/":
route.fulfill(status=200, content_type="text/html", body="<!doctype html><html><head><meta charset='utf-8'><style>html,body{margin:0;background:white}.score-page{display:block;break-after:page;width:210mm;height:297mm}.score-page:last-child{break-after:auto}@page{size:A4;margin:0}</style></head><body></body></html>")
return
relative = unquote(url.path.lstrip("/"))
file = assets / relative
# HF snapshots link verified assets into the shared blob cache.
# Whitelist logical names instead of rejecting those symlinks.
if relative not in asset_files or not file.is_file():
blocked.append(url.path)
route.abort()
return
route.fulfill(status=200, path=str(file), content_type=mimetypes.guess_type(file.name)[0] or "application/octet-stream")
context.route("**/*", route_handler)
page = context.new_page()
page.goto("https://render.invalid/")
page.add_script_tag(url="https://render.invalid/abcjs-basic-min.js")
page.add_script_tag(url="https://render.invalid/renderer.js")
if audio:
for part in requested:
selected = _select(tracks, part)
if not selected and part not in {"mix", "melody"}:
output["warnings"].append(f"No {part} track is available; skipped.")
continue
info = page.evaluate("renderAudio", {"tracks": selected, "duration": seconds})
target = destination / f"piano_{part}.wav" if destination is not None else None
temporary = target.with_suffix(".wav.partial") if target is not None else None
buffer = io.BytesIO() if target is None else None
try:
with wave.open(str(temporary) if temporary is not None else buffer, "wb") as stream:
stream.setnchannels(info["channels"])
stream.setsampwidth(2)
stream.setframerate(info["sample_rate"])
for start in range(0, info["frames"], 32768):
encoded = page.evaluate("audioBlock", {"start": start, "count": 32768})
stream.writeframesraw(base64.b64decode(encoded))
if temporary is not None:
temporary.replace(target)
output["audio"][part] = str(target) if target is not None else buffer.getvalue()
finally:
if temporary is not None:
temporary.unlink(missing_ok=True)
page.evaluate("window.renderedAudio = null")
if formats:
page.evaluate("""async ({font, license}) => {
window.renderFontData = font;
window.renderFontLicense = license;
const face = new FontFace('SheetSageSans', 'url(data:font/ttf;base64,' + font + ')');
document.fonts.add(await face.load());
}""", {"font": base64.b64encode((assets / "DejaVuSans.ttf").read_bytes()).decode("ascii"),
"license": (assets / "LICENSE.font").read_text(encoding="utf-8")})
info = page.evaluate("renderScore", abc_text)
for kind in formats:
output["score"][kind] = []
if "pdf" in formats:
if destination is None:
output["score"]["pdf"] = [page.pdf(format="A4", print_background=True, prefer_css_page_size=True)]
else:
target = destination / "score.pdf"
temporary = target.with_suffix(".pdf.partial")
page.pdf(path=str(temporary), format="A4", print_background=True, prefer_css_page_size=True)
temporary.replace(target)
output["score"]["pdf"] = [str(target)]
for index, svg in enumerate(info["pages"], 1):
if "svg" in formats:
if destination is None:
output["score"]["svg"].append(svg)
else:
target = destination / f"score_{index:03}.svg"
target.write_text(svg, encoding="utf-8")
output["score"]["svg"].append(str(target))
if "png" in formats:
element = page.locator(".score-page").nth(index-1)
if destination is None:
output["score"]["png"].append(element.screenshot(type="png"))
else:
target = destination / f"score_{index:03}.png"
element.screenshot(path=str(target))
output["score"]["png"].append(str(target))
output["warnings"].extend(info["warnings"])
if blocked:
raise RuntimeError(f"Rendering attempted to load unavailable assets: {', '.join(blocked)}")
except BrowserError as exc:
raise RuntimeError(str(exc).split("Call log:", 1)[0].strip()) from exc
finally:
browser.close()
# A failed render leaves existing pages intact. Only remove obsolete pages
# in the requested formats, within this renderer's numbered-file namespace.
if formats:
for number, path in previous_pages:
if number > len(info["pages"]):
path.unlink(missing_ok=True)
return output
+4
View File
@@ -0,0 +1,4 @@
playwright==1.58.0
pretty_midi==0.2.10
setuptools==78.1.1
mir_eval
+11
View File
@@ -0,0 +1,11 @@
torch==2.8.0
torchaudio==2.8.0
transformers==4.45.2
huggingface-hub==0.36.0
safetensors==0.5.3
numpy==1.24.3
scipy==1.13.1
mir_eval==0.8.2
pretty_midi==0.2.10
mido==1.3.3
setuptools==78.1.1
+122
View File
@@ -0,0 +1,122 @@
from __future__ import annotations
from dataclasses import dataclass
@dataclass(frozen=True)
class PromptTaskSpec:
name: str
sampling_group: str
output_field: str
@dataclass(frozen=True)
class TokenBlockSpec:
name: str
labels: tuple[str, ...]
output_field: str | None = None
@dataclass(frozen=True)
class PromptMultitaskSchema:
version: str
tasks: tuple[PromptTaskSpec, ...]
event_field_order: tuple[str, ...]
appended_token_blocks: tuple[TokenBlockSpec, ...] = ()
V1_TASKS = (
PromptTaskSpec("timestamp", "timestamp", "timestamp"),
PromptTaskSpec("downbeat_meter", "rhythm", "rhythm"),
PromptTaskSpec("structure", "structure", "structure"),
PromptTaskSpec("key", "key", "key"),
PromptTaskSpec("chord_majmin", "chord", "chord"),
PromptTaskSpec("chord_full", "chord", "chord"),
PromptTaskSpec("melody_vocal", "melody", "melody"),
PromptTaskSpec("melody_full", "melody", "melody"),
)
PROMPT_MULTITASK_SCHEMAS = {
"v1": PromptMultitaskSchema(
version="v1",
tasks=V1_TASKS,
event_field_order=(
"timestamp",
"rhythm",
"structure",
"key",
"chord",
"melody",
),
),
}
def _validate_schema(schema):
task_names = tuple(task.name for task in schema.tasks)
if len(task_names) != len(set(task_names)):
raise ValueError(f"Duplicate task name in schema {schema.version!r}")
field_names = tuple(schema.event_field_order)
if len(field_names) != len(set(field_names)):
raise ValueError(f"Duplicate output field in schema {schema.version!r}")
missing_fields = {
task.output_field for task in schema.tasks if task.output_field not in field_names
}
if missing_fields:
raise ValueError(
f"Tasks in schema {schema.version!r} use unknown output fields: "
f"{tuple(sorted(missing_fields))}"
)
block_names = tuple(block.name for block in schema.appended_token_blocks)
if len(block_names) != len(set(block_names)):
raise ValueError(f"Duplicate token block in schema {schema.version!r}")
for block in schema.appended_token_blocks:
if not block.labels or len(block.labels) != len(set(block.labels)):
raise ValueError(
f"Token block {block.name!r} must contain unique labels"
)
def register_prompt_multitask_schema(schema):
"""Register a frozen schema version without permitting redefinition."""
if not isinstance(schema, PromptMultitaskSchema):
raise TypeError("schema must be a PromptMultitaskSchema")
_validate_schema(schema)
existing = PROMPT_MULTITASK_SCHEMAS.get(schema.version)
if existing is not None and existing != schema:
raise ValueError(f"Schema {schema.version!r} is already registered")
PROMPT_MULTITASK_SCHEMAS[schema.version] = schema
return schema
def extend_prompt_multitask_schema(
base_version,
version,
tasks=(),
event_fields=(),
token_blocks=(),
):
"""Append tasks and vocabulary blocks while preserving every base token id."""
base = get_prompt_multitask_schema(base_version)
schema = PromptMultitaskSchema(
version=str(version),
tasks=base.tasks + tuple(tasks),
event_field_order=base.event_field_order + tuple(event_fields),
appended_token_blocks=base.appended_token_blocks + tuple(token_blocks),
)
return register_prompt_multitask_schema(schema)
def get_prompt_multitask_schema(version):
version = str(version)
try:
return PROMPT_MULTITASK_SCHEMAS[version]
except KeyError as exc:
raise ValueError(
f"Unknown prompt multitask schema {version!r}; "
f"available={tuple(PROMPT_MULTITASK_SCHEMAS)}"
) from exc
_validate_schema(PROMPT_MULTITASK_SCHEMAS["v1"])
+37
View File
@@ -0,0 +1,37 @@
"""Install the optional renderer and its matching headless browser."""
from pathlib import Path
import argparse
import subprocess
import sys
SCRIPT_DIR = Path(__file__).absolute().parent
sys.path.insert(0, str(SCRIPT_DIR))
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--with-deps", action="store_true", help="Also install Chromium OS libraries (may require sudo).")
args = parser.parse_args()
subprocess.run([sys.executable, "-m", "pip", "install", "-r", str(SCRIPT_DIR / "requirements-render.txt")], check=True)
command = [sys.executable, "-m", "playwright", "install", "--only-shell", "chromium"]
if args.with_deps:
command.append("--with-deps")
subprocess.run(command, check=True)
check = """from playwright.sync_api import sync_playwright
with sync_playwright() as p:
browser = p.chromium.launch(headless=True, args=['--disable-gpu'])
context = browser.new_context(offline=True)
page = context.new_page()
assert page.evaluate('typeof OfflineAudioContext') == 'function'
browser.close()
"""
result = subprocess.run([sys.executable, "-c", check], capture_output=True, text=True)
if result.returncode:
print("Chromium could not start. On Linux, run python setup_render.py --with-deps to install its system libraries.", file=sys.stderr)
return 1
print("Rendering is ready. Example: python render.py --input output --audio --score pdf")
return 0
if __name__ == "__main__":
raise SystemExit(main())
+99
View File
@@ -0,0 +1,99 @@
"""Aligned audio features and token predictions, in memory or on disk."""
import json
from pathlib import Path
import torch
from safetensors.torch import save_file
from .generation_sheetsage2 import inference_autocast
class WindowTensorWriter:
def __init__(self, output_dir, index, window, logits, scores):
self.directory = Path(output_dir) / "tensors" / f"window-{index:04d}" if output_dir is not None else None
if self.directory is not None:
self.directory.mkdir(parents=True, exist_ok=True)
self.window = window
self.logits, self.scores = logits, scores
self.pending, self.files = [], []
self.data = {"window_index": index, "token_batches": [], "decoder_batches": []} if self.directory is None else None
def _store(self, name, payload, group):
if self.directory is not None:
save_file(payload, str(self.directory / name))
self.files.append(name)
elif group == "audio":
self.data["audio"] = payload
else:
self.data[group].append(payload)
def audio_features(self, model, audio, dtype, hidden_states):
with inference_autocast(audio.device, dtype):
features = model.get_audio_features(audio, output_hidden_states=hidden_states)
payload = {}
for name, value in features.items():
if torch.is_tensor(value):
payload[name] = value.detach().cpu().contiguous().clone()
elif isinstance(value, tuple):
for index, layer in enumerate(value):
if torch.is_tensor(layer):
payload[f"{name}.{index}"] = layer.detach().cpu().contiguous().clone()
memory = features.encoder_last_hidden_state
frames = memory.shape[1]
valid_seconds = self.window["end"] - self.window["start"]
payload["frame_times"] = torch.arange(frames, dtype=torch.float64) / 25 + self.window["start"]
payload["feature_attention_mask"] = (torch.arange(frames) / 25 < valid_seconds)[None]
self._store("audio.safetensors", payload, "audio")
return memory
def capture(self, position, ids, logits, masked):
# Transcription processes one window at a time; keep at most 32 rows.
row = {"token_position": torch.tensor(position, dtype=torch.long)}
if self.logits:
row["logits"] = logits[0].detach().cpu().contiguous()
if self.scores:
row["scores"] = masked[0].detach().cpu().contiguous()
self.pending.append(row)
if len(self.pending) == 32:
self.flush()
def flush(self):
if not self.pending:
return
first = int(self.pending[0]["token_position"])
name = f"tokens-{first:04d}.safetensors"
self._store(name, {key: torch.stack([row[key] for row in self.pending]) for key in self.pending[0]},
"token_batches")
self.pending.clear()
def decoder_features(self, model, audio, tokens, dtype, memory=None):
with inference_autocast(audio.device, dtype):
if memory is None:
memory = model.encode(audio)
cache = None
for start in range(0, len(tokens) - 1, 64):
ids = tokens[start:min(start + 64, len(tokens) - 1)][None].to(audio.device)
output = model.decoder(
input_ids=ids,
attention_mask=ids.ne(model.tokenizer.pad_token) if cache is None else None,
encoder_hidden_states=memory, encoder_attention_mask=None,
past_key_values=cache, use_cache=True, return_dict=True,
)
cache = output.past_key_values
name = f"decoder-{start:04d}.safetensors"
self._store(name, {"decoder_hidden_state": output.last_hidden_state.detach().cpu().contiguous(),
"input_token_ids": ids.cpu(),
"token_positions": torch.arange(start, start + ids.shape[1])},
"decoder_batches")
def finish(self):
self.flush()
metadata = dict(self.window, frame_rate=25)
if self.directory is not None:
metadata["files"] = self.files
metadata["token_position_convention"] = "logits predict the token at token_position; decoder embeddings represent input_token_ids"
if self.directory is not None:
(self.directory / "index.json").write_text(json.dumps(metadata, indent=2) + "\n")
else:
self.data["metadata"] = metadata
return self.data
+732
View File
@@ -0,0 +1,732 @@
import re
import hashlib
import json
import mir_eval.chord
import numpy as np
from .durations_sheetsage2 import DURATION_TEMPLATES, duration_boundaries
from .labels_sheetsage2 import STRUCTURE_LABELS
from .schema_sheetsage2 import get_prompt_multitask_schema
PROMPT_ORDER = tuple(task.name for task in get_prompt_multitask_schema("v1").tasks)
CHROMATIC_SHARPS = (
"C",
"C#",
"D",
"D#",
"E",
"F",
"F#",
"G",
"G#",
"A",
"A#",
"B",
)
FULL_CHORD_QUALITIES = (
"maj",
"min",
"dim",
"aug",
"maj7",
"min7",
"7",
"hdim7",
"dim7",
"minmaj7",
"sus2",
"sus4",
"sus4(b7)",
"maj6",
"min6",
)
FULL_CHORD_INVERSIONS = {
"maj": ("/2", "/3", "/5"),
"min": ("/2", "/b3", "/5"),
"maj7": ("/3", "/5", "/7"),
"min7": ("/b3", "/5", "/b7"),
"7": ("/3", "/5", "/b7"),
}
def build_full_chord_vocabulary():
labels = ["N"]
for quality in FULL_CHORD_QUALITIES:
inversions = FULL_CHORD_INVERSIONS.get(quality, ())
for root in CHROMATIC_SHARPS:
for inversion in (*inversions, ""):
labels.append(f"{root}:{quality}{inversion}")
return tuple(labels)
FULL_CHORD_VOCABULARY = build_full_chord_vocabulary()
class SheetSage2Tokenizer:
"""Typed prompt and event vocabulary for prompt-conditioned transcription."""
meter_numerators = tuple(range(1, 33))
meter_denominators = (1, 2, 4, 8, 16, 32)
n_eighth_positions = 256
max_subbeat_shift = 256
prompt_capacity = 256
def __init__(
self,
audio_length_seconds=300.0,
time_hz=100,
schema_version="v1",
expected_fingerprint=None,
):
self.audio_length_seconds = float(audio_length_seconds)
self.time_hz = int(time_hz)
self.schema = get_prompt_multitask_schema(schema_version)
self.schema_version = self.schema.version
self.task_specs = self.schema.tasks
self.event_field_order = self.schema.event_field_order
self.n_time_tokens = int(round(self.audio_length_seconds * self.time_hz))
if self.n_time_tokens <= 0:
raise ValueError("audio_length_seconds must produce at least one time token")
self.pad_token = 0
self.sos_token = 1
self.bos_token = self.sos_token
self.eos_token = 2
self.out_token = 3
self.prompt_token_start = 4
self.prompt_names = tuple(task.name for task in self.task_specs)
if len(self.prompt_names) > self.prompt_capacity:
raise ValueError(
f"Schema {self.schema_version} has {len(self.prompt_names)} prompts, "
f"exceeding immutable capacity {self.prompt_capacity}"
)
self.prompt_to_id = {
name: self.prompt_token_start + index
for index, name in enumerate(self.prompt_names)
}
self.prompt_token_end = self.prompt_token_start + self.prompt_capacity
self.subbeat_shift_token_start = self.prompt_token_end
self.n_subbeat_shift_tokens = self.max_subbeat_shift + 1
self.subbeat_shift_token_end = (
self.subbeat_shift_token_start + self.n_subbeat_shift_tokens
)
self.time_token_start = self.subbeat_shift_token_end
self.time_token_end = self.time_token_start + self.n_time_tokens
self.meter_pairs = tuple(
(numerator, denominator)
for numerator in self.meter_numerators
for denominator in self.meter_denominators
)
self.meter_to_id = {meter: index for index, meter in enumerate(self.meter_pairs)}
self.meter_token_start = self.time_token_end
self.meter_token_end = self.meter_token_start + len(self.meter_pairs)
self.eighth_position_token_start = self.meter_token_end
self.eighth_position_token_end = (
self.eighth_position_token_start + self.n_eighth_positions
)
self.structure_labels = tuple(STRUCTURE_LABELS)
self.structure_token_start = self.eighth_position_token_end
self.structure_token_end = self.structure_token_start + len(self.structure_labels)
self.key_token_start = self.structure_token_end
self.n_key_tokens = 24
self.key_token_end = self.key_token_start + self.n_key_tokens
self.majmin_chord_labels = (
"N",
*(f"{root}:maj" for root in CHROMATIC_SHARPS),
*(f"{root}:min" for root in CHROMATIC_SHARPS),
)
self.majmin_chord_token_start = self.key_token_end
self.majmin_chord_token_end = (
self.majmin_chord_token_start + len(self.majmin_chord_labels)
)
self.full_chord_labels = FULL_CHORD_VOCABULARY
self.full_chord_to_id = {
label: index for index, label in enumerate(self.full_chord_labels)
}
self.full_chord_token_start = self.majmin_chord_token_end
self.full_chord_token_end = (
self.full_chord_token_start + len(self.full_chord_labels)
)
self.pitch_token_start = self.full_chord_token_end
self.n_pitch_tokens = 256
self.pitch_token_end = self.pitch_token_start + self.n_pitch_tokens
self.duration_templates = DURATION_TEMPLATES
self.duration_boundaries = duration_boundaries
self.duration_token_start = self.pitch_token_end
self.n_duration_tokens = len(self.duration_templates)
self.duration_token_end = self.duration_token_start + self.n_duration_tokens
self.appended_token_blocks = {}
next_token = self.duration_token_end
for block in self.schema.appended_token_blocks:
start = next_token
end = start + len(block.labels)
self.appended_token_blocks[block.name] = {
"start": start,
"end": end,
"labels": tuple(block.labels),
"output_field": block.output_field or block.name,
}
next_token = end
self.n_tokens = next_token
self._full_chord_templates = self._build_full_chord_templates()
self.vocab_fingerprint = self._compute_fingerprint()
if (
expected_fingerprint is not None
and str(expected_fingerprint) != self.vocab_fingerprint
):
raise ValueError(
f"Tokenizer fingerprint mismatch for schema {self.schema_version}: "
f"expected {expected_fingerprint}, got {self.vocab_fingerprint}"
)
def _compute_fingerprint(self):
payload = {
"schema_version": self.schema_version,
"audio_length_seconds": self.audio_length_seconds,
"time_hz": self.time_hz,
"prompt_capacity": self.prompt_capacity,
"prompt_names": self.prompt_names,
"event_field_order": self.event_field_order,
"meter_pairs": self.meter_pairs,
"structure_labels": self.structure_labels,
"majmin_chord_labels": self.majmin_chord_labels,
"full_chord_labels": self.full_chord_labels,
"duration_templates": tuple(int(value) for value in self.duration_templates),
"appended_token_blocks": tuple(
(name, block["labels"], block["output_field"])
for name, block in self.appended_token_blocks.items()
),
"n_tokens": self.n_tokens,
}
encoded = json.dumps(payload, sort_keys=True, separators=(",", ":")).encode(
"utf-8"
)
return hashlib.sha256(encoded).hexdigest()[:16]
def get_config(self):
return {
"schema_version": self.schema_version,
"audio_length_seconds": self.audio_length_seconds,
"time_hz": self.time_hz,
"vocab_fingerprint": self.vocab_fingerprint,
"n_tokens": self.n_tokens,
}
def normalize_prompts(self, prompts):
names = []
seen = set()
for prompt in prompts:
name = str(prompt).strip()
if name.startswith("<|") and name.endswith("|>"):
name = name[2:-2]
if name not in self.prompt_to_id:
raise ValueError(f"Unknown prompt: {prompt!r}")
if name not in seen:
names.append(name)
seen.add(name)
names.sort(key=self.prompt_names.index)
selected_groups = {}
task_by_name = {task.name: task for task in self.task_specs}
for name in names:
group = task_by_name[name].sampling_group
previous = selected_groups.get(group)
if previous is not None:
raise ValueError(
f"Prompts {previous!r} and {name!r} are mutually exclusive "
f"within sampling group {group!r}"
)
selected_groups[group] = name
if not names:
raise ValueError("At least one task prompt is required")
return tuple(names)
def prompt_to_token(self, prompt):
return self.prompt_to_id[self.normalize_prompts((prompt,))[0]]
def token_to_prompt(self, token):
token = int(token)
for prompt, prompt_token in self.prompt_to_id.items():
if token == prompt_token:
return prompt
raise ValueError(f"token {token} is not a prompt token")
def prompt_prefix(self, prompts):
prompts = self.normalize_prompts(prompts)
return [
self.sos_token,
*(self.prompt_to_id[prompt] for prompt in prompts),
self.out_token,
]
def subbeat_shift_to_tokens(self, shift):
shift = int(shift)
if shift < 0:
raise ValueError("subbeat shift must be non-negative")
tokens = []
while shift > self.max_subbeat_shift:
tokens.append(self.subbeat_shift_token_start + self.max_subbeat_shift)
shift -= self.max_subbeat_shift
tokens.append(self.subbeat_shift_token_start + shift)
return tokens
def token_to_subbeat_shift(self, token):
token = int(token)
if not self.subbeat_shift_token_start <= token < self.subbeat_shift_token_end:
raise ValueError(f"token {token} is not a subbeat shift token")
return token - self.subbeat_shift_token_start
def time_id_to_token(self, time_id):
time_id = int(time_id)
if not 0 <= time_id < self.n_time_tokens:
raise ValueError(f"time id {time_id} is outside [0, {self.n_time_tokens})")
return self.time_token_start + time_id
def token_to_time_id(self, token):
token = int(token)
if not self.time_token_start <= token < self.time_token_end:
raise ValueError(f"token {token} is not a time token")
return token - self.time_token_start
def meter_to_token(self, numerator, denominator):
meter = (int(numerator), int(denominator))
if meter not in self.meter_to_id:
raise ValueError(f"Unsupported meter {meter[0]}/{meter[1]}")
return self.meter_token_start + self.meter_to_id[meter]
def eighth_position_to_token(self, position):
position = int(position)
if not 0 <= position < self.n_eighth_positions:
raise ValueError(
f"eighth-note position {position} is outside [0, {self.n_eighth_positions})"
)
return self.eighth_position_token_start + position
def structure_to_token(self, label):
label = str(label)
if label not in self.structure_labels:
raise ValueError(f"Unknown structure label: {label!r}")
return self.structure_token_start + self.structure_labels.index(label)
def key_to_token(self, key_label, pitch_shift=0):
tonic, mode = str(key_label).split(":", 1)
tonic_id = mir_eval.chord.pitch_class_to_semitone(tonic)
if tonic_id < 0:
raise ValueError(f"Invalid key tonic: {key_label!r}")
tonic_id = (int(tonic_id) + int(pitch_shift)) % 12
minor_modes = {"minor", "dorian", "phrygian", "locrian"}
mode_id = 1 if mode.lower() in minor_modes else 0
return self.key_token_start + mode_id * 12 + tonic_id
@staticmethod
def _canonical_chord_parts(chord_label, pitch_shift=0):
chord_label = str(chord_label).strip()
if chord_label in {"N", "X", ""}:
return "N", None, None
root, suffix = chord_label.split(":", 1)
root_id = mir_eval.chord.pitch_class_to_semitone(root)
if root_id < 0:
return "N", None, None
root_id = (int(root_id) + int(pitch_shift)) % 12
canonical = f"{CHROMATIC_SHARPS[root_id]}:{suffix}"
quality = re.split(r"/", suffix, maxsplit=1)[0]
return canonical, root_id, quality
def chord_majmin_to_token(self, chord_label, pitch_shift=0):
canonical, root_id, quality = self._canonical_chord_parts(
chord_label,
pitch_shift,
)
if canonical == "N":
return self.majmin_chord_token_start
try:
original_root, chroma, _ = mir_eval.chord.encode(str(chord_label))
relative = mir_eval.chord.rotate_bitmap_to_root(chroma, original_root)
has_minor_third = bool(relative[3] > 0)
has_major_third = bool(relative[4] > 0)
except Exception:
has_minor_third = quality.startswith(("min", "dim", "hdim"))
has_major_third = not has_minor_third
if has_major_third and not has_minor_third:
quality_id = 0
elif has_minor_third and not has_major_third:
quality_id = 1
elif quality.startswith(("min", "dim", "hdim")):
quality_id = 1
elif has_major_third:
quality_id = 0
else:
return self.majmin_chord_token_start
return self.majmin_chord_token_start + 1 + quality_id * 12 + root_id
@staticmethod
def _chord_template(chord_label):
root, chroma, bass = mir_eval.chord.encode(chord_label)
root_chroma = np.zeros(12, dtype=np.float32)
root_chroma[root] = 1.0
relative_chroma = mir_eval.chord.rotate_bitmap_to_root(chroma, root).astype(
np.float32
)
bass_chroma = np.zeros(12, dtype=np.float32)
bass_chroma[(bass + root) % 12] = 1.0
return np.concatenate([root_chroma, relative_chroma, bass_chroma])
def _build_full_chord_templates(self):
templates = [np.zeros(36, dtype=np.float32)]
templates.extend(
self._chord_template(label) for label in self.full_chord_labels[1:]
)
return np.stack(templates)
def chord_full_to_token(self, chord_label, pitch_shift=0):
canonical, _root_id, _quality = self._canonical_chord_parts(
chord_label,
pitch_shift,
)
chord_id = self.full_chord_to_id.get(canonical)
if chord_id is None:
if canonical == "N":
chord_id = 0
else:
try:
target = self._chord_template(canonical)
distances = np.abs(self._full_chord_templates - target[None]).sum(
axis=1
)
chord_id = int(np.argmin(distances))
except Exception:
chord_id = 0
return self.full_chord_token_start + chord_id
def pitch_to_token(self, pitch, track=0, full_melody=False):
pitch = int(pitch)
track = int(track)
if not 0 <= pitch < 128:
raise ValueError(f"MIDI pitch {pitch} is outside [0, 128)")
if track not in (0, 1):
raise ValueError(f"melody track {track} is outside [0, 2)")
pitch_id = pitch + (128 if full_melody and track == 1 else 0)
return self.pitch_token_start + pitch_id
def duration_bin_to_token(self, duration_bin):
duration_bin = int(duration_bin)
if not 0 <= duration_bin < self.n_duration_tokens:
raise ValueError(
f"duration bin {duration_bin} is outside [0, {self.n_duration_tokens})"
)
return self.duration_token_start + duration_bin
def token_to_duration_bin(self, token):
token = int(token)
if not self.duration_token_start <= token < self.duration_token_end:
raise ValueError(f"token {token} is not a duration token")
return token - self.duration_token_start
def appended_label_to_token(self, block_name, label):
block = self.appended_token_blocks.get(str(block_name))
if block is None:
raise ValueError(f"unknown appended token block: {block_name!r}")
try:
index = block["labels"].index(str(label))
except ValueError as exc:
raise ValueError(
f"unknown label {label!r} for appended block {block_name!r}"
) from exc
return block["start"] + index
def token_to_appended_label(self, token):
token = int(token)
token_type = self.token_type(token)
block = self.appended_token_blocks.get(token_type)
if block is None:
raise ValueError(f"token {token} is not from an appended token block")
return token_type, block["labels"][token - block["start"]]
def _token_output_field(self, token_type):
built_in = {
"time": "timestamp",
"meter": "rhythm",
"eighth_position": "rhythm",
"structure": "structure",
"key": "key",
"chord_majmin": "chord",
"chord_full": "chord",
"pitch": "melody",
"duration": "melody",
}
if token_type in built_in:
return built_in[token_type]
block = self.appended_token_blocks.get(token_type)
return None if block is None else block["output_field"]
def _decode_field(self, field, field_tokens, prompts):
token_types = [self.token_type(token) for token in field_tokens]
if field == "timestamp":
return self.token_to_time_id(field_tokens[0]) / self.time_hz
if field == "rhythm":
rhythm = {}
for token, token_type in zip(field_tokens, token_types):
if token_type == "meter":
rhythm["meter"] = self.meter_pairs[token - self.meter_token_start]
elif token_type == "eighth_position":
rhythm["eighth_position"] = token - self.eighth_position_token_start
return rhythm
if field == "structure":
return self.structure_labels[field_tokens[0] - self.structure_token_start]
if field == "key":
key_id = field_tokens[0] - self.key_token_start
mode = "minor" if key_id >= 12 else "major"
return f"{CHROMATIC_SHARPS[key_id % 12]}:{mode}"
if field == "chord":
if token_types[0] == "chord_majmin":
return self.majmin_chord_labels[
field_tokens[0] - self.majmin_chord_token_start
]
return self.full_chord_labels[
field_tokens[0] - self.full_chord_token_start
]
if field == "melody":
notes = []
index = 0
while index < len(field_tokens):
pitch_id = field_tokens[index] - self.pitch_token_start
duration_bin = 0
if (
index + 1 < len(field_tokens)
and self.token_type(field_tokens[index + 1]) == "duration"
):
duration_bin = self.token_to_duration_bin(field_tokens[index + 1])
index += 2
else:
index += 1
notes.append(
{
"pitch": pitch_id % 128,
"track": int(pitch_id >= 128),
"duration_bin": duration_bin,
"duration_steps": int(self.duration_templates[duration_bin]),
}
)
return notes
values = []
for token in field_tokens:
block_name, label = self.token_to_appended_label(token)
values.append({"block": block_name, "label": label})
return values[0] if len(values) == 1 else values
def decode_sequence(self, tokens, strict=True):
"""Parse one prompt-conditioned sequence into timed, typed events."""
if hasattr(tokens, "detach"):
tokens = tokens.detach().cpu().tolist()
tokens = [int(token) for token in tokens]
while tokens and tokens[-1] == self.pad_token:
tokens.pop()
if not tokens or tokens[0] != self.sos_token:
raise ValueError("sequence must begin with <|sos|>")
try:
out_index = tokens.index(self.out_token, 1)
except ValueError as exc:
raise ValueError("sequence is missing <|out|>") from exc
prompts = tuple(self.token_to_prompt(token) for token in tokens[1:out_index])
if strict and self.normalize_prompts(prompts) != prompts:
raise ValueError("prompt tokens are not in canonical schema order")
active_fields = {
task.output_field for task in self.task_specs if task.name in prompts
}
events = []
position = out_index + 1
current_step = 0
saw_eos = False
while position < len(tokens):
token = tokens[position]
if token == self.eos_token:
saw_eos = True
position += 1
break
if self.token_type(token) != "subbeat_shift":
raise ValueError(f"event at token index {position} has no subbeat shift")
shift = 0
while (
position < len(tokens)
and self.token_type(tokens[position]) == "subbeat_shift"
):
shift += self.token_to_subbeat_shift(tokens[position])
position += 1
current_step += shift
tokens_by_field = {field: [] for field in self.event_field_order}
while position < len(tokens):
token = tokens[position]
token_type = self.token_type(token)
if token_type == "subbeat_shift" or token == self.eos_token:
break
field = self._token_output_field(token_type)
if field is None or field not in tokens_by_field:
raise ValueError(
f"token {token} ({token_type}) has no field in schema "
f"{self.schema_version}"
)
if strict and field not in active_fields:
raise ValueError(
f"token {token} belongs to inactive output field {field!r}"
)
tokens_by_field[field].append(token)
position += 1
tokens_by_field = {
field: values for field, values in tokens_by_field.items() if values
}
if not tokens_by_field:
if strict:
raise ValueError(f"empty event at subbeat {current_step}")
continue
if strict:
for field, values in tokens_by_field.items():
types = [self.token_type(value) for value in values]
if field == "timestamp" and types != ["time"]:
raise ValueError("timestamp event must contain exactly one time token")
if field == "rhythm":
if types not in (["eighth_position"], ["meter", "eighth_position"]):
raise ValueError(f"invalid rhythm payload: {types}")
if field in {"structure", "key", "chord"} and len(values) != 1:
raise ValueError(f"field {field!r} must contain exactly one token")
if field == "melody":
index = 0
while index < len(types):
if types[index] != "pitch":
raise ValueError(
"melody payload must contain pitch tokens with optional duration"
)
if index + 1 < len(types) and types[index + 1] == "duration":
index += 2
else:
index += 1
events.append(
{
"subbeat": current_step,
"tokens_by_field": tokens_by_field,
"values": {
field: self._decode_field(field, values, prompts)
for field, values in tokens_by_field.items()
},
}
)
if strict and not saw_eos:
raise ValueError("sequence is missing <|eos|>")
if strict and position != len(tokens):
raise ValueError("non-padding tokens follow <|eos|>")
return {
"schema_version": self.schema_version,
"prompts": prompts,
"events": events,
"has_eos": saw_eos,
}
def encode_decoded_sequence(self, decoded):
"""Re-encode the lossless representation returned by decode_sequence."""
prompts = self.normalize_prompts(decoded["prompts"])
output = self.prompt_prefix(prompts)
previous_step = 0
for event in decoded["events"]:
step = int(event["subbeat"])
if step < previous_step:
raise ValueError("events must be sorted by non-decreasing subbeat")
output.extend(self.subbeat_shift_to_tokens(step - previous_step))
previous_step = step
tokens_by_field = event["tokens_by_field"]
for field in self.event_field_order:
output.extend(int(token) for token in tokens_by_field.get(field, ()))
if decoded.get("has_eos", True):
output.append(self.eos_token)
return output
def token_type(self, token):
token = int(token)
for name, start, end in (
("subbeat_shift", self.subbeat_shift_token_start, self.subbeat_shift_token_end),
("time", self.time_token_start, self.time_token_end),
("meter", self.meter_token_start, self.meter_token_end),
("eighth_position", self.eighth_position_token_start, self.eighth_position_token_end),
("structure", self.structure_token_start, self.structure_token_end),
("key", self.key_token_start, self.key_token_end),
("chord_majmin", self.majmin_chord_token_start, self.majmin_chord_token_end),
("chord_full", self.full_chord_token_start, self.full_chord_token_end),
("pitch", self.pitch_token_start, self.pitch_token_end),
("duration", self.duration_token_start, self.duration_token_end),
):
if start <= token < end:
return name
if token in self.prompt_to_id.values():
return "prompt"
for name, block in self.appended_token_blocks.items():
if block["start"] <= token < block["end"]:
return name
if token == self.pad_token:
return "pad"
if token == self.sos_token:
return "sos"
if token == self.eos_token:
return "eos"
if token == self.out_token:
return "out"
raise ValueError(f"token {token} is outside vocabulary size {self.n_tokens}")
def describe(self, token):
token = int(token)
token_type = self.token_type(token)
if token_type == "prompt":
prompt_by_id = {value: key for key, value in self.prompt_to_id.items()}
return f"<|{prompt_by_id[token]}|>"
if token_type == "subbeat_shift":
return f"<subbeat_shift_{token - self.subbeat_shift_token_start}>"
if token_type == "time":
time_id = token - self.time_token_start
return f"<time_{time_id / self.time_hz:.2f}s>"
if token_type == "meter":
numerator, denominator = self.meter_pairs[token - self.meter_token_start]
return f"<meter_{numerator}/{denominator}>"
if token_type == "eighth_position":
return f"<eighth_pos_{token - self.eighth_position_token_start}>"
if token_type == "structure":
return f"<structure_{self.structure_labels[token - self.structure_token_start]}>"
if token_type == "key":
key_id = token - self.key_token_start
mode = "minor" if key_id >= 12 else "major"
return f"<key_{CHROMATIC_SHARPS[key_id % 12]}:{mode}>"
if token_type == "chord_majmin":
return f"<chord_majmin_{self.majmin_chord_labels[token - self.majmin_chord_token_start]}>"
if token_type == "chord_full":
return f"<chord_full_{self.full_chord_labels[token - self.full_chord_token_start]}>"
if token_type == "pitch":
pitch = token - self.pitch_token_start
track = 1 if pitch >= 128 else 0
return f"<pitch_{pitch % 128}_track_{track}>"
if token_type == "duration":
return f"<duration_{token - self.duration_token_start}>"
if token_type in self.appended_token_blocks:
block = self.appended_token_blocks[token_type]
label = block["labels"][token - block["start"]]
return f"<{token_type}_{label}>"
return f"<|{token_type}|>"
+15
View File
@@ -0,0 +1,15 @@
# Third-party code notices
The Oobleck VAE and SnakeBeta implementation in `modeling_vae.py` is derived
from stable-audio-tools commit `a6ae0cdf8b2eb1567a4b42ceadddec3712d99d45`.
The module hierarchy, weight normalization and activation equations preserve
the checkpoint's original inference implementation.
- Oobleck / stable-audio-tools: Copyright (c) 2023 Stability AI, MIT.
Full text: `licenses/stable-audio-tools-MIT.txt`.
- SnakeBeta / BigVGAN: Copyright (c) 2022 NVIDIA CORPORATION, MIT.
Full text: `licenses/SnakeBeta-NVIDIA-MIT.txt`.
These notices cover the identified source code and retain its original licenses.
The YuE2 model checkpoint weights are separately licensed under CC BY-NC 4.0;
see MODEL_LICENSE for the scope and full terms. This does not relicense third-party code.
+38
View File
@@ -0,0 +1,38 @@
{
"return_dict": true,
"output_hidden_states": false,
"torchscript": false,
"dtype": "bfloat16",
"tie_word_embeddings": false,
"is_encoder_decoder": false,
"is_decoder": false,
"add_cross_attention": false,
"architectures": [
"YuE2ForCausalLM"
],
"bos_token_id": null,
"pad_token_id": null,
"eos_token_id": null,
"decoder_start_token_id": null,
"transformers_version": "4.57.6",
"auto_map": {
"AutoConfig": "modeling_yue2.YuE2Config",
"AutoModelForCausalLM": "modeling_yue2.YuE2ForCausalLM"
},
"model_type": "yue2",
"hidden_size": 2048,
"num_hidden_layers": 28,
"num_attention_heads": 16,
"num_key_value_heads": 8,
"head_dim": 128,
"intermediate_size": 6144,
"vocab_size": 184704,
"rms_norm_eps": 1e-06,
"rope_theta": 1000000,
"max_position_embeddings": 24576,
"latent_type": "vae",
"latent_dim": 64,
"max_latent_frames": 24576,
"timestep_shift": 1.0,
"output_attentions": false
}
+11
View File
@@ -0,0 +1,11 @@
{
"bos_token_id": 151643,
"eos_token_id": 151852,
"pad_token_id": 151643,
"do_sample": true,
"temperature": 1.0,
"top_p": 0.95,
"top_k": 100,
"max_new_tokens": 9000,
"use_cache": true
}
+705
View File
@@ -0,0 +1,705 @@
"""YuE2 AR–NAR Mixture-of-Transformers, with checkpoint-compatible names.
This module is self contained for Transformers ``trust_remote_code`` loading.
It imports no CUDA extension and implements the released model architecture.
``generate`` returns token IDs; the package pipeline supplies song generation.
"""
from __future__ import annotations
import math
from typing import List, Optional, Tuple, Union
import torch
import torch.nn as nn
import torch.nn.functional as F
from transformers import GenerationMixin, PretrainedConfig, PreTrainedModel
from transformers.cache_utils import DynamicCache
from transformers.modeling_outputs import CausalLMOutputWithPast
def sdpa(query, key, value, *, attn_mask=None, is_causal=False):
"""Use native grouped-query attention, including a portable MPS fallback."""
grouped = query.shape[1] != key.shape[1]
if grouped and query.device.type == "mps":
# PyTorch's MPS attention does not implement enable_gqa on every release.
groups = query.shape[1] // key.shape[1]
key = key.repeat_interleave(groups, dim=1)
value = value.repeat_interleave(groups, dim=1)
grouped = False
return F.scaled_dot_product_attention(
query, key, value, attn_mask=attn_mask, is_causal=is_causal,
enable_gqa=grouped,
)
def _causal_mask(attention_mask, cache_position, key_length, batch_size):
"""Physical cache slots are causal; RoPE positions may exclude padding."""
device = cache_position.device
visible = torch.arange(key_length, device=device)[None, :] <= cache_position[:, None]
visible = visible[None, None].expand(batch_size, 1, -1, -1)
if attention_mask is None:
return visible
mask = attention_mask.to(device=device)
if mask.ndim == 2:
if mask.shape[0] != batch_size or mask.shape[1] > key_length:
raise ValueError("attention_mask must cover the batch and used cache slots")
# Static cache has unused capacity after the supplied 2D padding mask.
if mask.shape[1] < key_length:
mask = F.pad(mask, (0, key_length - mask.shape[1]), value=0)
return visible & mask[:, None, None, :].bool()
if mask.ndim != 4 or mask.shape[-2:] != visible.shape[-2:]:
raise ValueError("Expected a 2D padding mask or a matching 4D attention mask")
if mask.dtype == torch.bool:
return visible & mask
return mask.masked_fill(~visible, float("-inf"))
# ══════════════════════════════════════════════════════════════════════════════
# Config
# ══════════════════════════════════════════════════════════════════════════════
class YuE2Config(PretrainedConfig):
model_type = "yue2"
_hf_fields = frozenset({
"model_type", "architectures", "auto_map", "transformers_version",
"dtype", "torch_dtype", "return_dict", "output_hidden_states",
"output_attentions", "use_cache", "tie_word_embeddings", "torchscript",
"is_decoder", "is_encoder_decoder", "add_cross_attention",
"bos_token_id", "eos_token_id", "pad_token_id", "decoder_start_token_id",
"attn_implementation",
})
def to_dict(self):
return {key: value for key, value in super().to_dict().items()
if key in self._hf_fields or key in self._inference_fields}
_inference_fields = frozenset(['hidden_size', 'num_hidden_layers', 'num_attention_heads', 'num_key_value_heads', 'head_dim', 'intermediate_size', 'vocab_size', 'rms_norm_eps', 'rope_theta', 'max_position_embeddings', 'tie_word_embeddings', 'latent_type', 'latent_dim', 'max_latent_frames', 'timestep_shift'])
def __init__(
self,
hidden_size: int = 2048,
num_hidden_layers: int = 28,
num_attention_heads: int = 16,
num_key_value_heads: int = 8,
head_dim: int = 128,
intermediate_size: int = 6144,
vocab_size: int = 184704,
rms_norm_eps: float = 1e-6,
rope_theta: float = 1000000.0,
max_position_embeddings: int = 24576,
tie_word_embeddings: bool = False,
# Acoustic inference architecture
latent_type: str = "vae",
latent_dim: int = 64,
max_latent_frames: int = 24576,
timestep_shift: float = 1.0,
**kwargs,
):
if latent_type != "vae":
raise ValueError("YuE2 inference supports only latent_type='vae'")
# Serialize only the documented model and Transformers configuration.
kwargs = {key: value for key, value in kwargs.items() if key in self._hf_fields}
super().__init__(tie_word_embeddings=tie_word_embeddings, **kwargs)
self.hidden_size = hidden_size
self.num_hidden_layers = num_hidden_layers
self.num_attention_heads = num_attention_heads
self.num_key_value_heads = num_key_value_heads
self.head_dim = head_dim
self.intermediate_size = intermediate_size
self.vocab_size = vocab_size
self.rms_norm_eps = rms_norm_eps
self.rope_theta = rope_theta
self.max_position_embeddings = max_position_embeddings
self.latent_type = latent_type
self.latent_dim = latent_dim
self.max_latent_frames = max_latent_frames
self.timestep_shift = timestep_shift
# ══════════════════════════════════════════════════════════════════════════════
# Building blocks
# ══════════════════════════════════════════════════════════════════════════════
class RMSNorm(nn.Module):
def __init__(self, dim: int, eps: float = 1e-6):
super().__init__()
self.weight = nn.Parameter(torch.ones(dim))
self.eps = eps
def forward(self, x: torch.Tensor) -> torch.Tensor:
return x * torch.rsqrt(x.float().pow(2).mean(-1, keepdim=True) + self.eps).to(x.dtype) * self.weight
class RotaryEmbedding(nn.Module):
def __init__(self, head_dim: int, base: float = 1000000.0):
super().__init__()
self.head_dim = head_dim
self.base = base
self._inv_freq: Optional[torch.Tensor] = None
def forward(self, position_ids: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
if self._inv_freq is None or self._inv_freq.device != position_ids.device:
self._inv_freq = 1.0 / (self.base ** (
torch.arange(0, self.head_dim, 2, dtype=torch.float32, device=position_ids.device) / self.head_dim
))
pos = position_ids.float().unsqueeze(-1)
angles = pos * self._inv_freq
return angles.cos(), angles.sin()
def _apply_rotary(x: torch.Tensor, cos: torch.Tensor, sin: torch.Tensor) -> torch.Tensor:
half = x.shape[-1] // 2
x1, x2 = x[..., :half], x[..., half:]
cos, sin = cos.to(x.dtype), sin.to(x.dtype)
return torch.cat([x1 * cos - x2 * sin, x2 * cos + x1 * sin], dim=-1)
class Attention(nn.Module):
def __init__(self, config: YuE2Config):
super().__init__()
self.num_heads = config.num_attention_heads
self.num_kv_heads = config.num_key_value_heads
self.head_dim = config.head_dim
self.num_kv_groups = self.num_heads // self.num_kv_heads
self.q_proj = nn.Linear(config.hidden_size, self.num_heads * self.head_dim, bias=False)
self.k_proj = nn.Linear(config.hidden_size, self.num_kv_heads * self.head_dim, bias=False)
self.v_proj = nn.Linear(config.hidden_size, self.num_kv_heads * self.head_dim, bias=False)
self.o_proj = nn.Linear(self.num_heads * self.head_dim, config.hidden_size, bias=False)
self.q_norm = RMSNorm(self.head_dim, config.rms_norm_eps)
self.k_norm = RMSNorm(self.head_dim, config.rms_norm_eps)
def project_qkv(
self, x: torch.Tensor, cos: torch.Tensor, sin: torch.Tensor,
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
"""Project, normalize, and apply RoPE. No SDPA, no KV cache, no O proj.
Returns Q [B,T,num_heads,hd], K [B,T,num_kv_heads,hd], V [B,T,num_kv_heads,hd].
"""
B, T, _ = x.shape
q = self.q_proj(x).view(B, T, self.num_heads, self.head_dim)
k = self.k_proj(x).view(B, T, self.num_kv_heads, self.head_dim)
v = self.v_proj(x).view(B, T, self.num_kv_heads, self.head_dim)
q, k = self.q_norm(q), self.k_norm(k)
rc, rs = cos.unsqueeze(2), sin.unsqueeze(2)
q = _apply_rotary(q, rc, rs)
k = _apply_rotary(k, rc, rs)
return q, k, v
def forward(
self,
x: torch.Tensor,
cos: torch.Tensor,
sin: torch.Tensor,
past_key_value: Optional[DynamicCache] = None,
layer_idx: int = 0,
attention_mask: Optional[torch.Tensor] = None,
cache_position: Optional[torch.Tensor] = None,
) -> torch.Tensor:
B, T, _ = x.shape
q, k, v = self.project_qkv(x, cos, sin)
q, k, v = q.transpose(1, 2), k.transpose(1, 2), v.transpose(1, 2)
if past_key_value is not None:
k, v = past_key_value.update(k, v, layer_idx, {"cache_position": cache_position})
if attention_mask is not None:
out = sdpa(q, k, v, attn_mask=attention_mask[..., :k.shape[2]])
else:
out = sdpa(q, k, v, is_causal=(T > 1 and k.shape[2] == T))
return self.o_proj(out.transpose(1, 2).reshape(B, T, -1))
class MLP(nn.Module):
def __init__(self, config: YuE2Config):
super().__init__()
self.gate_proj = nn.Linear(config.hidden_size, config.intermediate_size, bias=False)
self.up_proj = nn.Linear(config.hidden_size, config.intermediate_size, bias=False)
self.down_proj = nn.Linear(config.intermediate_size, config.hidden_size, bias=False)
def forward(self, x: torch.Tensor) -> torch.Tensor:
return self.down_proj(F.silu(self.gate_proj(x)) * self.up_proj(x))
class DecoderLayer(nn.Module):
"""Transformer layer with full MoT: dual attention projections + dual MLP."""
def __init__(self, config: YuE2Config):
super().__init__()
# AR attention path
self.input_layernorm = RMSNorm(config.hidden_size, config.rms_norm_eps)
self.self_attn = Attention(config)
# NAR attention path (separate Q/K/V/O + layernorms)
self.nar_input_layernorm = RMSNorm(config.hidden_size, config.rms_norm_eps)
self.nar_self_attn = Attention(config)
# AR MLP path
self.post_attention_layernorm = RMSNorm(config.hidden_size, config.rms_norm_eps)
self.mlp = MLP(config)
# NAR MLP path
self.nar_pre_mlp_layernorm = RMSNorm(config.hidden_size, config.rms_norm_eps)
self.nar_mlp = MLP(config)
def forward(
self,
x: torch.Tensor,
cos: torch.Tensor,
sin: torch.Tensor,
past_key_value: Optional[DynamicCache] = None,
layer_idx: int = 0,
attention_mask: Optional[torch.Tensor] = None,
ar_mask: Optional[torch.Tensor] = None,
cache_position: Optional[torch.Tensor] = None,
) -> torch.Tensor:
if ar_mask is not None:
mask_3d = ar_mask.unsqueeze(-1) # [B, S, 1]
mask_4d = ar_mask.unsqueeze(-1).unsqueeze(-1) # [B, S, 1, 1]
# Per-type input layernorm
ln_ar = self.input_layernorm(x)
ln_nar = self.nar_input_layernorm(x)
# Per-type QKV projection (both process all tokens)
q_ar, k_ar, v_ar = self.self_attn.project_qkv(ln_ar, cos, sin)
q_nar, k_nar, v_nar = self.nar_self_attn.project_qkv(ln_nar, cos, sin)
# Merge Q/K/V per-position: AR positions use AR projections, NAR use NAR
query = torch.where(mask_4d, q_ar, q_nar) # [B, S, num_heads, hd]
# K/V have num_kv_heads (fewer), same mask broadcast works
key = torch.where(mask_4d, k_ar, k_nar) # [B, S, num_kv_heads, hd]
value = torch.where(mask_4d, v_ar, v_nar)
# Transpose to [B, H, S, D] for SDPA
B, S = x.shape[:2]
query = query.transpose(1, 2)
key = key.transpose(1, 2)
value = value.transpose(1, 2)
# Shared attention with hybrid mask
if attention_mask is not None and attention_mask.dtype != torch.bool:
attention_mask = attention_mask.to(query.dtype)
core_out = sdpa(query, key, value, attn_mask=attention_mask)
core_out = core_out.transpose(1, 2).reshape(B, S, -1)
# Per-type O projection, merge by mask
o_ar = self.self_attn.o_proj(core_out)
o_nar = self.nar_self_attn.o_proj(core_out)
h = torch.where(mask_3d, o_ar, o_nar)
x = x + h
# Per-type MLP
ar_out = self.mlp(self.post_attention_layernorm(x))
nar_out = self.nar_mlp(self.nar_pre_mlp_layernorm(x))
mlp_out = torch.where(mask_3d, ar_out, nar_out)
else:
# AR-only mode (generation): use AR path only
h = self.self_attn(self.input_layernorm(x), cos, sin, past_key_value, layer_idx,
attention_mask, cache_position)
x = x + h
mlp_out = self.mlp(self.post_attention_layernorm(x))
x = x + mlp_out
return x
# ══════════════════════════════════════════════════════════════════════════════
# NAR auxiliary modules
# ══════════════════════════════════════════════════════════════════════════════
class TimestepEmbedder(nn.Module):
"""Sinusoidal timestep → MLP → hidden_size (same as modules.py)."""
def __init__(self, hidden_size: int, frequency_embedding_size: int = 256):
super().__init__()
self.mlp = nn.Sequential(
nn.Linear(frequency_embedding_size, hidden_size),
nn.SiLU(),
nn.Linear(hidden_size, hidden_size),
)
self.frequency_embedding_size = frequency_embedding_size
def forward(self, t):
half = self.frequency_embedding_size // 2
freqs = torch.exp(
-math.log(10000) * torch.arange(half, device=t.device, dtype=torch.float32) / half
)
args = t.float().unsqueeze(-1) * freqs.unsqueeze(0)
emb = torch.cat([torch.cos(args), torch.sin(args)], dim=-1)
return self.mlp(emb.to(next(self.parameters()).dtype))
class AudioPositionEmbedding(nn.Module):
"""Non-learnable 1D sinusoidal PE for audio latent frames."""
def __init__(self, max_frames: int, hidden_size: int):
super().__init__()
pe = torch.zeros(max_frames, hidden_size)
position = torch.arange(0, max_frames, dtype=torch.float32).unsqueeze(1)
div_term = torch.exp(
torch.arange(0, hidden_size, 2, dtype=torch.float32) * (-math.log(10000.0) / hidden_size)
)
pe[:, 0::2] = torch.sin(position * div_term)
pe[:, 1::2] = torch.cos(position * div_term)
self.register_buffer("pe", pe)
def forward(self, position_ids):
return self.pe[position_ids]
# ══════════════════════════════════════════════════════════════════════════════
# Static KV Cache
# ══════════════════════════════════════════════════════════════════════════════
class StaticKVCache:
"""Bounded, append-only cache for the explicit single-request AR loop.
Returns views of the used prefix and never reallocates/copies its history.
Standard HF ``generate`` also supports Transformers' own StaticCache.
"""
def __init__(
self, num_layers: int, batch_size: int, num_kv_heads: int,
max_seq_len: int, head_dim: int, dtype: torch.dtype, device: torch.device,
):
self.num_layers = num_layers
self.max_seq_len = max_seq_len
self._seen_tokens = 0
self.key_cache: List[torch.Tensor] = [
torch.zeros(batch_size, num_kv_heads, max_seq_len, head_dim, dtype=dtype, device=device)
for _ in range(num_layers)
]
self.value_cache: List[torch.Tensor] = [
torch.zeros(batch_size, num_kv_heads, max_seq_len, head_dim, dtype=dtype, device=device)
for _ in range(num_layers)
]
def get_seq_length(self, layer_idx=0) -> int:
return self._seen_tokens
def update(self, key_states, value_states, layer_idx, cache_kwargs=None):
T = key_states.shape[2]
pos = self._seen_tokens
end = pos + T
if end > self.max_seq_len:
raise ValueError(f"KV cache capacity {self.max_seq_len} exceeded by {end}; generation was not shortened")
self.key_cache[layer_idx][:, :, pos:end] = key_states
self.value_cache[layer_idx][:, :, pos:end] = value_states
if layer_idx == self.num_layers - 1:
self._seen_tokens = end
return self.key_cache[layer_idx][:, :, :end], self.value_cache[layer_idx][:, :, :end]
def reset(self):
self._seen_tokens = 0
def reorder_cache(self, beam_idx):
self.key_cache = [v.index_select(0, beam_idx.to(v.device)) for v in self.key_cache]
self.value_cache = [v.index_select(0, beam_idx.to(v.device)) for v in self.value_cache]
# ══════════════════════════════════════════════════════════════════════════════
# Model
# ══════════════════════════════════════════════════════════════════════════════
class Backbone(nn.Module):
"""Transformer backbone with MoT dual MLP."""
def __init__(self, config: YuE2Config):
super().__init__()
self.embed_tokens = nn.Embedding(config.vocab_size, config.hidden_size)
self.layers = nn.ModuleList([DecoderLayer(config) for _ in range(config.num_hidden_layers)])
self.norm = RMSNorm(config.hidden_size, config.rms_norm_eps)
self.rotary_emb = RotaryEmbedding(config.head_dim, config.rope_theta)
def forward(
self,
input_ids: Optional[torch.LongTensor] = None,
position_ids: Optional[torch.LongTensor] = None,
past_key_values=None,
use_cache: bool = True,
attention_mask: Optional[torch.Tensor] = None,
ar_mask: Optional[torch.Tensor] = None,
inputs_embeds: Optional[torch.Tensor] = None,
cache_position: Optional[torch.Tensor] = None,
) -> Tuple[torch.Tensor, ...]:
if inputs_embeds is not None:
x = inputs_embeds
else:
x = self.embed_tokens(input_ids)
cos, sin = self.rotary_emb(position_ids)
if use_cache and past_key_values is None:
past_key_values = DynamicCache()
for i, layer in enumerate(self.layers):
x = layer(x, cos, sin, past_key_values if use_cache else None,
layer_idx=i, attention_mask=attention_mask, ar_mask=ar_mask,
cache_position=cache_position)
return self.norm(x), past_key_values
class YuE2PreTrainedModel(PreTrainedModel):
config_class = YuE2Config
base_model_prefix = "model"
supports_gradient_checkpointing = True
_no_split_modules = ["DecoderLayer"]
_supports_sdpa = True
def _init_weights(self, module):
if isinstance(module, nn.Linear):
nn.init.normal_(module.weight, std=0.01)
if module.bias is not None:
nn.init.zeros_(module.bias)
elif isinstance(module, nn.Embedding):
nn.init.normal_(module.weight, std=0.01)
class YuE2ForCausalLM(YuE2PreTrainedModel, GenerationMixin):
"""YuE2 model: AR causal LM (generate) + NAR flow matching (ODE)."""
def __init__(self, config: YuE2Config):
super().__init__(config)
self.model = Backbone(config)
self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
# NAR auxiliary
self.llm2vae = nn.Linear(config.hidden_size, config.latent_dim)
self.vae2llm = nn.Linear(config.latent_dim, config.hidden_size)
self.time_embedder = TimestepEmbedder(config.hidden_size)
self.latent_pos_embed = AudioPositionEmbedding(config.max_latent_frames, config.hidden_size)
self.post_init()
def get_input_embeddings(self):
return self.model.embed_tokens
def set_input_embeddings(self, value):
self.model.embed_tokens = value
def get_output_embeddings(self):
return self.lm_head
def set_output_embeddings(self, new_embeddings):
self.lm_head = new_embeddings
# ── AR forward (standard causal LM, KV cached) ───────────────────
def forward(
self,
input_ids: Optional[torch.LongTensor] = None,
attention_mask: Optional[torch.Tensor] = None,
position_ids: Optional[torch.LongTensor] = None,
past_key_values=None,
inputs_embeds: Optional[torch.FloatTensor] = None,
labels: Optional[torch.LongTensor] = None,
use_cache: Optional[bool] = None,
cache_position: Optional[torch.LongTensor] = None,
logits_to_keep: Union[int, torch.Tensor] = 0,
return_dict: Optional[bool] = None,
**kwargs,
) -> Union[Tuple, CausalLMOutputWithPast]:
if (input_ids is None) == (inputs_embeds is None):
raise ValueError("Supply exactly one of input_ids or inputs_embeds")
use_cache = use_cache if use_cache is not None else getattr(self.config, "use_cache", True)
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
tensor = input_ids if input_ids is not None else inputs_embeds
batch_size, seq_len = tensor.shape[:2]
if not seq_len:
raise ValueError("Input must contain at least one token")
device = tensor.device
past_len = past_key_values.get_seq_length() if past_key_values is not None and use_cache else 0
if cache_position is None:
cache_position = torch.arange(past_len, past_len + seq_len, device=device)
else:
cache_position = cache_position.to(device=device, dtype=torch.long)
if cache_position.ndim != 1 or cache_position.numel() != seq_len:
raise ValueError("cache_position must identify each current token's physical cache slot")
if position_ids is None:
if attention_mask is not None and attention_mask.ndim == 2:
position_ids = attention_mask.long().cumsum(-1) - 1
position_ids.masked_fill_(attention_mask == 0, 0)
position_ids = position_ids[:, -seq_len:].to(device)
else:
position_ids = cache_position[None]
else:
position_ids = position_ids.to(device=device, dtype=torch.long)
if position_ids.shape[-1] != seq_len:
raise ValueError("position_ids must cover the current input tokens")
key_length = past_len + seq_len
if use_cache and past_key_values is not None and hasattr(past_key_values, "get_max_cache_shape"):
capacity = past_key_values.get_max_cache_shape()
if capacity is not None and capacity > 0:
key_length = capacity
# No explicit mask is needed for unpadded prefill or single-token dynamic
# decode. Chunked prefill needs bottom-right causal alignment; a full
# static cache additionally needs to hide all unfilled slots.
needs_mask = attention_mask is not None or key_length != past_len + seq_len or (past_len > 0 and seq_len > 1)
causal_mask = _causal_mask(attention_mask, cache_position, key_length, batch_size) if needs_mask else None
hidden_states, past_key_values = self.model(
input_ids=input_ids, position_ids=position_ids,
past_key_values=past_key_values, use_cache=use_cache,
attention_mask=causal_mask, inputs_embeds=inputs_embeds,
cache_position=cache_position,
)
if isinstance(logits_to_keep, int):
if logits_to_keep < 0:
raise ValueError("logits_to_keep must be nonnegative")
selected = hidden_states[:, -logits_to_keep:, :] if logits_to_keep else hidden_states
else:
selected = hidden_states[:, logits_to_keep.to(device), :]
if labels is not None and selected.shape[1] != hidden_states.shape[1]:
raise ValueError("Loss computation requires logits_to_keep=0")
logits = self.lm_head(selected)
loss = None
if labels is not None:
shift_logits = logits[..., :-1, :].contiguous()
shift_labels = labels[..., 1:].contiguous()
loss = F.cross_entropy(shift_logits.view(-1, shift_logits.size(-1)), shift_labels.view(-1))
if not return_dict:
output = (logits, past_key_values) if use_cache else (logits,)
return ((loss,) + output) if loss is not None else output
return CausalLMOutputWithPast(loss=loss, logits=logits, past_key_values=past_key_values if use_cache else None)
def prepare_inputs_for_generation(
self, input_ids, past_key_values=None, attention_mask=None,
inputs_embeds=None, cache_position=None, position_ids=None, **kwargs,
):
"""Keep physical cache slots separate from padding-aware RoPE positions."""
past_len = past_key_values.get_seq_length() if past_key_values is not None else 0
if cache_position is None:
total = inputs_embeds.shape[1] if inputs_embeds is not None and past_len == 0 else input_ids.shape[1]
count = max(total - past_len, 1) if past_len else total
cache_position = torch.arange(past_len, past_len + count, device=input_ids.device)
count = cache_position.numel()
use_embeds = inputs_embeds is not None and past_len == 0
if use_embeds:
current_ids, current_embeds = None, inputs_embeds[:, -count:]
else:
current_ids, current_embeds = input_ids[:, -count:].contiguous(), None
if position_ids is None and attention_mask is not None and attention_mask.ndim == 2:
position_ids = attention_mask.long().cumsum(-1) - 1
position_ids.masked_fill_(attention_mask == 0, 0)
if position_ids is not None:
position_ids = position_ids[:, -count:].contiguous()
return {
"input_ids": current_ids, "inputs_embeds": current_embeds,
"past_key_values": past_key_values, "attention_mask": attention_mask,
"position_ids": position_ids, "cache_position": cache_position,
"use_cache": kwargs.get("use_cache", True),
"logits_to_keep": kwargs.get("logits_to_keep", 1),
}
# ── NAR velocity (flow matching, no KV cache) ────────────────────
def _shift_t_value(self, t_value: float, device: torch.device, dtype: torch.dtype) -> torch.Tensor:
t_sig = torch.sigmoid(torch.tensor(t_value, dtype=dtype, device=device))
shift = self.config.timestep_shift
return shift * t_sig / (1 + (shift - 1) * t_sig)
@torch.no_grad()
def nar_velocity(
self,
tokens: torch.LongTensor,
ar_mask: torch.BoolTensor,
nar_mask: torch.BoolTensor,
nar_content_mask: torch.BoolTensor,
x_t: torch.Tensor,
t_value: float,
nar_cond_end: int = 0,
) -> torch.Tensor:
"""Compute v_theta(x_t, t) — flow-matching velocity field.
Args:
tokens: [1, S] full sequence (AR + NAR tokens)
ar_mask: [1, S] True for AR positions
nar_mask: [1, S] True for NAR positions
nar_content_mask: [1, S] True for actual latent positions (not LATENT_START/END)
x_t: [T_lat, D] current ODE state
t_value: raw timestep (will be sigmoid-shifted)
nar_cond_end: if > 0, NAR only sees positions < nar_cond_end (text-only mode)
Returns:
v_pred: [T_lat, D] predicted velocity
"""
device = tokens.device
dtype = next(self.parameters()).dtype
B, S = tokens.shape
# 1. Token embeddings
token_emb = self.model.embed_tokens(tokens) # [B, S, H]
# 2. Build latent hidden for ALL NAR positions (START + content + END)
# Training injects vae2llm(x_t) + time_emb + pos_emb at ALL NAR positions,
# including LATENT_START (clean=0) and LATENT_END (clean=0).
# NAR position IDs via cumsum: START=0, content=[1..T_lat], END=T_lat+1.
t_shifted = self._shift_t_value(t_value, device, dtype)
T_lat = x_t.shape[0]
nar_indices = nar_mask[0].nonzero(as_tuple=True)[0] # all NAR positions
content_indices = nar_content_mask[0].nonzero(as_tuple=True)[0]
N_nar = nar_indices.shape[0] # START + T_lat + END
# Build x_t for all NAR positions: zeros for START/END, actual x_t for content
x_nar = torch.zeros(N_nar, x_t.shape[1], device=device, dtype=dtype)
x_nar[1:1 + T_lat] = x_t.to(dtype) # content frames at positions [1, T_lat]
latent_hidden_nar = self.vae2llm(x_nar.unsqueeze(0)) # [1, N_nar, H]
# Timestep embedding (same t for all NAR positions)
time_emb = self.time_embedder(t_shifted.expand(N_nar)).unsqueeze(0)
latent_hidden_nar = latent_hidden_nar + time_emb
# Position embedding: cumsum-style [0, 1, 2, ..., N_nar-1]
pos_ids = torch.arange(N_nar, device=device).clamp(max=self.config.max_latent_frames - 1)
pos_emb = self.latent_pos_embed(pos_ids).unsqueeze(0)
latent_hidden_nar = latent_hidden_nar + pos_emb
# Inject at ALL NAR positions (matching training's torch.where)
token_emb[0, nar_indices] = latent_hidden_nar[0]
# 3. Build hybrid attention mask [B, 1, S, S]
# AR→AR: causal, NAR→AR: full, NAR→NAR: bidirectional, AR→NAR: blocked
ar_q = ar_mask.unsqueeze(2).float() # [B, S, 1]
ar_k = ar_mask.unsqueeze(1).float() # [B, 1, S]
nar_q = nar_mask.unsqueeze(2).float()
nar_k = nar_mask.unsqueeze(1).float()
causal = torch.tril(torch.ones(S, S, device=device))
if nar_cond_end > 0:
# Codec dropout: NAR only sees positions < nar_cond_end (text) + NAR
text_k = torch.zeros(1, 1, S, device=device)
text_k[0, 0, :nar_cond_end] = 1.0
mask = (ar_q * ar_k * causal) + (nar_q * text_k) + (nar_q * nar_k)
else:
mask = (ar_q * ar_k * causal) + (nar_q * ar_k) + (nar_q * nar_k)
# Convert to additive: 0 → attend, -inf → block
attn_mask = mask.unsqueeze(1) # [B, 1, S, S]
attn_mask = attn_mask.masked_fill(attn_mask == 0, float("-inf")).masked_fill(attn_mask > 0, 0.0)
# 4. Position IDs + RoPE
position_ids = torch.arange(S, device=device).unsqueeze(0)
# 5. Forward through decoder (with MoT routing)
ar_mask_bt = ar_mask # [B, S] bool for MoT routing
hidden_states, _ = self.model(
inputs_embeds=token_emb, position_ids=position_ids,
use_cache=False, attention_mask=attn_mask, ar_mask=ar_mask_bt,
)
# 6. NAR head at content positions
nar_pred = self.llm2vae(hidden_states) # [B, S, D]
v_pred = nar_pred[0, content_indices] # [T_lat, D]
return v_pred
# Keep custom code + auto_map when a local user calls save_pretrained as well
# as when the release builder creates a Hub repository.
YuE2Config.register_for_auto_class()
YuE2ForCausalLM.register_for_auto_class("AutoModelForCausalLM")
+151643
View File
File diff suppressed because it is too large Load Diff
+9
View File
@@ -0,0 +1,9 @@
{
"schema": 1,
"files": {
"model.safetensors": {
"bytes": 7261441640,
"sha256": "1d55c42c1a9875c34f5d736e15078449992b044e807ce2a138e6cf289a1e59e9"
}
}
}
+24
View File
@@ -0,0 +1,24 @@
{
"abc": {
"temperature": 0.7,
"top_p": 0.9,
"top_k": 30,
"repetition_penalty": 1.005,
"penalty_window": 100,
"min_tokens": 32,
"max_tokens": 4096
},
"semantic": {
"temperature": 1.0,
"top_p": 0.95,
"top_k": 100,
"repetition_penalty": 1.2,
"penalty_window": 50,
"min_tokens": 200,
"max_tokens": 9000
},
"ode_steps": 32,
"ode_method": "midpoint",
"context": 24576,
"version": "yue2-native-v1"
}
Binary file not shown.
Binary file not shown.
+20 -2
View File
@@ -1,4 +1,22 @@
from comfy_api.latest import ComfyExtension, io
from typing_extensions import override
from .yue_node import YUE_SM_Model,YUE_SM_Vae, YUE_SM_Sampler,YUE_SM_Clip,YUE_SM_Cond
WEB_DIRECTORY = "./js"
class YUE_SM_Extension(ComfyExtension):
@override
async def get_node_list(self) -> list[type[io.ComfyNode]]:
return [
YUE_SM_Model,
YUE_SM_Vae,
YUE_SM_Sampler,
YUE_SM_Clip,
YUE_SM_Cond,
]
async def comfy_entrypoint() -> YUE_SM_Extension: # ComfyUI calls this to load your extension and its nodes.
return YUE_SM_Extension()
from .yue_node import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
__all__ = ['NODE_CLASS_MAPPINGS', 'NODE_DISPLAY_NAME_MAPPINGS']
Binary file not shown.

After

Width:  |  Height:  |  Size: 398 KiB

Binary file not shown.
Binary file not shown.

After

Width:  |  Height:  |  Size: 168 KiB

+498
View File
@@ -0,0 +1,498 @@
<svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:inkscape="http://www.inkscape.org/namespaces/inkscape" version="1.1" width="396" height="170.87078" viewBox="0 0 396 170.87078">
<defs>
<clipPath id="clip_1">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<path id="font_2_6" d="M.46191407 .33007813C.46191407 .27539063 .45760093 .22688802 .4489746 .18457031 .4403483 .14225261 .4272461 .10668945 .40966798 .07788086 .39208985 .049072267 .36979167 .02726237 .34277345 .012451172 .31575523-.0023600262 .28385417-.009765625 .24707031-.009765625 .17805989-.009765625 .12597656 .019205729 .09082031 .07714844 .055664064 .13509114 .038085939 .21940105 .038085939 .33007813 .038085939 .43847657 .055664064 .521403 .09082031 .5788574 .12597656 .6363118 .17936199 .66503909 .25097657 .66503909 .31998698 .66503909 .37239585 .6366374 .40820313 .579834 .4440104 .5230306 .46191407 .43977867 .46191407 .33007813M.3720703 .33032228C.3720703 .3771566 .3700358 .4189504 .3659668 .45570375 .3618978 .49245707 .35506187 .5234375 .34545899 .548645 .3358561 .57385256 .32299806 .593043 .30688478 .60621646 .29077149 .61938986 .27083335 .62597659 .24707031 .62597659 .22298177 .62597659 .20320638 .61938986 .18774414 .60621646 .1722819 .593043 .16015625 .57385256 .15136719 .548645 .14257813 .5234375 .13647461 .49245707 .13305664 .45570375 .12963867 .4189504 .12792969 .3771566 .12792969 .33032228 .12792969 .28348796 .12963867 .24145 .13305664 .20420838 .13647461 .16696675 .14257813 .13541667 .15136719 .109558109 .16015625 .08369955 .1722819 .06385803 .18774414 .05003357 .20320638 .036209108 .22298177 .029296875 .24707031 .029296875 .27083335 .029296875 .29077149 .036209108 .30688478 .05003357 .32299806 .06385803 .3358561 .08369955 .34545899 .109558109 .35506187 .13541667 .3618978 .16696675 .3659668 .20420838 .3700358 .24145 .3720703 .28348796 .3720703 .33032228Z"/>
<path id="font_2_8" d="M.44482423 0H.043945314V.071777347C.076822917 .102376308 .10701498 .12980144 .13452149 .15405274 .162028 .17830403 .18676758 .20092774 .20874024 .22192383 .23071289 .24291992 .24991863 .26302085 .26635743 .28222657 .28279624 .30143229 .2964681 .3214518 .30737306 .34228517 .31827799 .3631185 .32641603 .38557945 .3317871 .40966798 .3371582 .4337565 .33984376 .4609375 .33984376 .49121095 .33984376 .5335286 .33024089 .5657552 .31103517 .5878906 .29182945 .61002609 .26041667 .62109377 .21679688 .62109377 .20703125 .62109377 .197347 .6203613 .18774414 .6188965 .17814128 .61744186 .16894531 .61549887 .16015625 .6130676 .15136719 .61064657 .14314778 .6078949 .13549805 .6048126 .12784831 .60173037 .12109375 .5985718 .115234378 .5953369L.09814453 .515625H.06591797V.6411896C.090657558 .6470286 .114990238 .6519725 .13891602 .6560211 .1628418 .66007998 .18880208 .6621094 .21679688 .6621094 .28841148 .6621094 .34220378 .6472168 .37817384 .61743167 .4141439 .5876465 .4321289 .54557296 .4321289 .49121095 .4321289 .46451823 .42862956 .43977867 .42163087 .4169922 .41463218 .39420573 .4046224 .37231446 .39160157 .35131837 .37858073 .33032228 .36263023 .3096517 .34375 .28930665 .32486979 .26896159 .3033854 .24788411 .27929688 .22607422 .25520835 .20426433 .22875977 .18107097 .19995117 .15649414 .17114258 .13191732 .14046224 .10481771 .107910159 .07519531H.44482423V0Z"/>
<path id="font_2_11" d="M.2368164 .38378907C.27327476 .38378907 .3055013 .3801168 .3334961 .37277223 .36149089 .36542765 .38492839 .3540853 .4038086 .33874513 .4226888 .32341514 .43693034 .30391947 .4465332 .28025819 .45613609 .2565969 .4609375 .22828675 .4609375 .19532776 .4609375 .16433208 .45629884 .13627117 .44702149 .11114502 .43774415 .08601888 .4235026 .064478557 .40429688 .046524049 .38509117 .028579712 .36092124 .014709473 .3317871 .00491333 .30265299-.00487264 .26839195-.009765625 .2290039-.009765625 .19840496-.009765625 .16935222-.008051555 .1418457-.004623413 .114339198-.0011952719 .08821615 .0041097009 .06347656 .011291504L.05810547 .14941406H.09033203L.11230469 .057617189C.11816406 .05436198 .12524414 .05110677 .13354492 .047851564 .1418457 .044596357 .1508789 .041748048 .16064453 .03930664 .17041016 .036865236 .1805013 .03491211 .19091797 .033447267 .20133464 .031982423 .21142578 .03125 .2211914 .03125 .25146485 .03125 .2762858 .035079957 .2956543 .04273987 .3150228 .05039978 .33032228 .061238607 .34155274 .07525635 .3527832 .08928426 .3605143 .1060791 .3647461 .12564087 .36897788 .14520264 .37109376 .1668803 .37109376 .19067383 .37109376 .21610515 .36889649 .23827617 .36450196 .2571869 .36010743 .27609764 .35205079 .2919108 .34033204 .30462647 .32861329 .31734214 .3125814 .32687889 .29223634 .3332367 .27189128 .3395945 .24576824 .34277345 .21386719 .34277345 .19466146 .34277345 .17773438 .34147135 .16308594 .3388672 .1484375 .33626304 .13639324 .33365885 .12695313 .3310547H.080078128V.65527346H.41210938V.5800781H.12402344V.3720703C.12988281 .3733724 .13647461 .37467448 .14379883 .37597657 .15112305 .37727867 .15926107 .37849937 .16821289 .37963868 .17716472 .38077799 .18725586 .38175456 .19848633 .38256837 .2097168 .38338218 .22249349 .38378907 .2368164 .38378907Z"/>
<path id="font_2_13" d="M.09814453 .50097659H.06591797V.65527346H.4711914V.61743167L.17919922 0H.11621094L.40283204 .5800781H.115234378L.09814453 .50097659Z"/>
<path id="font_2_7" d="M.30615235 .0390625 .4399414 .025878907V0H.087890628V.025878907L.22216797 .0390625V.5727539L.08984375 .5253906V.5513611L.28076173 .66015627H.30615235V.0390625Z"/>
<path id="font_2_27" d="M.067871097 .17675781H.099609378L.11669922 .088378909C.12288412 .080566409 .13151042 .073160808 .14257813 .06616211 .15364583 .05916341 .16593425 .052978517 .17944336 .047607423 .19295247 .042236329 .20719402 .03800456 .22216797 .03491211 .23714192 .03181966 .25179038 .030273438 .26611329 .030273438 .29085288 .030273438 .31233726 .03359477 .3305664 .040237428 .34879557 .04689026 .36385093 .056055707 .37573243 .067733768 .38761393 .07941183 .39648438 .093195598 .40234376 .10908508 .40820313 .12497457 .4111328 .1423238 .4111328 .16113281 .4111328 .18457031 .4061686 .2039388 .39624024 .21923828 .38631187 .23453777 .37329103 .24747722 .35717774 .25805665 .34106446 .26863609 .3227539 .2775879 .3022461 .2849121 .28173829 .29223634 .2606608 .29964195 .23901367 .3071289 .21736653 .31461589 .19628906 .32291667 .17578125 .33203126 .15527344 .34114585 .13696289 .3527018 .12084961 .36669923 .10473633 .38069663 .09171549 .3980306 .08178711 .41870118 .07185873 .43937174 .06689453 .46484376 .06689453 .4951172 .06689453 .5205078 .071777347 .5435384 .08154297 .564209 .091308597 .5848796 .10563151 .6024577 .12451172 .61694338 .14339192 .631429 .16658528 .6425781 .1940918 .6503906 .22159831 .6582031 .2529297 .6621094 .28808595 .6621094 .3203125 .6621094 .35091148 .65999349 .3798828 .6557617 .40885417 .65152999 .43554688 .64664718 .45996095 .6411133V.5048828H.42822267L.4111328 .58496096C.3968099 .5953776 .37923179 .6040039 .35839845 .61083987 .3375651 .6176758 .3141276 .62109377 .28808595 .62109377 .2639974 .62109377 .24324544 .61848959 .22583008 .61328127 .20841472 .60807296 .19401042 .60099288 .18261719 .592041 .17122396 .5830892 .16276042 .57250979 .15722656 .56030276 .1516927 .5480957 .14892578 .53499349 .14892578 .5209961 .14892578 .49983726 .15388997 .48225913 .16381836 .46826173 .17374675 .4542643 .18676758 .44230143 .20288086 .43237306 .21899414 .42244468 .23738607 .41389976 .25805665 .40673829 .2787272 .3995768 .29988609 .39217124 .3215332 .38452149 .34318034 .37687174 .3643392 .36824546 .38500978 .35864259 .40568034 .3490397 .42407228 .33683268 .44018556 .32202149 .45629884 .3072103 .46931968 .2891439 .47924806 .26782228 .48917643 .24650066 .49414063 .22021485 .49414063 .18896485 .49414063 .1586914 .48950196 .13134766 .4802246 .106933597 .47094728 .08251953 .45694987 .061604818 .43823243 .044189454 .41951499 .026774088 .39607749 .013427734 .36791993 .0041503908 .33976237-.005126953 .30664063-.009765625 .2685547-.009765625 .24837239-.009765625 .22851563-.008783977 .20898438-.0068206789 .18945313-.0048675539 .1710612-.0022583008 .1538086 .0010070801 .13655599 .004272461 .12060547 .008026123 .10595703 .012268066 .091308597 .01651001 .07861328 .020751954 .067871097 .024993897V.17675781Z"/>
<path id="font_2_45" d="M.46191407 .23217774C.46191407 .15429688 .4444987 .09450277 .40966798 .05279541 .37483726 .011088054 .32063804-.009765625 .24707031-.009765625 .17805989-.009765625 .12597656 .010925293 .09082031 .05230713 .055664064 .093688968 .038085939 .15364583 .038085939 .23217774 .038085939 .30973307 .055664064 .3690389 .09082031 .4100952 .12597656 .45115153 .17936199 .4716797 .25097657 .4716797 .32063804 .4716797 .37320964 .45155845 .4086914 .41131593 .4441732 .3710734 .46191407 .3113607 .46191407 .23217774M.37402345 .23242188C.37402345 .2639974 .37190757 .29223634 .36767579 .31713868 .363444 .34204103 .35636393 .3630371 .34643556 .38012696 .33650718 .3972168 .32340495 .41023765 .3071289 .41918946 .29085288 .42814128 .27083335 .4326172 .24707031 .4326172 .22298177 .4326172 .203125 .42814128 .1875 .41918946 .171875 .41023765 .1595052 .3972168 .15039063 .38012696 .14127605 .3630371 .13492839 .34204103 .13134766 .31713868 .12776692 .29223634 .12597656 .2639974 .12597656 .23242188 .12597656 .20052083 .12776692 .17203777 .13134766 .14697266 .13492839 .121907558 .14127605 .10066732 .15039063 .08325195 .1595052 .065836589 .171875 .052490236 .1875 .04321289 .203125 .033935548 .22298177 .029296875 .24707031 .029296875 .27083335 .029296875 .29085288 .033935548 .3071289 .04321289 .32340495 .052490236 .33650718 .065836589 .34643556 .08325195 .35636393 .10066732 .363444 .121907558 .36767579 .14697266 .37190757 .17203777 .37402345 .20052083 .37402345 .23242188Z"/>
<path id="font_2_44" d="M.15820313 .42236329C.1673177 .42757163 .17814128 .43310548 .19067383 .43896485 .20320638 .44482423 .2163086 .4501953 .22998047 .45507813 .24365235 .45996095 .25732423 .46394859 .2709961 .46704103 .28466798 .47013346 .29736329 .4716797 .30908204 .4716797 .32666017 .4716797 .34277345 .46923319 .35742188 .4643402 .3720703 .4594574 .38468424 .4516398 .39526368 .44088746 .4058431 .4301351 .4141439 .41612245 .42016603 .3988495 .42618815 .38157655 .42919923 .36071778 .42919923 .3362732V.034179689L.48486329 .021972657V0H.28710938V.021972657L.34814454 .034179689V.32714845C.34814454 .35416667 .34155274 .3754069 .32836915 .39086915 .31518556 .4063314 .29475913 .4140625 .26708985 .4140625 .25797526 .4140625 .24837239 .41358949 .23828125 .41264344 .22819011 .41170756 .21826172 .41060893 .2084961 .40934754 .19873047 .40808616 .1895345 .4065908 .1809082 .40486146 .1722819 .4031423 .16503906 .401652 .15917969 .40039063V.034179689L.2211914 .021972657V0H.022949219V.021972657L.078125 .034179689V.4248047L.022949219 .43701173V.45898438H.1538086L.15820313 .42236329Z"/>
<path id="font_2_38" d="M.4248047 .31445313C.4248047 .26171876 .40901695 .22184246 .3774414 .19482422 .34586589 .16780599 .30045573 .15429688 .24121094 .15429688 .22786458 .15429688 .21443685 .1550293 .20092774 .15649414 .18741863 .15795899 .17610677 .15966797 .16699219 .1616211L.13623047 .09729004C.13720703 .09172567 .14355469 .08648682 .15527344 .08157349 .16699219 .07667033 .18164063 .07421875 .19921875 .07421875H.33496095C.3844401 .07421875 .42114259 .06347656 .44506837 .041992189 .46899415 .020507813 .48095704-.009114583 .48095704-.046875 .48095704-.06803385 .4766439-.088785808 .46801759-.10913086 .45939128-.1294759 .4453939-.14754232 .4260254-.16333008 .4066569-.17911785 .38134767-.19181316 .35009767-.20141602 .31884767-.21101888 .2804362-.21582031 .23486328-.21582031 .20003255-.21582031 .17041016-.212972 .1459961-.20727539 .12158203-.20157878 .10172526-.19368489 .08642578-.18359375 .071126308-.17350261 .060058595-.16178386 .053222658-.1484375 .04638672-.13509114 .04296875-.12060547 .04296875-.10498047 .04296875-.09423828 .045003255-.08406576 .049072267-.07446289 .053141279-.06486002 .05883789-.055582685 .06616211-.04663086 .07348633-.037679037 .08219401-.028971354 .092285159-.020507813 .102376308-.0120442709 .11328125-.0035807293 .125 .0048828127 .119140628 .00684611 .11238607 .01003011 .10473633 .014434814 .097086589 .018839518 .08984375 .02447001 .08300781 .031326295 .076171878 .038182577 .07047526 .0460968 .06591797 .05506897 .061360677 .06405131 .05908203 .07409159 .05908203 .08518982L.13623047 .17236328C.08479818 .19645183 .05908203 .24381511 .05908203 .31445313 .05908203 .34016929 .063313808 .36287437 .071777347 .38256837 .08024088 .40226237 .09236654 .41870118 .1081543 .43188478 .123942058 .44506837 .14339192 .45499674 .1665039 .46166993 .18961589 .4683431 .21582031 .4716797 .24511719 .4716797 .25423179 .4716797 .26350913 .4711914 .27294923 .47021485 .2823893 .46923829 .29125978 .46818034 .29956056 .46704103 .30786134 .4659017 .31510417 .4645996 .32128907 .46313478 .32747398 .46166993 .33203126 .46044923 .33496095 .45947267L.4428711 .5136719 .45996095 .49267579 .39208985 .42236329C.40315757 .4099935 .41137696 .39444987 .41674806 .37573243 .42211915 .35701499 .4248047 .33658854 .4248047 .31445313M.40478517-.06201172C.40478517-.04345703 .39908854-.028971354 .3876953-.018554688 .3763021-.0081380209 .35904948-.0029296876 .3359375-.0029296876H.15820313C.15136719-.0087890629 .1451009-.01554362 .1394043-.02319336 .13370769-.0308431 .12882488-.03889974 .12475586-.04736328 .12068685-.055826826 .11751302-.06437174 .115234378-.07299805 .11295573-.08162435 .111816409-.09000651 .111816409-.09814453 .111816409-.10986328 .11368815-.120524089 .11743164-.13012696 .12117513-.13972982 .12768555-.14794922 .13696289-.15478516 .14624024-.1616211 .15885417-.16699219 .17480469-.17089844 .1907552-.17480469 .21077474-.17675781 .23486328-.17675781 .26416017-.17675781 .2894694-.1739095 .31079103-.16821289 .33211265-.16251628 .34977214-.15454102 .36376954-.14428711 .37776695-.1340332 .3881022-.121907558 .3947754-.107910159 .40144859-.09391276 .40478517-.07861328 .40478517-.06201172M.2421875 .19140625C.27766929 .19140625 .30281578 .20157878 .31762696 .22192383 .33243815 .24226888 .33984376 .27311198 .33984376 .31445313 .33984376 .33496095 .33813478 .3527832 .3347168 .36791993 .33129884 .38305665 .32576499 .3955078 .31811524 .40527345 .31046549 .41503907 .30045573 .42236329 .28808595 .4272461 .27571617 .4321289 .2607422 .4345703 .24316406 .4345703 .22526042 .4345703 .21004232 .4321289 .19750977 .4272461 .18497722 .42236329 .17472331 .41503907 .16674805 .40527345 .15877278 .3955078 .1529948 .38305665 .14941406 .36791993 .14583333 .3527832 .14404297 .33496095 .14404297 .31445313 .14404297 .2939453 .14583333 .2759603 .14941406 .26049806 .1529948 .24503581 .1586914 .23217774 .1665039 .22192383 .1743164 .21166992 .18440755 .20402019 .19677735 .19897461 .20914714 .19392903 .22428386 .19140625 .2421875 .19140625Z"/>
<path id="font_2_3" d="M0 0Z"/>
<path id="font_2_47" d="M.39697267 .4814453H.43115235V-.17773438L.4819336-.18962097V-.21289063H.28710938V-.18962097L.35009767-.17773438V-.050598146C.35009767-.0382487 .3504232-.025009156 .35107423-.010879517 .35172526 .0032399495 .35302735 .017120362 .35498048 .030761719 .34033204 .01936849 .32218424 .009765625 .3005371 .001953125 .27888999-.005859375 .25341798-.009765625 .2241211-.009765625 .15999349-.009765625 .11263021 .010681152 .08203125 .051574708 .051432294 .09246826 .036132814 .15136719 .036132814 .22827149 .036132814 .26574708 .040364583 .29955546 .048828126 .32969667 .057291669 .35983787 .07006836 .38541667 .0871582 .4064331 .10424805 .42744956 .12589519 .4435781 .15209961 .45481874 .17830403 .46605937 .20898438 .4716797 .24414063 .4716797 .2607422 .4716797 .2787272 .47070313 .2980957 .46875 .3174642 .46679688 .33577476 .46402995 .35302735 .46044923L.39697267 .4814453M.12402344 .22825623C.12402344 .1950124 .12687175 .16657512 .13256836 .14294434 .13826497 .11931356 .14648438 .09991964 .15722656 .08476257 .16796875 .069615688 .18123372 .0585378 .19702149 .05152893 .21280925 .04452006 .23079427 .041015626 .25097657 .041015626 .25878907 .041015626 .2672526 .041341146 .2763672 .041992189 .28548179 .04264323 .29451499 .043701173 .3034668 .045166017 .3124186 .04663086 .32096354 .048339845 .32910157 .05029297 .3372396 .052246095 .34423829 .05452474 .35009767 .057128908V.42332459C.33642579 .42593894 .32063804 .42797853 .30273438 .42944337 .28483073 .4309082 .26757813 .43164063 .25097657 .43164063 .20800781 .43164063 .17610677 .41419984 .15527344 .37931825 .13444011 .3444468 .12402344 .2940928 .12402344 .22825623Z"/>
<path id="font_2_51" d="M.15283203 .13085938C.15283203 .10384115 .15893555 .083089198 .17114258 .068603519 .18334961 .05411784 .20328777 .046875 .23095703 .046875 .2491862 .046875 .2680664 .048095704 .28759767 .05053711 .3071289 .052978517 .32600913 .056803388 .34423829 .06201172V.4248047L.27490235 .43701173V.45898438H.4248047V.034179689L.48291017 .021972657V0H.3491211L.34521485 .037109376C.33577476 .031901044 .32454429 .026529948 .31152345 .020996094 .2985026 .015462239 .2849121 .010416667 .27075196 .005859375 .2565918 .0013020834 .24235027-.0024414063 .22802735-.0053710939 .21370442-.008300781 .2006836-.009765625 .18896485-.009765625 .17138672-.009765625 .1554362-.0073242189 .14111328-.0024414063 .12679036 .0024414063 .11450195 .010253906 .10424805 .020996094 .09399414 .03173828 .08601888 .045654298 .080322269 .06274414 .07462565 .079833988 .071777347 .10058594 .071777347 .125V.4248047L.013183594 .43701173V.45898438H.15283203V.13085938Z"/>
<path id="font_2_32" d="M.22705078 .46972657C.24788411 .46972657 .2672526 .46776835 .28515626 .46385194 .3030599 .45994569 .31852214 .45326744 .33154298 .44381715 .3445638 .43436686 .35473634 .42157493 .36206056 .40544129 .36938478 .38931785 .37304688 .3690338 .37304688 .34458924V.034179689L.43017579 .021972657V0H.30419923L.29492188 .045898439C.29003907 .041015626 .28344728 .03531901 .27514649 .028808594 .2668457 .022298178 .25683595 .016194663 .24511719 .010498047 .23339844 .004801432 .21980794 0 .2043457-.00390625 .18888347-.0078125 .17171224-.009765625 .15283203-.009765625 .13069661-.009765625 .11206055-.00634257 .09692383 .00050354006 .08178711 .00734965 .06966146 .016886393 .060546876 .02911377 .051432294 .04135132 .044921876 .055862428 .041015626 .072647098 .037109376 .08944193 .03515625 .10762533 .03515625 .12719727 .03515625 .14741008 .037597658 .16493225 .04248047 .1797638 .04736328 .19460552 .05419922 .20708211 .06298828 .2171936 .071777347 .2273051 .08211263 .23553975 .09399414 .24189759 .10587565 .24825542 .11873373 .2532247 .13256836 .25680543 .146403 .26039634 .16105144 .2628428 .17651367 .2641449 .1919759 .26545716 .20751953 .26627604 .22314453 .26660157L.2919922 .2685547V.34033204C.2919922 .3540039 .29085288 .36645509 .28857423 .37768556 .28629557 .38891603 .2824707 .39860026 .2770996 .40673829 .27172853 .4148763 .2644857 .42122398 .2553711 .42578126 .24625652 .43033854 .23486328 .4326172 .2211914 .4326172 .2055664 .4326172 .18977864 .4305013 .17382813 .42626954 .15787761 .42203776 .1438802 .4165039 .13183594 .40966798L.115234378 .35253907H.087890628V.45263673C.10904948 .457194 .13094075 .46118165 .15356446 .4645996 .17618816 .46801759 .2006836 .46972657 .22705078 .46972657M.2919922 .234375 .22802735 .23242188C.20882161 .23177083 .19222005 .229894 .17822266 .22679138 .16422527 .22368877 .15266927 .21838379 .14355469 .21087647 .13444011 .20336914 .12760417 .1930898 .123046878 .18003845 .118489589 .1669871 .11621094 .15033977 .11621094 .13009644 .11621094 .07266235 .13948567 .043945314 .18603516 .043945314 .20817058 .043945314 .22729492 .046468099 .2434082 .051513673 .25952149 .056559247 .27571617 .06298828 .2919922 .07080078V.234375Z"/>
<path id="font_2_42" d="M.17919922 .034179689 .2578125 .021972657V0H.020019532V.021972657L.09814453 .034179689V.66015627L.020019532 .67204287V.69433596H.17919922V.034179689Z"/>
<path id="font_2_40" d="M.1850586 .6086426C.1850586 .6014506 .18367513 .5946655 .1809082 .58828738 .17814128 .5819092 .1743164 .5762685 .1694336 .57136538 .16455078 .5664622 .15885417 .562617 .15234375 .5598297 .14583333 .5570526 .13899739 .55566409 .13183594 .55566409 .12467448 .55566409 .11791992 .5570526 .111572269 .5598297 .10522461 .562617 .099609378 .5664622 .09472656 .57136538 .08984375 .5762685 .08601888 .5819092 .08325195 .58828738 .08048502 .5946655 .07910156 .6014506 .07910156 .6086426 .07910156 .61583456 .08048502 .622701 .08325195 .62924197 .08601888 .6357829 .08984375 .64150497 .09472656 .6464081 .099609378 .6513112 .10522461 .65515139 .111572269 .65792849 .11791992 .66071578 .12467448 .6621094 .13183594 .6621094 .13899739 .6621094 .14583333 .66071578 .15234375 .65792849 .15885417 .65515139 .16455078 .6513112 .1694336 .6464081 .1743164 .64150497 .17814128 .6357829 .1809082 .62924197 .18367513 .622701 .1850586 .61583456 .1850586 .6086426M.18017578 .034179689 .25878907 .021972657V0H.020996094V.021972657L.099121097 .034179689V.4248047L.034179689 .43701173V.45898438H.18017578V.034179689Z"/>
<path id="font_2_50" d="M.16308594-.009765625C.13183594-.009765625 .10847982-.00048828127 .09301758 .018066407 .077555339 .036621095 .06982422 .06266276 .06982422 .096191409V.41796876H.009765625V.4399414L.07080078 .45898438 .12011719 .56347659H.1508789V.45898438H.25585938V.41796876H.1508789V.10498047C.1508789 .08382162 .15568035 .067871097 .1652832 .057128908 .17488607 .04638672 .1875 .041015626 .203125 .041015626 .21516927 .041015626 .22713216 .041829427 .23901367 .04345703 .25089518 .045084638 .2618815 .046875 .27197267 .048828126V.017089844C.26708985 .013834636 .2606608 .010579427 .25268556 .0073242189 .24471028 .0040690104 .23592122 .0012207031 .22631836-.0012207031 .2167155-.0036621094 .20654297-.0056966149 .19580078-.0073242189 .1850586-.008951823 .17415364-.009765625 .16308594-.009765625Z"/>
<path id="font_2_54" d="M.4482422 .4267578 .26904298-.028808594C.2586263-.05517578 .24804688-.0797526 .23730469-.10253906 .2265625-.12532552 .21459961-.1451009 .20141602-.16186524 .18823242-.17862956 .17317708-.19181316 .15625-.20141602 .13932292-.21101888 .119628909-.21582031 .09716797-.21582031 .08870443-.21582031 .08129883-.21565755 .07495117-.21533203 .068603519-.21500652 .06266276-.21443685 .057128908-.21362305 .05159505-.21280925 .0460612-.21191406 .040527345-.2109375 .03499349-.20996094 .028808594-.20865886 .021972657-.20703125V-.107910159H.044921876L.061035158-.15478516C.071126308-.16227214 .085123699-.16601563 .103027347-.16601563 .11702474-.16601563 .12972005-.16292317 .14111328-.15673828 .15250652-.15056356 .16300456-.14186096 .17260742-.1306305 .18221028-.119410198 .19091797-.106155399 .19873047-.09086609 .20654297-.07557678 .21402996-.05865987 .2211914-.040115358L.23388672-.004989624 .05908203 .4248047 .012207031 .43701173V.45898438H.22509766V.43701173L.15283203 .42382813 .27685548 .103027347 .39697267 .4248047 .3251953 .43701173V.45898438H.49609376V.43701173L.4482422 .4267578Z"/>
<path id="font_2_35" d="M.35302735 .034179689C.33870445 .022786459 .32088218 .012613933 .29956056 .0036621094 .27823893-.0052897136 .25309245-.009765625 .2241211-.009765625 .09879557-.009765625 .036132814 .068603519 .036132814 .2253418 .036132814 .26345826 .040283204 .29774986 .048583986 .32821656 .056884767 .35869346 .06966146 .38460288 .08691406 .40594483 .104166667 .42728678 .12597656 .4435781 .15234375 .45481874 .17871094 .46605937 .20996094 .4716797 .24609375 .4716797 .2626953 .4716797 .28035484 .47070313 .29907228 .46875 .3177897 .46679688 .33577476 .46402995 .35302735 .46044923 .3523763 .46402995 .35188804 .46930949 .3515625 .47628785 .35123698 .48327638 .35099284 .49074809 .35083009 .498703 .35066734 .5066579 .35050456 .51453146 .3503418 .5223236 .35017906 .5301158 .35009767 .5364482 .35009767 .5413208V.66015627L.27294923 .67204287V.69433596H.43115235V.034179689L.48779298 .021972657V0H.35888673L.35302735 .034179689M.12402344 .22532654C.12402344 .19110616 .1270345 .16226197 .13305664 .13879395 .13907878 .11532593 .1476237 .096338909 .1586914 .081832889 .16975911 .067337039 .18310547 .0569102 .19873047 .05055237 .21435547 .04419454 .23177083 .041015626 .25097657 .041015626 .2705078 .041015626 .28889976 .04288737 .30615235 .04663086 .32340495 .050374349 .33805339 .05485026 .35009767 .060058595V.42332459C.33642579 .42593894 .32063804 .42797853 .30273438 .42944337 .28483073 .4309082 .26757813 .43164063 .25097657 .43164063 .2076823 .43164063 .17569988 .41419984 .1550293 .37931825 .13435872 .3444468 .12402344 .29311625 .12402344 .22532654Z"/>
<path id="font_2_36" d="M.12695313 .23144531V.22264099C.12695313 .19881694 .12866211 .17596944 .13208008 .15409851 .13549805 .13222759 .14233399 .11288961 .15258789 .096084598 .1628418 .07927958 .1772461 .06589762 .19580078 .05593872 .21435547 .04598999 .23876953 .041015626 .26904298 .041015626 .2788086 .041015626 .2890625 .041422529 .2998047 .042236329 .31054688 .04305013 .32128907 .044108076 .33203126 .045410158 .34277345 .04671224 .3531901 .048177083 .36328126 .049804689 .3733724 .051432294 .38264976 .053222658 .39111329 .05517578V.027832032C.3836263 .022949219 .3745931 .018310547 .36401368 .013916016 .35343424 .009521484 .34179688 .005533854 .32910157 .001953125 .31640626-.0016276041 .30289714-.0044759118 .28857423-.006591797 .2742513-.008707683 .25976563-.009765625 .24511719-.009765625 .20703125-.009765625 .17488607-.0045522057 .14868164 .005874634 .12247721 .016301474 .10123698 .031778974 .08496094 .05230713 .0686849 .07283529 .056966146 .09825134 .049804689 .1285553 .04264323 .15885926 .0390625 .19372559 .0390625 .2331543 .0390625 .3133138 .055826826 .3731079 .08935547 .41253663 .12288412 .45196534 .17073567 .4716797 .23291016 .4716797 .25732423 .4716797 .28019206 .46842448 .30151368 .46191407 .3228353 .45540367 .34147135 .4444987 .35742188 .42919923 .3733724 .41389976 .38598634 .39339195 .39526368 .36767579 .40454103 .34195964 .4091797 .30989585 .4091797 .27148438V.23144531H.12695313M.23291016 .4326172C.21468099 .4326172 .19897461 .42879234 .18579102 .42114259 .17260742 .41349284 .16170247 .40266929 .15307617 .38867188 .14444988 .37467448 .13810222 .35766603 .1340332 .33764649 .12996419 .31762696 .12792969 .2952474 .12792969 .2705078H.32421876C.32421876 .2952474 .3228353 .31762696 .32006837 .33764649 .31730143 .35766603 .3124186 .37467448 .30541993 .38867188 .29842124 .40266929 .2890625 .41349284 .27734376 .42114259 .265625 .42879234 .2508138 .4326172 .23291016 .4326172Z"/>
<path id="font_2_53" d="M.43408204 .032714845 .48779298 .02230835V0H.27978517V.02229309L.3408203 .033691408 .23486328 .195755 .110839847 .032714845 .17382813 .02230835V0H.0087890629V.022338868L.06201172 .030273438 .21289063 .22898865 .080078128 .4248047 .025878907 .43701173V.45898438H.23388672V.43701173L.17285156 .42382813 .26123048 .2922058 .36279298 .4248047 .2998047 .43701173V.45898438H.46484376V.43701173L.41210938 .4267578 .28320313 .25976563 .43408204 .032714845Z"/>
<path id="font_2_55" d="M.2290039 .45361329C.2179362 .4441732 .20442708 .4346517 .18847656 .42504884 .17252605 .41544596 .15397136 .40559898 .1328125 .3955078V.43066407C.15494792 .44954429 .1751302 .4695638 .19335938 .49072267 .21158855 .51188156 .22753906 .5358073 .24121094 .5625H.25878907C.27246095 .5358073 .28841148 .51188156 .30664063 .49072267 .32486979 .4695638 .3450521 .44954429 .3671875 .43066407V.3955078C.34602867 .40559898 .32747398 .41544596 .31152345 .42504884 .2955729 .4346517 .2820638 .4441732 .2709961 .45361329V-.029296875H.2290039V.45361329Z"/>
<path id="font_2_28" d="M.1538086 0V.025878907L.2578125 .0390625V.61328127H.23291016C.19026692 .61328127 .15445964 .61230978 .12548828 .6103668 .09651693 .6084239 .07600912 .6061554 .063964847 .6035614L.05078125 .5019531H.018066407V.65527346H.5942383V.5019531H.56103518L.54785159 .6035614C.5419922 .60485336 .5332845 .6059825 .5217285 .60694888 .51017257 .6079254 .49674479 .6088155 .4814453 .60961917 .46614585 .6104329 .4494629 .611084 .43139649 .61157229 .41333009 .61206057 .39485679 .6123047 .37597657 .6123047H.35205079V.0390625L.4560547 .025878907V0H.1538086Z"/>
<path id="font_2_43" d="M.15917969 .421875C.16829427 .4271342 .17911785 .4327189 .19165039 .43862916 .20418294 .44454957 .21712239 .4499766 .23046875 .45491029 .24381511 .45984397 .25732423 .46387229 .2709961 .46699525 .28466798 .4701182 .29736329 .4716797 .30908204 .4716797 .33154298 .4716797 .35229493 .46740724 .3713379 .4588623 .39038087 .45032756 .4046224 .43669639 .4140625 .41796876 .42447917 .423879 .43701173 .43003846 .45166017 .43644715 .4663086 .44285584 .4815267 .4486847 .49731446 .45393373 .51310226 .4591929 .52872726 .46346537 .54418948 .4667511 .5596517 .47003684 .5735677 .4716797 .5859375 .4716797 .6035156 .4716797 .6194661 .46923319 .63378909 .4643402 .648112 .4594574 .6604004 .4516398 .6706543 .44088746 .6809082 .4301351 .6888835 .41612245 .6945801 .3988495 .7002767 .38157655 .703125 .36071778 .703125 .3362732V.034179689L.76220706 .021972657V0H.55371096V.021972657L.6220703 .034179689V.32714845C.6220703 .35416667 .6159668 .3749186 .60375979 .3894043 .59155276 .40388999 .57161459 .4111328 .5439453 .4111328 .53548178 .4111328 .52563479 .41048179 .5144043 .4091797 .5031738 .4078776 .49194337 .40641276 .4807129 .40478517 .46948243 .40315757 .45874024 .4012858 .44848634 .39916993 .43823243 .39705406 .4296875 .39534507 .42285157 .39404298 .4283854 .37646485 .43115235 .35709635 .43115235 .3359375V.034179689L.5 .021972657V0H.28222657V.021972657L.35009767 .034179689V.32714845C.35009767 .35416667 .34318034 .3749186 .3293457 .3894043 .31551109 .40388999 .29475913 .4111328 .26708985 .4111328 .25797526 .4111328 .24845378 .41064454 .23852539 .40966798 .228597 .4086914 .21883138 .4075521 .20922852 .40625 .19962566 .4049479 .19051107 .4034017 .18188477 .40161134 .17325847 .39982096 .16601563 .39827476 .16015625 .39697267V.034179689L.2290039 .021972657V0H.020996094V.021972657L.07910156 .034179689V.4248047L.020996094 .43701173V.45898438H.15527344L.15917969 .421875Z"/>
<clipPath id="clip_3">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_4">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_5">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_6">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_7">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_8">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_9">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_10">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_11">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_12">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_13">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_14">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_15">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_16">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_17">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_18">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<clipPath id="clip_19">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74003V132.17078H39.6Z"/>
</clipPath>
<path id="font_2_17" d="M.46777345 .49635316C.46777345 .51559957 .46492515 .5323944 .45922853 .5467377 .4535319 .5610911 .44433595 .5730794 .43164063 .58270266 .4189453 .59232589 .4025065 .59950259 .38232423 .6042328 .36214195 .608963 .33740235 .6113281 .30810548 .6113281H.20703125V.36328126H.31396485C.34423829 .36328126 .36930339 .36646018 .38916017 .372818 .40901695 .3791758 .4247233 .3881429 .4362793 .39971925 .4478353 .41130577 .4559733 .4252523 .46069337 .44155885 .46541343 .4578654 .46777345 .47613017 .46777345 .49635316M.51708987 .18652344C.51708987 .20930989 .51342776 .2290039 .5061035 .24560547 .4987793 .26220704 .48738609 .2759603 .47192384 .28686524 .45646159 .29777018 .43652345 .3059082 .41210938 .3112793 .3876953 .3166504 .35839845 .31933595 .32421876 .31933595H.20703125V.043945314C.2220052 .04329427 .2376302 .04280599 .25390626 .04248047 .26790367 .041829427 .28336589 .041422529 .30029298 .041259767 .31722007 .041097005 .33398438 .041015626 .35058595 .041015626 .3811849 .041015626 .4070638 .044433595 .42822267 .05126953 .4493815 .05810547 .46655274 .06778971 .47973634 .080322269 .49291993 .09285482 .5024414 .108072917 .5083008 .12597656 .51416018 .1438802 .51708987 .1640625 .51708987 .18652344M.028808594 0V.025878907L.11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.328125C.37402345 .65527346 .41227214 .651769 .4428711 .64476016 .47347007 .6377513 .49804688 .6275635 .51660159 .6141968 .53515627 .6008301 .54833987 .58461 .55615237 .5655365 .56396487 .546463 .5678711 .5250244 .5678711 .5012207 .5678711 .48068238 .5645345 .46193443 .5578613 .4449768 .5511882 .4280192 .5419108 .4131012 .5300293 .40022279 .51814779 .38734437 .50390627 .3765869 .4873047 .36795045 .47070313 .35931397 .45263673 .35287477 .43310548 .3486328 .46142579 .34570313 .48706056 .3400065 .51000979 .33154298 .532959 .32307945 .5524089 .3120931 .5683594 .29858399 .5843099 .28507487 .5965983 .26904298 .6052246 .25048829 .6138509 .2319336 .61816409 .21126302 .61816409 .18847656 .61816409 .16015625 .6133626 .13419597 .60375979 .1105957 .5941569 .086995448 .57910159 .06681315 .55859377 .050048829 .53808596 .033284505 .51171877 .020263672 .4794922 .010986328 .44726563 .0017089844 .40836589-.0029296876 .36279298-.0029296876 .32633464-.0029296876 .29003907-.0024414063 .25390626-.0014648438 .21777344-.00048828127 .18440755 0 .1538086 0H.028808594Z"/>
<path id="font_2_31" d="M.4091797 .25801087V.0390625L.5131836 .025878907V0H.2109375V.025878907L.3149414 .0390625V.25506593L.08496094 .61621096 .011230469 .6290741V.65527346H.28808595V.6290741L.20019531 .61621096 .3881836 .31419374 .56689456 .61621096 .48388673 .6290741V.65527346H.69677737V.6290741L.625 .61621096 .4091797 .25801087Z"/>
<path id="font_2_20" d="M.028808594 .025878907 .11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.52001956V.49902345H.48779298L.47216798 .6045227C.4617513 .60581466 .44913737 .60694888 .43432618 .6079254 .41951499 .608902 .40445964 .60962936 .38916017 .6101074 .3738607 .6105957 .359375 .6109212 .34570313 .611084 .33203126 .61124679 .3214518 .6113281 .31396485 .6113281H.20703125V.35546876H.38378907L.39892579 .43359376H.43017579V.23242188H.39892579L.38378907 .31152345H.20703125V.043945314H.3359375C.35611979 .043945314 .37516276 .044189454 .3930664 .044677736 .41097007 .045166017 .42708335 .045735677 .44140626 .04638672 .45572917 .04703776 .46801759 .047851564 .47827149 .048828126 .4885254 .049804689 .49609376 .05078125 .50097659 .051757814L.5288086 .17285156H.56103518L.5517578 0H.028808594V.025878907Z"/>
<path id="font_2_19" d="M.5800781 .3322754C.5800781 .38049317 .57340499 .42211405 .5600586 .45713807 .5467122 .49216209 .5275879 .52107748 .50268557 .5438843 .4777832 .5666911 .4477539 .5836334 .41259767 .5947113 .3774414 .6057892 .33821617 .6113281 .29492188 .6113281H.20703125V.045898439C.21679688 .045247396 .2273763 .044759115 .23876953 .044433595 .25016276 .044108076 .2618815 .043701173 .27392579 .04321289 .28597007 .04272461 .2980957 .04239909 .31030274 .042236329 .32250978 .042073568 .33447267 .041992189 .3461914 .041992189 .38753257 .041992189 .42293296 .04810079 .45239259 .060317995 .4818522 .072535198 .50602218 .090779628 .52490237 .11505127 .54378256 .13932292 .55769857 .1695404 .5666504 .20570374 .57560226 .24186707 .5800781 .28405763 .5800781 .3322754M.32617188 .65527346C.38053385 .65527346 .4296061 .6495717 .47338868 .63816836 .5171712 .62676498 .5545247 .608195 .5854492 .5824585 .6163737 .5567322 .6402181 .52334597 .6569824 .4822998 .67374679 .44125367 .6821289 .39092 .6821289 .33129884 .6821289 .2778727 .67529299 .23047383 .6616211 .18910218 .6479492 .14773052 .6272786 .11287435 .5996094 .08453369 .5719401 .056193036 .537028 .0346934 .49487306 .02003479 .4527181 .00537618 .40315757-.001953125 .3461914-.001953125 .32763673-.001953125 .30737306-.0018717448 .2854004-.0017089844 .26342774-.001546224 .24186199-.0013020834 .22070313-.0009765625 .19954427-.0006510417 .17944336-.00040690104 .16040039-.00024414063 .14135742-.00008138021 .12548828 0 .11279297 0H.028808594V.025878907L.11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.32617188Z"/>
<path id="font_2_37" d="M.10986328 .41796876H.030761719V.44189454L.10986328 .4609375V.49316407C.10986328 .52766928 .11336263 .5580241 .12036133 .5842285 .12736003 .6104329 .13745117 .6324056 .15063477 .6501465 .16381836 .6678874 .17993164 .6813151 .19897461 .6904297 .21801758 .69954428 .23942058 .70410159 .2631836 .70410159 .27783204 .70410159 .29085288 .70320639 .3022461 .701416 .3136393 .6996257 .32389323 .6974284 .3330078 .6948242V.59472659H.30908204L.28710938 .65478518C.28190104 .65804037 .27620445 .6605632 .27001954 .6623535 .26383464 .66414389 .2561849 .66503909 .24707031 .66503909 .23567708 .66503909 .22639974 .6625163 .21923828 .6574707 .21207683 .6524251 .2063802 .6446126 .20214844 .6340332 .19791667 .6234538 .19498699 .61002609 .19335938 .59375 .19173177 .57747396 .19091797 .5579427 .19091797 .53515627V.45898438H.31298829V.41796876H.19091797V.038085939L.29003907 .021972657V0H.041992189V.021972657L.10986328 .038085939V.41796876Z"/>
<path id="font_2_26" d="M.20703125 .28710938V.0390625L.30615235 .025878907V0H.03515625V.025878907L.11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.31152345C.35742188 .65527346 .39575196 .65119937 .42651368 .64305117 .4572754 .63490298 .4819336 .6232503 .5004883 .60809329 .51904299 .5929362 .53222659 .5746002 .54003909 .5530853 .54785159 .53157046 .5517578 .5072835 .5517578 .4802246 .5517578 .455129 .5480957 .4327189 .5407715 .41299439 .53344729 .39326988 .5236003 .37591044 .51123049 .36091615 .4988607 .34592185 .48453776 .3334554 .46826173 .32351686 .4519857 .31357829 .4350586 .30599977 .41748048 .30078126L.59472659 .0390625 .66552737 .025878907V0H.50878909L.32470704 .28710938H.20703125M.45458985 .47338868C.45458985 .4994812 .4514974 .5213318 .4453125 .5389404 .4391276 .5565491 .42944337 .57072958 .41625978 .58148196 .40307618 .59224447 .38614909 .5999095 .36547853 .6044769 .34480796 .6090444 .31982423 .6113281 .29052735 .6113281H.20703125V.3310547H.29345704C.32340495 .3310547 .3486328 .33374534 .36914063 .3391266 .38964845 .34450785 .40625 .35298667 .4189453 .364563 .43164063 .3761393 .44075523 .39089457 .44628907 .40882875 .4518229 .4267629 .45458985 .44828288 .45458985 .47338868Z"/>
<path id="font_2_39" d="M.15917969 .49560548C.15917969 .4910482 .15909831 .4855143 .15893555 .4790039 .15877278 .4724935 .15861003 .46573893 .15844727 .45874024 .1582845 .45174156 .15795899 .44498698 .1574707 .43847657 .15698242 .43196617 .15641277 .42659507 .15576172 .42236329 .1648763 .42757163 .17594402 .43310548 .18896485 .43896485 .20198567 .44482423 .21557617 .4501953 .22973633 .45507813 .24389649 .45996095 .25805665 .46394859 .2722168 .46704103 .28637696 .47013346 .2993164 .4716797 .31103517 .4716797 .32861329 .4716797 .34472657 .46923319 .359375 .4643402 .37402345 .4594574 .38663737 .4516398 .3972168 .44088746 .40779624 .4301351 .41609703 .41612245 .42211915 .3988495 .42814128 .38157655 .43115235 .36071778 .43115235 .3362732V.034179689L.4868164 .021972657V0H.2890625V.021972657L.35009767 .034179689V.33007813C.35009767 .35709635 .34350587 .3778483 .33032228 .39233399 .31713868 .40681968 .29671226 .4140625 .26904298 .4140625 .25992839 .4140625 .25024415 .41358949 .23999024 .41264344 .22973633 .41170756 .2195638 .41060893 .20947266 .40934754 .19938152 .40808616 .1899414 .4065908 .18115235 .40486146 .17236328 .4031423 .16503906 .401652 .15917969 .40039063V.034179689L.2211914 .021972657V0H.020019532V.021972657L.078125 .034179689V.66015627L.009765625 .67204287V.69433596H.15917969V.49560548Z"/>
<path id="font_2_23" d="M.42089845 0H.4038086L.1640625 .5629883V.0390625L.25195313 .025878907V0H.028808594V.025878907L.11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.22705078L.4399414 .15686035 .6723633 .65527346H.8598633V.6290741L.7758789 .61621096V.0390625L.8598633 .025878907V0H.5942383V.025878907L.6821289 .0390625V.5629883L.42089845 0Z"/>
<path id="font_2_49" d="M.35302735 .12904358C.35302735 .10851542 .34985353 .08969625 .34350587 .07258606 .3371582 .055486043 .32714845 .040822347 .31347657 .02859497 .2998047 .016377768 .28214518 .0069274904 .26049806 .00024414063 .2388509-.0064290368 .21272786-.009765625 .1821289-.009765625 .16682942-.009765625 .1517741-.008870442 .13696289-.007080078 .122151698-.0052897136 .10839844-.0031738282 .095703128-.0007324219 .08300781 .0017089844 .0719401 .004231771 .0625 .0068359377 .053059896 .0094401049 .046223958 .011555989 .041992189 .013183594V.12597656H.063964847L.087890628 .062332155C.09798177 .053268434 .1110026 .045496625 .12695313 .039016725 .14290364 .032536825 .1616211 .029296875 .18310547 .029296875 .2133789 .029296875 .23673503 .035858156 .25317384 .048980714 .26961265 .06210327 .27783204 .08243815 .27783204 .10998535 .27783204 .12628174 .27441407 .13972473 .26757813 .15031433 .2607422 .16090393 .25179038 .16977947 .24072266 .17694092 .22965496 .18411255 .21704102 .1900584 .20288086 .19477844 .1887207 .19950867 .17423503 .20431519 .15942383 .209198 .14461263 .21409099 .13012696 .21962993 .1159668 .22581482 .10180664 .23200989 .08919271 .23999532 .078125 .24977112 .06705729 .2595469 .05810547 .27176414 .05126953 .28642274 .044433595 .30109153 .041015626 .31934104 .041015626 .34117127 .041015626 .36202494 .044759115 .38059999 .052246095 .39689637 .059733076 .41319276 .07023112 .42687989 .083740238 .43795777 .09724935 .44903565 .11336263 .45742289 .13208008 .4631195 .15079753 .4688263 .17138672 .4716797 .19384766 .4716797 .21598308 .4716797 .2376302 .47013346 .25878907 .46704103 .2799479 .46394859 .30029298 .46044923 .31982423 .45654298V.3564453H.296875L.2763672 .40966798C.26790367 .41715495 .25634767 .42285157 .24169922 .4267578 .22705078 .43066407 .21142578 .4326172 .19482422 .4326172 .16845703 .4326172 .14835613 .4260966 .13452149 .41305543 .12068685 .40001426 .11376953 .38241069 .11376953 .36024476 .11376953 .34525047 .1171875 .33294679 .12402344 .32333375 .13085938 .31373088 .13989258 .30558778 .15112305 .29890443 .16235352 .29222108 .1751302 .2864329 .18945313 .28153993 .20377605 .2766571 .21842449 .27160646 .23339844 .26638795 .24837239 .2611796 .26302085 .25523377 .27734376 .24855042 .29166667 .24187725 .30444337 .23332723 .31567384 .22290039 .3269043 .21247356 .3359375 .19976299 .34277345 .18476868 .34960938 .16978455 .35302735 .15120952 .35302735 .12904358Z"/>
<path id="font_2_22" d="M.30810548 .6290741 .20703125 .61621096V.041992189H.3359375C.3766276 .041992189 .40983073 .04313151 .43554688 .045410158 .46126304 .047698976 .4790039 .04982503 .48876954 .05178833L.51904299 .18847656H.55078127L.5419922 0H.028808594V.025878907L.11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.30810548V.6290741Z"/>
<path id="font_2_29" d="M.7109375 .65527346V.6290741L.63916018 .61621096 .37597657-.015625H.35107423L.08496094 .61621096 .011230469 .6290741V.65527346H.2758789V.6290741L.18798828 .61621096 .38623048 .13400269 .5839844 .61621096 .49804688 .6290741V.65527346H.7109375Z"/>
<path id="font_2_34" d="M.41308595 .027832032C.4046224 .021647135 .39453126 .016194663 .3828125 .011474609 .37109376 .006754557 .3585612 .0028483074 .34521485-.00024414063 .3318685-.0033365887 .31795249-.0056966149 .3034668-.0073242189 .2889811-.008951823 .27490235-.009765625 .26123048-.009765625 .22151692-.009765625 .18758138-.004308065 .15942383 .0066070558 .13126628 .017522177 .10823568 .033406576 .09033203 .054260255 .07242838 .07511393 .059244794 .10061137 .05078125 .13075257 .042317708 .16089376 .038085939 .19502767 .038085939 .2331543 .038085939 .27486167 .04353841 .31078593 .05444336 .34092713 .065348308 .37106834 .080566409 .39583335 .100097659 .41522218 .119628909 .434611 .14282227 .4488678 .16967774 .45799256 .1965332 .4671173 .22591146 .4716797 .2578125 .4716797 .2841797 .4716797 .30973307 .47013346 .33447267 .46704103 .35921226 .46394859 .3816732 .46044923 .40185548 .45654298V.32861329H.375L.3540039 .40966798C.34195964 .4165039 .32739259 .42203776 .31030274 .42626954 .2932129 .4305013 .27539063 .4326172 .25683595 .4326172 .23567708 .4326172 .21704102 .42878724 .20092774 .42112733 .18481446 .4134674 .17114258 .40148927 .15991211 .38519288 .14868164 .36889649 .1402181 .34820048 .13452149 .32310487 .12882488 .29800926 .12597656 .26802574 .12597656 .2331543 .12597656 .20381673 .12841797 .17733257 .13330078 .15370178 .1381836 .130071 .14680989 .109944667 .15917969 .093322757 .17154949 .076700847 .18823242 .063827518 .20922852 .05470276 .23022461 .045578004 .25683595 .041015626 .2890625 .041015626 .30013023 .041015626 .31144206 .041422529 .32299806 .042236329 .33455406 .04305013 .34578453 .044189454 .35668946 .045654298 .3675944 .04711914 .3778483 .048828126 .38745118 .05078125 .39705406 .052734376 .40559898 .05485026 .41308595 .057128908V.027832032Z"/>
<path id="font_2_9" d="M.4609375 .17848206C.4609375 .14881897 .45572917 .1223348 .4453125 .09902954 .43489585 .07572428 .41984049 .055999757 .40014649 .039855958 .38045249 .02372233 .35620118 .01141866 .32739259 .0029449464 .29858399-.005528768 .26578776-.009765625 .2290039-.009765625 .19677735-.009765625 .16552735-.0078074138 .1352539-.0038909913 .10498047 .000025431315 .07763672 .005086263 .053222658 .011291504L.047851564 .14941406H.080078128L.10205078 .057617189C.107910159 .05436198 .1155599 .05110677 .125 .047851564 .13444011 .044596357 .14461263 .041748048 .15551758 .03930664 .16642253 .036865236 .177653 .03491211 .18920899 .033447267 .20076497 .031982423 .21142578 .03125 .2211914 .03125 .25146485 .03125 .2762858 .03499349 .2956543 .04248047 .3150228 .04996745 .33032228 .060465497 .34155274 .07397461 .3527832 .08748373 .3605143 .10359701 .3647461 .12231445 .36897788 .1410319 .37109376 .16145833 .37109376 .18359375 .37109376 .20898438 .36751304 .23006185 .36035157 .24682617 .3531901 .26359049 .3433431 .2770996 .33081056 .28735353 .31827799 .29760743 .30362956 .3050944 .28686524 .30981446 .27010093 .31453453 .25211589 .3173828 .23291016 .31835938L.16308594 .32226563V.3623047L.23291016 .36668397C.269694 .36863709 .296875 .38000999 .31445313 .4008026 .33203126 .42159526 .3408203 .45311485 .3408203 .49536134 .3408203 .5164795 .33862306 .5349172 .33422853 .55067446 .32983399 .5664317 .32291667 .5795085 .31347657 .5899048 .30403648 .6003011 .29174806 .6080983 .27661134 .6132965 .2614746 .6184947 .2430013 .62109377 .2211914 .62109377 .21142578 .62109377 .20157878 .6203613 .19165039 .6188965 .181722 .61744186 .17220052 .61549887 .16308594 .6130676 .15397136 .61064657 .14550781 .6078949 .13769531 .6048126 .12988281 .60173037 .123046878 .5985718 .1171875 .5953369L.100097659 .515625H.067871097V.6411896C.07926432 .6441091 .090657558 .64686587 .10205078 .64945986 .11344401 .65205386 .12532552 .6542409 .13769531 .6560211 .15006511 .65781149 .16308594 .65927127 .17675781 .6604004 .19042969 .66153976 .20524089 .6621094 .2211914 .6621094 .2898763 .6621094 .34204103 .6489461 .37768556 .6226196 .41333009 .59629318 .43115235 .55582687 .43115235 .5012207 .43115235 .4807434 .4283854 .46164958 .42285157 .4439392 .41731773 .42622886 .4086914 .41054789 .39697267 .39689637 .3852539 .38324485 .37036134 .3717855 .35229493 .3625183 .33422853 .35325114 .31282554 .34683229 .28808595 .34326173 .34733073 .33641563 .39095054 .3193817 .4189453 .29216004 .4469401 .26494853 .4609375 .22705586 .4609375 .17848206Z"/>
<path id="font_2_5" d="M.18408203 .044433595C.18408203 .036295576 .18253581 .028645834 .17944336 .021484375 .1763509 .014322917 .17220052 .008056641 .16699219 .0026855469 .16178386-.0026855469 .15551758-.006917318 .14819336-.010009766 .14086914-.013102214 .13313802-.0146484379 .125-.0146484379 .11653646-.0146484379 .10872396-.013102214 .1015625-.010009766 .09440104-.006917318 .08821615-.0026855469 .08300781 .0026855469 .07779948 .008056641 .073649089 .014322917 .07055664 .021484375 .067464198 .028645834 .06591797 .036295576 .06591797 .044433595 .06591797 .052571615 .067464198 .060302736 .07055664 .06762695 .073649089 .07495117 .07779948 .081217449 .08300781 .08642578 .08821615 .09163412 .09440104 .09578451 .1015625 .09887695 .10872396 .1019694 .11653646 .103515628 .125 .103515628 .13313802 .103515628 .14086914 .1019694 .14819336 .09887695 .15551758 .09578451 .16178386 .09163412 .16699219 .08642578 .17220052 .081217449 .1763509 .07495117 .17944336 .06762695 .18253581 .060302736 .18408203 .052571615 .18408203 .044433595Z"/>
<path id="font_2_12" d="M.47021485 .2033844C.47021485 .16948955 .4659017 .13934326 .4572754 .11294556 .44864909 .08654785 .4358724 .064219158 .4189453 .045959474 .40201823 .027709961 .38110353 .013860066 .35620118 .00440979 .33129884-.005040487 .30257163-.009765625 .27001954-.009765625 .23421224-.009765625 .20222982-.002766927 .17407227 .011230469 .14591472 .025227866 .122151698 .046142579 .1027832 .07397461 .08341471 .10180664 .068603519 .13647461 .05834961 .17797852 .048095704 .21948242 .04296875 .26790367 .04296875 .3232422 .04296875 .38248698 .04972331 .433431 .06323242 .47607423 .07674154 .51871749 .094807948 .5538737 .11743164 .58154299 .14005535 .6092122 .16617839 .6295573 .19580078 .6425781 .22542317 .65559896 .2561849 .6621094 .28808595 .6621094 .3125 .6621094 .3372396 .66048178 .3623047 .65722659 .38736979 .6539714 .4099935 .64990237 .43017579 .64501956V.53222659H.39794923L.38085938 .5991211C.375 .6023763 .36824546 .60530599 .3605957 .60791018 .35294596 .61051437 .3449707 .61279299 .33666993 .6147461 .32836915 .6166992 .32006837 .6182454 .31176759 .61938479 .3034668 .6205241 .2955729 .62109377 .28808595 .62109377 .26497398 .62109377 .244222 .61557009 .22583008 .6045227 .20743816 .59347537 .19156902 .5767415 .17822266 .5543213 .1648763 .53190109 .15437825 .5037944 .14672852 .47000123 .13907878 .4362081 .13460286 .39640299 .13330078 .35058595 .15673828 .36295573 .18237305 .37304688 .21020508 .38085938 .23803711 .38867188 .265625 .39257813 .29296876 .39257813 .3203125 .39257813 .3449707 .38874818 .36694337 .38108827 .38891603 .37342835 .4075521 .36177574 .42285157 .34613038 .43815104 .33048503 .44986979 .3107656 .4580078 .28697206 .46614585 .2631887 .47021485 .23532613 .47021485 .2033844M.2680664 .029296875C.28889976 .029296875 .30647788 .032714845 .32080079 .03955078 .3351237 .04638672 .3466797 .056722005 .35546876 .07055664 .3642578 .08439127 .37052409 .10164388 .37426759 .12231445 .37801109 .14298503 .3798828 .16699219 .3798828 .19433594 .3798828 .24772136 .37150065 .28629557 .35473634 .3100586 .33797203 .33382163 .3113607 .34570313 .27490235 .34570313 .25276695 .34570313 .22957357 .34358726 .20532227 .33935548 .18107097 .3351237 .15690105 .32910157 .1328125 .32128907 .1328125 .27441407 .13533528 .23291016 .14038086 .19677735 .14542644 .16064453 .15348308 .13012696 .16455078 .10522461 .17561849 .080322269 .18961589 .06144206 .20654297 .048583986 .22347005 .03572591 .24397786 .029296875 .2680664 .029296875Z"/>
<path id="font_2_16" d="M.22509766 .025878907V0H.009765625V.025878907L.083984378 .0390625 .3071289 .66015627H.39990235L.63183596 .0390625 .71484377 .025878907V0H.43798829V.025878907L.5258789 .0390625 .4609375 .22851563H.203125L.13720703 .0390625 .22509766 .025878907M.33007813 .58984377 .21777344 .27246095H.44580079L.33007813 .58984377Z"/>
<path id="font_2_18" d="M.3779297-.009765625C.32454429-.009765625 .27693687-.0022786458 .23510742 .0126953129 .193278 .027669272 .15795899 .049316408 .12915039 .07763672 .1003418 .10595703 .07845052 .14054363 .06347656 .18139649 .048502607 .22224935 .041015626 .26839195 .041015626 .31982423 .041015626 .37841798 .048583986 .42919923 .0637207 .47216798 .07885742 .5151367 .10083008 .5506999 .12963867 .5788574 .15844727 .60701498 .19384766 .6279297 .23583985 .64160159 .27783204 .65527346 .32584635 .6621094 .3798828 .6621094 .40234376 .6621094 .4235026 .66137698 .44335938 .6599121 .46321617 .65844729 .48217774 .6565755 .50024417 .6542969 .51831057 .65201827 .53548178 .64941409 .5517578 .6464844 .5680339 .6435547 .5838216 .6404622 .5991211 .63720706L.6020508 .49414063H.5698242L.5551758 .57910159C.53238937 .59309896 .50594076 .60392257 .47583009 .61157229 .4457194 .619222 .41503907 .6230469 .38378907 .6230469 .34570313 .6230469 .31176759 .6178436 .28198243 .60743716 .25219728 .59703066 .22705078 .58003237 .20654297 .55644229 .18603516 .53286239 .17032878 .50180056 .15942383 .46325685 .14851888 .4247233 .1430664 .37731935 .1430664 .32104493 .1430664 .27160646 .1484375 .22859192 .15917969 .19200135 .16992188 .15541077 .18538411 .125 .2055664 .10076904 .2257487 .076538089 .2504069 .05840556 .27954103 .04637146 .30867515 .03433736 .34179688 .028320313 .37890626 .028320313 .39908854 .028320313 .41837565 .02961731 .43676759 .032211305 .45515953 .034805299 .47224937 .03837077 .4880371 .042907716 .5038249 .047454835 .51798507 .05272929 .5305176 .05873108 .5430501 .06473287 .55354818 .0709788 .5620117 .07746887L.5800781 .17480469H.6118164L.6088867 .020996094C.59521487 .017089844 .57958987 .013264974 .5620117 .009521484 .5444336 .0057779948 .5257161 .0024414063 .5058594-.00048828127 .4860026-.0034179688 .46516929-.0056966149 .44335938-.0073242189 .42154948-.008951823 .3997396-.009765625 .3779297-.009765625Z"/>
<path id="font_2_4" d="M.037109376 .19824219V.2734375H.296875V.19824219H.037109376Z"/>
<path id="font_2_46" d="M.07421875 .4248047 .021972657 .43701173V.45898438H.1508789L.15185547 .4326172C.1586914 .43847657 .16674805 .44376628 .17602539 .44848634 .18530274 .4532064 .1953125 .4572754 .20605469 .46069337 .21679688 .46411134 .22819011 .46679688 .24023438 .46875 .25227867 .47070313 .2644857 .4716797 .27685548 .4716797 .3055013 .4716797 .33121745 .46662904 .3540039 .4565277 .37679038 .4464264 .39615885 .4313558 .41210938 .41131593 .4280599 .39127604 .44018556 .36651103 .44848634 .33702088 .4567871 .30753074 .4609375 .27355958 .4609375 .23510742 .4609375 .19764202 .45670573 .1638387 .4482422 .13369751 .43977867 .10355631 .42708335 .077809657 .41015626 .05645752 .39322917 .03511556 .37198893 .01874288 .34643556 .0073394777 .32088218-.0040639245 .29101563-.009765625 .25683595-.009765625 .24023438-.009765625 .22273763-.008870442 .2043457-.007080078 .18595378-.0052897136 .16845703-.0026041668 .15185547 .0009765625 .15218099-.0029296876 .15258789-.0074005129 .15307617-.012435913 .15356446-.017471314 .15388997-.022664389 .15405274-.028015137 .1542155-.033376058 .15437825-.038324994 .15454102-.04286194 .15470378-.047409059 .15478516-.05114746 .15478516-.05407715V-.17773438L.23486328-.18962097V-.21289063H.016113282V-.18962097L.07421875-.17773438V.4248047M.37304688 .23509217C.37304688 .2682546 .37027995 .2965393 .3647461 .3199463 .35921226 .34335328 .3511556 .36245219 .34057618 .37724305 .32999674 .39204408 .31705729 .40285746 .3017578 .40968324 .28645835 .416509 .26920573 .41992188 .25 .41992188 .234375 .41992188 .21769206 .41853843 .19995117 .41577149 .18221028 .41300456 .16715496 .40901695 .15478516 .4038086V.037582399C.16845703 .034988405 .18359375 .032958986 .20019531 .03149414 .21679688 .030029297 .23339844 .029296875 .25 .029296875 .29296876 .029296875 .32421876 .047093709 .34375 .08268738 .36328126 .11829122 .37304688 .16909282 .37304688 .23509217Z"/>
<path id="font_2_21" d="M.028808594 0V.025878907L.11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.29101563V.6290741L.20703125 .61621096V.359375H.5151367V.61621096L.43115235 .6290741V.65527346H.6928711V.6290741L.6088867 .61621096V.0390625L.6928711 .025878907V0H.43115235V.025878907L.5151367 .0390625V.3154297H.20703125V.0390625L.29101563 .025878907V0H.028808594Z"/>
<path id="font_2_48" d="M.32421876 .4716797V.34765626H.30322267L.27490235 .4013672C.26578776 .4013672 .25602214 .40071617 .24560547 .39941407 .2351888 .39811198 .22477214 .39639793 .21435547 .39427186 .2039388 .39215599 .19392903 .38962809 .18432617 .38668824 .17472331 .38375855 .16634114 .3806661 .15917969 .3774109V.034179689L.23779297 .021972657V0H.020019532V.021972657L.078125 .034179689V.4248047L.020019532 .43701173V.45898438H.1538086L.15820313 .40234376C.16569011 .40852867 .17594402 .41560874 .18896485 .42358399 .20198567 .43155924 .21606446 .4391276 .23120117 .44628907 .24633789 .45345054 .2614746 .45947267 .27661134 .46435548 .29174806 .46923829 .30517579 .4716797 .31689454 .4716797H.32421876Z"/>
<path id="font_2_52" d="M.4560547 .4267578 .27197267-.009765625H.23583985L.046875 .4248047 0 .43701173V.45898438H.21386719V.43701173L.14111328 .42382813 .27490235 .106933597 .40283204 .4248047 .33007813 .43701173V.45898438H.5V.43701173L.4560547 .4267578Z"/>
<path id="font_2_30" d="M.8847656 .61621096 .67089846-.015625H.64501956L.47509767 .43652345 .30078126-.015625H.27490235L.05810547 .61621096 .0009765625 .6290741V.65527346H.25097657V.6290741L.15478516 .61621096 .31054688 .15429688 .4868164 .609375H.50878909L.67871096 .15429688 .82714846 .61621096 .72509768 .6290741V.65527346H.94189456V.6290741L.8847656 .61621096Z"/>
<path id="font_2_10" d="M.3955078 .14453125V0H.31152345V.14453125H.01953125V.20996094L.33935548 .6582031H.3955078V.21484375H.484375V.14453125H.3955078M.31152345 .54345706H.30908204L.07470703 .21484375H.31152345V.54345706Z"/>
<path id="font_20_10" d="M.44482423 .25801087V.048828126L.5488281 .03564453V0H.18652344V.03564453L.29052735 .048828126V.25506593L.091308597 .6064453 .017578125 .6192627V.65527346H.3544922V.6192627L.26660157 .6064453 .41601563 .33200074 .5605469 .6064453 .48046876 .6192627V.65527346H.703125V.6192627L.63378909 .6064453 .44482423 .25801087Z"/>
<path id="font_20_12" d="M.34423829 .03955078C.33414714 .03434245 .32307945 .028645834 .31103517 .022460938 .29899089 .016276041 .28629557 .010579427 .27294923 .0053710939 .25960288 .00016276042 .24576824-.0041503908 .23144531-.0075683596 .21712239-.010986328 .20263672-.0126953129 .18798828-.0126953129 .1694336-.0126953129 .15234375-.01024882 .13671875-.005355835 .12109375-.00047302247 .107666019 .007344564 .09643555 .018096924 .08520508 .028859457 .0764974 .042872114 .0703125 .060134889 .0641276 .07740784 .061035158 .0982666 .061035158 .12271118V.41503907L.015136719 .4267578V.45898438H.20214844V.14170838C.20214844 .11433411 .20792644 .09298706 .21948242 .07766724 .2310384 .062347413 .2475586 .0546875 .26904298 .0546875 .29378257 .0546875 .31852214 .06022644 .34326173 .07130432V.41503907L.30126954 .4267578V.45898438H.484375V.043945314L.5292969 .032226564V0H.35107423L.34423829 .03955078Z"/>
<path id="font_20_9" d="M.017089844 .03564453 .10107422 .048828126V.6064453L.017089844 .6192627V.65527346H.57470706V.4892578H.53027346L.51464846 .5947571C.50423178 .596049 .49161784 .5971832 .47680665 .5981598 .46199546 .59913638 .4469401 .5998637 .43164063 .6003418 .41634117 .6008301 .40185548 .6011556 .3881836 .60131838 .37451173 .60148116 .36393229 .6015625 .3564453 .6015625H.2548828V.36132813H.42626954L.44140626 .43359376H.48486329V.23242188H.44140626L.42626954 .30664063H.2548828V.053710939H.37841798C.39860026 .053710939 .41764323 .053955079 .43554688 .05444336 .45345054 .05493164 .4695638 .0555013 .48388673 .056152345 .49820964 .056803388 .51049807 .057617189 .52075198 .05859375 .53100588 .059570314 .5385742 .060546876 .54345706 .061523439L.57128909 .18261719H.61572268L.6064453 0H.017089844V.03564453Z"/>
<path id="font_20_6" d="M.45703126 0H.041992189V.092285159C.06998698 .12223307 .09586588 .14900716 .119628909 .17260742 .14339192 .19620769 .16495769 .21818035 .18432617 .23852539 .20369466 .25887046 .22086589 .27823893 .23583985 .29663087 .2508138 .3150228 .26326499 .33406578 .27319337 .35375978 .28312174 .37345378 .2906901 .39453126 .29589845 .4169922 .30110679 .43945313 .30371095 .4650065 .30371095 .49365235 .30371095 .5135091 .30110679 .5308431 .29589845 .5456543 .2906901 .5604655 .2837728 .57283529 .27514649 .5827637 .26652018 .5926921 .2565104 .60009768 .24511719 .60498049 .23372396 .6098633 .22167969 .6123047 .20898438 .6123047 .18880208 .6123047 .1726888 .6101888 .16064453 .60595706 .14860027 .6017253 .13753255 .5953776 .1274414 .58691409L.10644531 .4921875H.063964847V.6411133C.09000651 .64697268 .115478519 .6519368 .14038086 .65600588 .1652832 .6600749 .19238281 .6621094 .22167969 .6621094 .29361979 .6621094 .3487142 .64729818 .3869629 .6176758 .42521159 .5880534 .44433595 .54589846 .44433595 .49121095 .44433595 .46516929 .4412435 .4416504 .4350586 .4206543 .4288737 .3996582 .41984049 .3798828 .40795899 .36132813 .39607749 .34277345 .38134767 .32470704 .36376954 .3071289 .3461914 .28955079 .32592774 .2709961 .30297853 .25146485 .2800293 .2319336 .25455729 .21077474 .2265625 .18798828 .1985677 .16520183 .16829427 .13932292 .13574219 .11035156H.45703126V0Z"/>
<path id="font_20_3" d="M0 0Z"/>
<path id="font_20_4" d="M.1772461 .24145508C.1772461 .20172119 .17862956 .16239421 .18139649 .12347412 .1841634 .08455404 .1899414 .047749837 .19873047 .013061523 .20751953-.021626791 .2199707-.05354309 .23608399-.08268738 .25219728-.111841838 .27376304-.13667806 .30078126-.15719605V-.21289063C.25716148-.18976848 .21931966-.16460674 .18725586-.1374054 .15519206-.11021423 .12849935-.07878622 .107177738-.043121339 .08585612-.0074564616 .06998698 .03357951 .059570314 .07998657 .049153646 .12640381 .043945314 .1803894 .043945314 .24194336 .043945314 .30317179 .049153646 .35682679 .059570314 .40290834 .06998698 .44900004 .08585612 .48979698 .107177738 .5252991 .12849935 .5608012 .15519206 .59206649 .18725586 .61909487 .21931966 .6461334 .25716148 .6712138 .30078126 .69433596V.63864138C.27376304 .61812338 .25219728 .59336856 .23608399 .56437686 .2199707 .5353953 .20751953 .50356039 .19873047 .46887208 .1899414 .43418376 .1841634 .39754234 .18139649 .35894776 .17862956 .32035319 .1772461 .28118897 .1772461 .24145508Z"/>
<path id="font_20_8" d="M.42871095 .49342347C.42871095 .51201888 .4268392 .52808126 .4230957 .5416107 .4193522 .55515035 .4132487 .56640627 .40478517 .5753784 .39632163 .5843506 .3853353 .59095767 .37182618 .5951996 .35831706 .5994415 .34195964 .6015625 .3227539 .6015625H.2548828V.37304688H.32666017C.34651695 .37304688 .3630371 .37581889 .3762207 .38136292 .3894043 .38690696 .39990235 .39481608 .40771485 .40509034 .41552735 .41537477 .42097984 .42793785 .42407228 .44277955 .4271647 .45762126 .42871095 .47450257 .42871095 .49342347M.47802735 .19091797C.47802735 .23421224 .46720378 .2664388 .44555665 .28759767 .42390953 .3087565 .38850913 .31933595 .33935548 .31933595H.2548828V.053710939C.2705078 .053059896 .28548179 .052571615 .2998047 .052246095 .31184898 .05159505 .324056 .051188154 .33642579 .05102539 .34879557 .05086263 .3585612 .05078125 .36572267 .05078125 .38623048 .05078125 .40364585 .054036458 .41796876 .060546876 .43229167 .06705729 .44384767 .07633463 .45263673 .088378909 .46142579 .10042318 .46785484 .11507162 .47192384 .13232422 .47599284 .14957683 .47802735 .16910808 .47802735 .19091797M.016601563 0V.03564453L.10107422 .048828126V.6064453L.016601563 .6192627V.65527346H.33447267C.38297526 .65527346 .4235026 .6518504 .4560547 .6450043 .48860679 .63815817 .5147298 .62837728 .5344238 .6156616 .55411788 .602946 .56811526 .5874583 .576416 .5691986 .5847168 .5509389 .5888672 .5302327 .5888672 .5070801 .5888672 .48523966 .5852051 .46551515 .57788088 .4479065 .57055667 .43029786 .56070968 .41489158 .54833987 .40168763 .53597006 .38848368 .52164718 .37756349 .5053711 .368927 .48909507 .36029054 .47200523 .35401408 .45410157 .35009767 .48632813 .34684245 .5140788 .3408203 .5373535 .33203126 .56062826 .3232422 .579834 .31201173 .5949707 .29833985 .6101074 .28466798 .62125656 .26879884 .62841799 .25073243 .6355794 .23266602 .63916018 .21272786 .63916018 .19091797 .63916018 .16259766 .63411459 .13655599 .62402346 .11279297 .6139323 .089029949 .5979004 .068603519 .57592776 .051513673 .5539551 .034423829 .52579757 .021077475 .49145509 .011474609 .45711265 .0018717448 .4156901-.0029296876 .3671875-.0029296876 .328125-.0029296876 .28938804-.0024414063 .25097657-.0014648438 .21256511-.00048828127 .17708333 0 .14453125 0H.016601563Z"/>
<path id="font_20_11" d="M.46191407 .23217774C.46191407 .19307454 .45792643 .15853374 .44995118 .1285553 .44197593 .098576869 .42936198 .07332357 .41210938 .05279541 .39485679 .032267255 .37263999 .016708374 .34545899 .0061187746 .31827799-.004470825 .28548179-.009765625 .24707031-.009765625 .2109375-.009765625 .1796875-.0045522057 .15332031 .005874634 .12695313 .016301474 .10522461 .031778974 .088134769 .05230713 .07104492 .07283529 .05843099 .09816996 .05029297 .12831116 .04215495 .15845235 .038085939 .19307454 .038085939 .23217774 .038085939 .2709554 .04215495 .30525209 .05029297 .33506776 .05843099 .36488343 .071126308 .38989259 .088378909 .4100952 .10563151 .43029786 .12768555 .44561259 .15454102 .45603944 .18139649 .46646629 .21354167 .4716797 .25097657 .4716797 .32356773 .4716797 .37687174 .45139567 .41088868 .41082765 .4449056 .3702596 .46191407 .31070964 .46191407 .23217774M.31884767 .2319336C.31884767 .26350913 .3178711 .29117839 .31591798 .3149414 .31396485 .33870445 .31038413 .3585612 .30517579 .37451173 .29996745 .39046226 .292806 .40234376 .2836914 .41015626 .2745768 .41796876 .2626953 .421875 .24804688 .421875 .23372396 .421875 .22216797 .41796876 .2133789 .41015626 .20458985 .40234376 .19783528 .39046226 .19311524 .37451173 .18839519 .3585612 .18522136 .33870445 .18359375 .3149414 .18196614 .29117839 .18115235 .26350913 .18115235 .2319336 .18115235 .20003255 .18196614 .17211914 .18359375 .14819336 .18522136 .12426758 .18839519 .104166667 .19311524 .087890628 .19783528 .071614589 .20458985 .05940755 .2133789 .05126953 .22216797 .04313151 .23372396 .0390625 .24804688 .0390625 .2626953 .0390625 .2745768 .04313151 .2836914 .05126953 .292806 .05940755 .29996745 .071614589 .30517579 .087890628 .31038413 .104166667 .31396485 .12426758 .31591798 .14819336 .3178711 .17211914 .31884767 .20003255 .31884767 .2319336Z"/>
<path id="font_20_7" d="M.45166017 .49378968C.45166017 .45800273 .4428711 .4275055 .42529298 .40229798 .40771485 .37709046 .38297526 .35879518 .35107423 .3474121 .3683268 .3409017 .3841146 .33243308 .3984375 .32200624 .4127604 .31157939 .42496745 .2992808 .4350586 .28511048 .44514976 .27094016 .45296226 .25481669 .4584961 .23674011 .46402995 .21866353 .46679688 .19871013 .46679688 .17687989 .46679688 .114990238 .4489746 .06841024 .41333009 .037139894 .37768556 .0058695476 .32226563-.009765625 .24707031-.009765625 .17578125-.009765625 .12231445 .0056254069 .08666992 .03640747 .05102539 .06718954 .033203126 .11401367 .033203126 .17687989 .033203126 .1983846 .035888673 .21817525 .041259767 .23625183 .04663086 .2543284 .054280599 .27045188 .064208988 .2846222 .07413737 .2987925 .08610026 .3111725 .100097659 .32176209 .114095058 .33235169 .12988281 .3409017 .14746094 .3474121 .115885417 .3594462 .09147135 .37798564 .07421875 .4030304 .056966146 .42807517 .048339845 .45881654 .048339845 .49525453 .048339845 .5209503 .052652997 .5442861 .061279298 .56526187 .0699056 .5862376 .08268229 .604126 .099609378 .618927 .11653646 .633728 .13769531 .6451111 .16308594 .6530762 .18847656 .66105148 .21777344 .66503909 .25097657 .66503909 .28320313 .66503909 .31176759 .66105148 .33666993 .6530762 .36157228 .6451111 .38256837 .63364669 .3996582 .61868289 .41674806 .60371908 .4296875 .5857493 .43847657 .56477358 .44726563 .5437978 .45166017 .52013656 .45166017 .49378968M.328125 .17700196C.328125 .20007324 .32674156 .22046407 .3239746 .23817444 .32120768 .2558848 .3166504 .27083335 .31030274 .28302003 .30395509 .2952067 .2956543 .30446879 .2854004 .31080628 .27514649 .31714378 .26236979 .3203125 .24707031 .3203125 .23209636 .3203125 .21980794 .31714378 .21020508 .31080628 .20060222 .30446879 .19295247 .2952067 .18725586 .28302003 .18155925 .27083335 .17757161 .2558848 .17529297 .23817444 .17301433 .22046407 .171875 .20007324 .171875 .17700196 .171875 .15490723 .17301433 .1353302 .17529297 .118270877 .17757161 .10121155 .18155925 .08691406 .18725586 .07537842 .19295247 .06384277 .20060222 .05506897 .21020508 .049057008 .21980794 .043045045 .23209636 .040039064 .24707031 .040039064 .26236979 .040039064 .27514649 .043045045 .2854004 .049057008 .2956543 .05506897 .30395509 .06384277 .31030274 .07537842 .3166504 .08691406 .32120768 .10121155 .3239746 .118270877 .32674156 .1353302 .328125 .15490723 .328125 .17700196M.31298829 .49365235C.31298829 .51115927 .31201173 .5273692 .3100586 .5422821 .30810548 .557195 .3046061 .5700836 .29956056 .5809479 .29451499 .59181216 .2878418 .60024008 .27954103 .6062317 .27124024 .61223348 .2607422 .6152344 .24804688 .6152344 .23567708 .6152344 .22558594 .61223348 .21777344 .6062317 .20996094 .60024008 .20377605 .59181216 .19921875 .5809479 .19466146 .5700836 .19148763 .557195 .18969727 .5422821 .1879069 .5273692 .18701172 .51115927 .18701172 .49365235 .18701172 .47485353 .1879069 .45799256 .18969727 .44306947 .19148763 .42815653 .19466146 .41551209 .19921875 .4051361 .20377605 .3947703 .20996094 .38683067 .21777344 .38131715 .22558594 .37580363 .23567708 .37304688 .24804688 .37304688 .2607422 .37304688 .27124024 .37580363 .27954103 .38131715 .2878418 .38683067 .29451499 .3947703 .29956056 .4051361 .3046061 .41551209 .30810548 .42815653 .3100586 .44306947 .31201173 .45799256 .31298829 .47485353 .31298829 .49365235Z"/>
<path id="font_20_5" d="M.032226564-.21289063V-.15719605C.05891927-.13667806 .08040365-.111841838 .09667969-.08268738 .11295573-.05354309 .12548828-.021626791 .13427735 .013061523 .1430664 .047749837 .1488444 .08455404 .15161133 .12347412 .15437825 .16239421 .15576172 .20172119 .15576172 .24145508 .15576172 .28118897 .15437825 .32035319 .15161133 .35894776 .1488444 .39754234 .1430664 .43418376 .13427735 .46887208 .12548828 .50356039 .11295573 .5353953 .09667969 .56437686 .08040365 .59336856 .05891927 .61812338 .032226564 .63864138V.69433596C.075520839 .6712138 .11328125 .6461334 .14550781 .61909487 .17773438 .59206649 .20450847 .5608012 .22583008 .5252991 .24715169 .48979698 .26302085 .44900004 .2734375 .40290834 .28385417 .35682679 .2890625 .30317179 .2890625 .24194336 .2890625 .1803894 .28385417 .12640381 .2734375 .07998657 .26302085 .03357951 .24715169-.0074564616 .22583008-.043121339 .20450847-.07878622 .17773438-.11021423 .14550781-.1374054 .11328125-.16460674 .075520839-.18976848 .032226564-.21289063Z"/>
<path id="font_2_41" d="M.16796875 .2211914 .35595704 .42382813 .30810548 .43701173V.45898438H.47021485V.43701173L.41308595 .42578126 .28222657 .29203797 .4501953 .033203126 .5 .021972657V0H.31201173V.021972657L.3540039 .034179689 .22802735 .2318573 .16796875 .16601563V.034179689L.21679688 .021972657V0H.028808594V.021972657L.08691406 .034179689V.66015627L.019042969 .67204287V.69433596H.16796875V.2211914Z"/>
<path id="font_2_15" d="M.032226564 .45507813C.032226564 .48860679 .037109376 .51831057 .046875 .54418948 .056640626 .57006838 .07063802 .5917155 .08886719 .60913088 .10709635 .6265462 .12923177 .6397298 .15527344 .64868167 .18131511 .6576335 .21061199 .6621094 .24316406 .6621094 .28027345 .6621094 .31233726 .65559896 .33935548 .6425781 .3663737 .6295573 .38875328 .60945639 .40649415 .5822754 .42423503 .5550944 .4374186 .5205078 .44604493 .47851563 .45467124 .43652345 .45898438 .38671876 .45898438 .32910157 .45898438 .28710938 .45581056 .2495931 .4494629 .21655274 .44311524 .18351238 .43424479 .15445964 .42285157 .12939453 .41145835 .10432943 .39786784 .08292643 .38208009 .06518555 .36629234 .04744466 .34895835 .033040365 .33007813 .021972657 .3111979 .010904948 .29109703 .0028483074 .2697754-.0021972657 .24845378-.0072428386 .2265625-.009765625 .20410156-.009765625 .17545574-.009765625 .14949544-.008382161 .1262207-.0056152346 .10294596-.0028483074 .08024088 .0013020834 .05810547 .0068359377V.12011719H.08984375L.106933597 .050186159C.11311849 .047276815 .120035808 .04468791 .12768555 .042419435 .13533528 .04015096 .14331055 .03812663 .15161133 .036346437 .15991211 .034566244 .16837566 .033269247 .17700196 .032455446 .18562825 .031651815 .19401042 .03125 .20214844 .03125 .22493489 .03125 .24617513 .036051435 .26586915 .045654298 .28556315 .05525716 .30273438 .07080078 .3173828 .092285159 .33203126 .11376953 .3438314 .14168294 .3527832 .17602539 .36173503 .21036785 .36702476 .25227867 .36865235 .3017578 .34716798 .28957115 .32389323 .27952577 .29882813 .2716217 .27376304 .26371766 .2467448 .25976563 .21777344 .25976563 .19042969 .25976563 .16536458 .26391603 .14257813 .2722168 .119791667 .28051759 .10017904 .29288737 .083740238 .30932618 .06730143 .32576499 .05460612 .3461914 .045654298 .37060548 .036702474 .39501954 .032226564 .4231771 .032226564 .45507813M.24414063 .6230469C.20214844 .6230469 .17130535 .6085002 .15161133 .57940676 .13191732 .5503235 .12207031 .50831606 .12207031 .4533844 .12207031 .4267324 .12475586 .4041443 .13012696 .38562013 .13549805 .36709596 .14331055 .35206095 .15356446 .34051515 .16381836 .3289795 .1763509 .32061259 .19116211 .31541444 .20597331 .31021629 .22298177 .3076172 .2421875 .3076172 .26367188 .3076172 .28515626 .30989585 .30664063 .31445313 .328125 .3190104 .34895835 .32535807 .36914063 .3334961 .36914063 .38126628 .36686198 .42326866 .3623047 .45950318 .3577474 .4957377 .35058595 .52596029 .3408203 .5501709 .3310547 .57438156 .31819663 .59258016 .3022461 .60476687 .28629557 .61695358 .2669271 .6230469 .24414063 .6230469Z"/>
<path id="font_2_24" d="M.4189453 .4611969C.4189453 .48727928 .41609703 .5097707 .4104004 .52867129 .40470378 .547582 .39542643 .56315109 .38256837 .5753784 .3697103 .5876058 .3528646 .59665426 .33203126 .6025238 .3111979 .6083934 .28548179 .6113281 .2548828 .6113281H.20703125V.30078126H.2578125C.28841148 .30078126 .31404624 .30444847 .3347168 .31178285 .35538737 .31911723 .37198893 .32963053 .38452149 .34332276 .39705406 .35702516 .40592448 .37381999 .4111328 .39370729 .41634117 .41359458 .4189453 .4360911 .4189453 .4611969M.20703125 .25683595V.0390625L.31103517 .025878907V0H.03515625V.025878907L.11279297 .0390625V.61621096L.028808594 .6290741V.65527346H.2758789C.32177735 .65527346 .36010743 .65030416 .39086915 .6403656 .42163087 .63042709 .44628907 .6167348 .46484376 .59928897 .48339845 .5818532 .49658204 .56140139 .50439456 .53793337 .51220706 .5144755 .5161133 .48921714 .5161133 .4621582 .5161133 .43543498 .5122884 .40968833 .5046387 .3849182 .49698893 .3601481 .48413087 .33831278 .46606446 .31941224 .44799806 .3005117 .42382813 .2853546 .3935547 .27394105 .36328126 .26253764 .3256836 .25683595 .28076173 .25683595H.20703125Z"/>
<path id="font_2_25" d="M.1430664 .32739259C.1430664 .28186036 .14648438 .24071758 .15332031 .20396424 .16015625 .16721089 .171875 .13590496 .18847656 .11004639 .20507813 .08418783 .2273763 .06426493 .2553711 .05027771 .28336589 .036290487 .31852214 .029296875 .36083985 .029296875 .40315757 .029296875 .4383138 .036290487 .4663086 .05027771 .49430339 .06426493 .5166829 .08418783 .53344729 .11004639 .5502116 .13590496 .5620117 .16721089 .56884768 .20396424 .5756836 .24071758 .57910159 .28186036 .57910159 .32739259 .57910159 .37259928 .5756836 .41349793 .56884768 .4500885 .5620117 .48667909 .5502116 .5177409 .53344729 .5432739 .5166829 .56880697 .49430339 .5884857 .4663086 .6023102 .4383138 .61613467 .40315757 .6230469 .36083985 .6230469 .31852214 .6230469 .28336589 .61613467 .2553711 .6023102 .2273763 .5884857 .20507813 .56880697 .18847656 .5432739 .171875 .5177409 .16015625 .48667909 .15332031 .4500885 .14648438 .41349793 .1430664 .37259928 .1430664 .32739259M.041015626 .32714845C.041015626 .38509117 .048095704 .43513999 .06225586 .47729493 .076416019 .5194499 .09700521 .5541992 .12402344 .58154299 .15104167 .6088867 .18440755 .6291504 .2241211 .642334 .26383464 .6555176 .30940757 .6621094 .36083985 .6621094 .41064454 .6621094 .45532228 .6555176 .49487306 .642334 .5344238 .6291504 .5680339 .6088867 .5957031 .58154299 .6233724 .5541992 .64453127 .5194499 .6591797 .47729493 .6738281 .43513999 .68115237 .38509117 .68115237 .32714845 .68115237 .28548179 .67732748 .2479655 .66967776 .21459961 .662028 .18123372 .65079757 .1517741 .6359863 .1262207 .6211751 .10066732 .60302737 .07877604 .58154299 .060546876 .5600586 .042317708 .53548178 .02750651 .5078125 .016113282L.53222659-.014053345C.54622396-.03156026 .5592448-.046635946 .57128909-.059280397 .5833333-.07193502 .5946452-.0822347 .6052246-.09017944 .615804-.09812418 .62597659-.10395813 .6357422-.107681278 .6455078-.11141459 .65527346-.11328125 .66503909-.11328125 .6673177-.11328125 .67041018-.11319987 .6743164-.11303711 .67822268-.11287435 .6821289-.112711589 .68603518-.11254883 .6899414-.11238607 .6936849-.112223308 .6972656-.11206055 .7008464-.11189779 .7034505-.11165365 .7050781-.111328128V-.14385987C.7014974-.14550781 .6961263-.14731853 .68896487-.14929199 .6818034-.15126546 .67374679-.15323384 .6647949-.15519715 .6558431-.15717061 .6464844-.15881348 .63671877-.16012573 .6269531-.16144817 .6176758-.16210938 .6088867-.16210938 .59033206-.16210938 .5736491-.15974935 .5588379-.1550293 .5440267-.15030925 .529541-.14282227 .51538088-.13256836 .5012207-.12231445 .48673503-.1089681 .47192384-.0925293 .45711265-.07609049 .4404297-.056315107 .421875-.033203126L.40185548-.0078125C.3959961-.008463542 .38972984-.008951823 .38305665-.009277344 .37638346-.009602864 .36897788-.009765625 .36083985-.009765625 .31103517-.009765625 .2664388-.003092448 .22705078 .010253906 .18766277 .02360026 .15413411 .044108076 .12646485 .071777347 .09879557 .09944662 .07763672 .13444011 .06298828 .17675781 .048339845 .21907552 .041015626 .26920573 .041015626 .32714845Z"/>
<path id="font_2_14" d="M.44189454 .4951172C.44189454 .4593099 .43318687 .42895509 .41577149 .40405274 .3983561 .3791504 .37483726 .3601888 .34521485 .34716798 .36279298 .34065757 .3787435 .332194 .3930664 .32177735 .4073893 .3113607 .41967774 .29907228 .42993165 .2849121 .44018556 .27075196 .44807945 .25463868 .45361329 .23657227 .45914714 .21850586 .46191407 .1985677 .46191407 .17675781 .46191407 .11490885 .4444987 .068359378 .40966798 .037109376 .37483726 .005859375 .32063804-.009765625 .24707031-.009765625 .17740886-.009765625 .12516277 .0056152346 .09033203 .036376954 .0555013 .06713867 .038085939 .11393229 .038085939 .17675781 .038085939 .22005208 .048502607 .25594078 .06933594 .28442384 .09016927 .3129069 .11832682 .33382163 .1538086 .34716798 .12548828 .3601888 .10245768 .379069 .0847168 .4038086 .066975917 .4285482 .05810547 .45898438 .05810547 .4951172 .05810547 .5208333 .062093099 .54418948 .07006836 .56518557 .07804362 .58618167 .09000651 .60408529 .10595703 .6188965 .121907558 .6337077 .14192708 .6451009 .16601563 .6530762 .19010417 .66105148 .21842449 .66503909 .25097657 .66503909 .28222657 .66503909 .30973307 .6611328 .3334961 .6533203 .35725913 .6455078 .37719728 .63427737 .39331056 .6196289 .40942384 .60498049 .42154948 .5871582 .4296875 .5661621 .43782554 .545166 .44189454 .5214844 .44189454 .4951172M.37402345 .17700196C.37402345 .20072429 .37190757 .22176616 .36767579 .24012757 .363444 .25848899 .35636393 .27400718 .34643556 .28668214 .33650718 .2993571 .32340495 .3089447 .3071289 .31544496 .29085288 .3219452 .27083335 .3251953 .24707031 .3251953 .22298177 .3251953 .203125 .3219452 .1875 .31544496 .171875 .3089447 .1595052 .2993571 .15039063 .28668214 .14127605 .27400718 .13492839 .25848899 .13134766 .24012757 .12776692 .22176616 .12597656 .20072429 .12597656 .17700196 .12597656 .1529541 .12776692 .13174947 .13134766 .11338806 .13492839 .09502665 .14127605 .07967123 .15039063 .06732178 .1595052 .054972333 .171875 .045547487 .1875 .03904724 .203125 .032546998 .22298177 .029296875 .24707031 .029296875 .27083335 .029296875 .29085288 .032546998 .3071289 .03904724 .32340495 .045547487 .33650718 .054972333 .34643556 .06732178 .35636393 .07967123 .363444 .09502665 .36767579 .11338806 .37190757 .13174947 .37402345 .1529541 .37402345 .17700196M.3540039 .4951172C.3540039 .51432296 .35229493 .53190109 .34887696 .54785159 .34545899 .56380209 .339681 .57755538 .33154298 .5891113 .32340495 .6006673 .3125 .6097005 .29882813 .61621096 .28515626 .6227214 .26822917 .62597659 .24804688 .62597659 .22786458 .62597659 .21118164 .6227214 .19799805 .61621096 .18481446 .6097005 .17439778 .6006673 .16674805 .5891113 .15909831 .57755538 .15372722 .56380209 .15063477 .54785159 .14754232 .53190109 .1459961 .51432296 .1459961 .4951172 .1459961 .47558595 .14754232 .4580078 .15063477 .4423828 .15372722 .4267578 .15909831 .41341148 .16674805 .40234376 .17439778 .39127604 .18481446 .3828125 .19799805 .37695313 .21118164 .37109376 .22786458 .36816407 .24804688 .36816407 .26822917 .36816407 .28515626 .37109376 .29882813 .37695313 .3125 .3828125 .32340495 .39127604 .33154298 .40234376 .339681 .41341148 .34545899 .4267578 .34887696 .4423828 .35229493 .4580078 .3540039 .47558595 .3540039 .4951172Z"/>
<path id="font_2_33" d="M.37402345 .2421875C.37402345 .27539063 .37101237 .30330406 .36499024 .32592774 .3589681 .34855143 .3503418 .36686198 .33911134 .38085938 .32788087 .39485679 .31437174 .40486656 .29858399 .41088868 .28279624 .4169108 .26529948 .41992188 .24609375 .41992188 .23828125 .41992188 .2298991 .41959635 .22094727 .4189453 .21199544 .41829429 .203125 .41723634 .19433594 .41577149 .18554688 .41430665 .17708333 .41259767 .16894531 .41064454 .1608073 .4086914 .1538086 .40641276 .14794922 .4038086V.040039064C.1616211 .037434896 .1772461 .03548177 .19482422 .034179689 .21240235 .032877607 .22949219 .032226564 .24609375 .032226564 .29101563 .032226564 .32356773 .049804689 .34375 .08496094 .36393229 .12011719 .37402345 .17252605 .37402345 .2421875M.06689453 .66015627 0 .67204287V.69433596H.14794922V.52996829C.14794922 .5237732 .14786785 .51667788 .14770508 .50868228 .14754232 .50069686 .14737956 .49238078 .1472168 .48373414 .14705403 .47509767 .14672852 .4664561 .14624024 .45780946 .14575196 .4491628 .14534505 .44092814 .14501953 .43310548 .15966797 .4446411 .17749024 .45395408 .19848633 .4610443 .21948242 .46813456 .24267578 .4716797 .2680664 .4716797 .3305664 .4716797 .37849937 .45269776 .41186524 .4147339 .4452311 .37677003 .46191407 .31934104 .46191407 .2424469 .46191407 .20366924 .45768229 .16872151 .44921876 .13760376 .44075523 .106486 .42773438 .08000692 .41015626 .058166505 .39257813 .03633626 .37036134 .01955668 .34350587 .007827759 .3166504-.0039011639 .28483073-.009765625 .24804688-.009765625 .23242188-.009765625 .21655274-.008870442 .20043946-.007080078 .18432617-.0052897136 .16845703-.0029296876 .15283203 0 .13720703 .0029296876 .12207031 .0064290368 .107421878 .010498047 .09277344 .014567058 .07926432 .019042969 .06689453 .023925782V.66015627Z"/>
</defs>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M0 0H396V170.87078H0Z" fill="#ffffff"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M39.6 19.562769H322.74V132.17078H39.6Z" fill="#ffffff"/>
<g clip-path="url(#clip_1)">
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".45" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#e9eef1" d="M50.184675 50.361539H314.80149"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".45" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#e9eef1" d="M50.184675 74.42308H314.80149"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".45" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#e9eef1" d="M50.184675 98.48461H314.80149"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".45" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#e9eef1" d="M50.184675 122.54615H314.80149"/>
</g>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M50.184675 26.3V24"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M50.184675 26.3V24"/>
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,48.184675,154.91765)"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M116.33888 26.3V24"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M116.33888 26.3V24"/>
<use data-text="2" xlink:href="#font_2_8" transform="matrix(8,0,0,-8,112.338878,154.91765)"/>
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8,0,0,-8,116.338878,154.91765)"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M182.49309 26.3V24"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M182.49309 26.3V24"/>
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8,0,0,-8,178.49309,154.91765)"/>
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,182.49309,154.91765)"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M248.6473 26.3V24"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M248.6473 26.3V24"/>
<use data-text="7" xlink:href="#font_2_13" transform="matrix(8,0,0,-8,244.6473,154.91765)"/>
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8,0,0,-8,248.6473,154.91765)"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M314.80149 26.3V24"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M314.80149 26.3V24"/>
<use data-text="1" xlink:href="#font_2_7" transform="matrix(8,0,0,-8,308.80149,154.91765)"/>
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,312.80149,154.91765)"/>
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,316.80149,154.91765)"/>
<use data-text="S" xlink:href="#font_2_27" transform="matrix(9.5,0,0,-9.5,141.71688,166.55828)"/>
<use data-text="o" xlink:href="#font_2_45" transform="matrix(9.5,0,0,-9.5,146.99887,166.55828)"/>
<use data-text="n" xlink:href="#font_2_44" transform="matrix(9.5,0,0,-9.5,151.74887,166.55828)"/>
<use data-text="g" xlink:href="#font_2_38" transform="matrix(9.5,0,0,-9.5,156.49887,166.55828)"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(9.5,0,0,-9.5,161.24887,166.55828)"/>
<use data-text="q" xlink:href="#font_2_47" transform="matrix(9.5,0,0,-9.5,163.62387,166.55828)"/>
<use data-text="u" xlink:href="#font_2_51" transform="matrix(9.5,0,0,-9.5,168.37387,166.55828)"/>
<use data-text="a" xlink:href="#font_2_32" transform="matrix(9.5,0,0,-9.5,173.12387,166.55828)"/>
<use data-text="l" xlink:href="#font_2_42" transform="matrix(9.5,0,0,-9.5,177.34188,166.55828)"/>
<use data-text="i" xlink:href="#font_2_40" transform="matrix(9.5,0,0,-9.5,179.98288,166.55828)"/>
<use data-text="t" xlink:href="#font_2_50" transform="matrix(9.5,0,0,-9.5,182.62387,166.55828)"/>
<use data-text="y" xlink:href="#font_2_54" transform="matrix(9.5,0,0,-9.5,185.26486,166.55828)"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(9.5,0,0,-9.5,190.01486,166.55828)"/>
<use data-text="i" xlink:href="#font_2_40" transform="matrix(9.5,0,0,-9.5,192.38986,166.55828)"/>
<use data-text="n" xlink:href="#font_2_44" transform="matrix(9.5,0,0,-9.5,195.03087,166.55828)"/>
<use data-text="d" xlink:href="#font_2_35" transform="matrix(9.5,0,0,-9.5,199.78087,166.55828)"/>
<use data-text="e" xlink:href="#font_2_36" transform="matrix(9.5,0,0,-9.5,204.53087,166.55828)"/>
<use data-text="x" xlink:href="#font_2_53" transform="matrix(9.5,0,0,-9.5,208.74887,166.55828)"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(9.5,0,0,-9.5,213.49887,166.55828)"/>
<use data-text="&#x2191;" xlink:href="#font_2_55" transform="matrix(9.5,0,0,-9.5,215.87387,166.55828)"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M50.184675 26.3H47.884675"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M50.184675 26.3H47.884675"/>
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,41.384675,147.34421)"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M50.184675 50.361539H47.884675"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M50.184675 50.361539H47.884675"/>
<use data-text="2" xlink:href="#font_2_8" transform="matrix(8,0,0,-8,37.384675,123.28267)"/>
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8,0,0,-8,41.384675,123.28267)"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M50.184675 74.42308H47.884675"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M50.184675 74.42308H47.884675"/>
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8,0,0,-8,37.384675,99.22113)"/>
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,41.384675,99.22113)"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M50.184675 98.48461H47.884675"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M50.184675 98.48461H47.884675"/>
<use data-text="7" xlink:href="#font_2_13" transform="matrix(8,0,0,-8,37.384675,75.1596)"/>
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8,0,0,-8,41.384675,75.1596)"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M50.184675 122.54615H47.884675"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".75" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#000000" d="M50.184675 122.54615H47.884675"/>
<use data-text="1" xlink:href="#font_2_7" transform="matrix(8,0,0,-8,33.384675,51.09806)"/>
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,37.384675,51.09806)"/>
<use data-text="0" xlink:href="#font_2_6" transform="matrix(8,0,0,-8,41.384675,51.09806)"/>
<use data-text="T" xlink:href="#font_2_28" transform="matrix(0,-9.5,-9.5,-0,26.228423,138.88681)"/>
<use data-text="e" xlink:href="#font_2_36" transform="matrix(0,-9.5,-9.5,-0,26.228423,133.73856)"/>
<use data-text="x" xlink:href="#font_2_53" transform="matrix(0,-9.5,-9.5,-0,26.228423,129.52056)"/>
<use data-text="t" xlink:href="#font_2_50" transform="matrix(0,-9.5,-9.5,-0,26.228423,124.77056)"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(0,-9.5,-9.5,-0,26.228423,122.12956)"/>
<use data-text="a" xlink:href="#font_2_32" transform="matrix(0,-9.5,-9.5,-0,26.228423,119.75456)"/>
<use data-text="l" xlink:href="#font_2_42" transform="matrix(0,-9.5,-9.5,-0,26.228423,115.53656)"/>
<use data-text="i" xlink:href="#font_2_40" transform="matrix(0,-9.5,-9.5,-0,26.228423,112.89555)"/>
<use data-text="g" xlink:href="#font_2_38" transform="matrix(0,-9.5,-9.5,-0,26.228423,110.254558)"/>
<use data-text="n" xlink:href="#font_2_44" transform="matrix(0,-9.5,-9.5,-0,26.228423,105.504558)"/>
<use data-text="m" xlink:href="#font_2_43" transform="matrix(0,-9.5,-9.5,-0,26.228423,100.754558)"/>
<use data-text="e" xlink:href="#font_2_36" transform="matrix(0,-9.5,-9.5,-0,26.228423,93.363559)"/>
<use data-text="n" xlink:href="#font_2_44" transform="matrix(0,-9.5,-9.5,-0,26.228423,89.14555)"/>
<use data-text="t" xlink:href="#font_2_50" transform="matrix(0,-9.5,-9.5,-0,26.228423,84.39555)"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(0,-9.5,-9.5,-0,26.228423,81.754558)"/>
<use data-text="i" xlink:href="#font_2_40" transform="matrix(0,-9.5,-9.5,-0,26.228423,79.379558)"/>
<use data-text="n" xlink:href="#font_2_44" transform="matrix(0,-9.5,-9.5,-0,26.228423,76.738559)"/>
<use data-text="d" xlink:href="#font_2_35" transform="matrix(0,-9.5,-9.5,-0,26.228423,71.988559)"/>
<use data-text="e" xlink:href="#font_2_36" transform="matrix(0,-9.5,-9.5,-0,26.228423,67.238559)"/>
<use data-text="x" xlink:href="#font_2_53" transform="matrix(0,-9.5,-9.5,-0,26.228423,63.020555)"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(0,-9.5,-9.5,-0,26.228423,58.270555)"/>
<use data-text="&#x2191;" xlink:href="#font_2_55" transform="matrix(0,-9.5,-9.5,-0,26.228423,55.895555)"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".8" stroke-linecap="square" stroke-miterlimit="10" stroke-linejoin="miter" fill="none" stroke="#000000" d="M50.184675 26.3V122.54615"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".8" stroke-linecap="square" stroke-miterlimit="10" stroke-linejoin="miter" fill="none" stroke="#000000" d="M50.184675 26.3H314.80149"/>
<g clip-path="url(#clip_3)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M281.9906 109.89433C282.79344 109.89433 283.56349 110.213298 284.13117 110.78098 284.69886 111.34866 285.01783 112.11871 285.01783 112.92154 285.01783 113.724369 284.69886 114.494419 284.13117 115.062099 283.56349 115.62978 282.79344 115.948749 281.9906 115.948749 281.18778 115.948749 280.41773 115.62978 279.85005 115.062099 279.28236 114.494419 278.9634 113.724369 278.9634 112.92154 278.9634 112.11871 279.28236 111.34866 279.85005 110.78098 280.41773 110.213298 281.18778 109.89433 281.9906 109.89433Z" fill="#87b3cb"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width="1" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#293844" d="M281.9906 109.89433C282.79344 109.89433 283.56349 110.213298 284.13117 110.78098 284.69886 111.34866 285.01783 112.11871 285.01783 112.92154 285.01783 113.724369 284.69886 114.494419 284.13117 115.062099 283.56349 115.62978 282.79344 115.948749 281.9906 115.948749 281.18778 115.948749 280.41773 115.62978 279.85005 115.062099 279.28236 114.494419 278.9634 113.724369 278.9634 112.92154 278.9634 112.11871 279.28236 111.34866 279.85005 110.78098 280.41773 110.213298 281.18778 109.89433 281.9906 109.89433Z"/>
</g>
<g clip-path="url(#clip_4)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M272.79505 97.80651C273.65718 97.80651 274.4841 98.14904 275.09373 98.75865 275.70335 99.36826 276.04588 100.19519 276.04588 101.05731 276.04588 101.91943 275.70335 102.74637 275.09373 103.35598 274.4841 103.96559 273.65718 104.30811 272.79505 104.30811 271.93293 104.30811 271.10603 103.96559 270.4964 103.35598 269.88679 102.74637 269.54426 101.91943 269.54426 101.05731 269.54426 100.19519 269.88679 99.36826 270.4964 98.75865 271.10603 98.14904 271.93293 97.80651 272.79505 97.80651Z" fill="#87b3cb"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#456e87" d="M272.79505 97.80651C273.65718 97.80651 274.4841 98.14904 275.09373 98.75865 275.70335 99.36826 276.04588 100.19519 276.04588 101.05731 276.04588 101.91943 275.70335 102.74637 275.09373 103.35598 274.4841 103.96559 273.65718 104.30811 272.79505 104.30811 271.93293 104.30811 271.10603 103.96559 270.4964 103.35598 269.88679 102.74637 269.54426 101.91943 269.54426 101.05731 269.54426 100.19519 269.88679 99.36826 270.4964 98.75865 271.10603 98.14904 271.93293 97.80651 272.79505 97.80651Z"/>
</g>
<g clip-path="url(#clip_5)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M266.0198 101.86222C266.84117 101.86222 267.62898 102.188549 268.20973 102.76932 268.7905 103.35009 269.11683 104.1379 269.11683 104.959239 269.11683 105.78057 268.7905 106.56838 268.20973 107.149158 267.62898 107.72993 266.84117 108.05625 266.0198 108.05625 265.1985 108.05625 264.41069 107.72993 263.8299 107.149158 263.2491 106.56838 262.9228 105.78057 262.9228 104.959239 262.9228 104.1379 263.2491 103.35009 263.8299 102.76932 264.41069 102.188549 265.1985 101.86222 266.0198 101.86222Z" fill="#87b3cb"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#456e87" d="M266.0198 101.86222C266.84117 101.86222 267.62898 102.188549 268.20973 102.76932 268.7905 103.35009 269.11683 104.1379 269.11683 104.959239 269.11683 105.78057 268.7905 106.56838 268.20973 107.149158 267.62898 107.72993 266.84117 108.05625 266.0198 108.05625 265.1985 108.05625 264.41069 107.72993 263.8299 107.149158 263.2491 106.56838 262.9228 105.78057 262.9228 104.959239 262.9228 104.1379 263.2491 103.35009 263.8299 102.76932 264.41069 102.188549 265.1985 101.86222 266.0198 101.86222Z"/>
</g>
<g clip-path="url(#clip_6)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M261.87083 105.26901C262.6437 105.26901 263.38505 105.57608 263.93156 106.12259 264.47807 106.6691 264.78514 107.41042 264.78514 108.183307 264.78514 108.956188 264.47807 109.69752 263.93156 110.244029 263.38505 110.790538 262.6437 111.0976 261.87083 111.0976 261.09794 111.0976 260.3566 110.790538 259.8101 110.244029 259.26359 109.69752 258.9565 108.956188 258.9565 108.183307 258.9565 107.41042 259.26359 106.6691 259.8101 106.12259 260.3566 105.57608 261.09794 105.26901 261.87083 105.26901Z" fill="#87b3cb"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#456e87" d="M261.87083 105.26901C262.6437 105.26901 263.38505 105.57608 263.93156 106.12259 264.47807 106.6691 264.78514 107.41042 264.78514 108.183307 264.78514 108.956188 264.47807 109.69752 263.93156 110.244029 263.38505 110.790538 262.6437 111.0976 261.87083 111.0976 261.09794 111.0976 260.3566 110.790538 259.8101 110.244029 259.26359 109.69752 258.9565 108.956188 258.9565 108.183307 258.9565 107.41042 259.26359 106.6691 259.8101 106.12259 260.3566 105.57608 261.09794 105.26901 261.87083 105.26901Z"/>
</g>
<g clip-path="url(#clip_7)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M249.76737 105.46063C250.5765 105.46063 251.35262 105.782108 251.92476 106.354259 252.49692 106.92641 252.81839 107.702518 252.81839 108.51166 252.81839 109.3208 252.49692 110.09691 251.92476 110.66906 251.35262 111.24121 250.5765 111.56268 249.76737 111.56268 248.95822 111.56268 248.18212 111.24121 247.60996 110.66906 247.03781 110.09691 246.71634 109.3208 246.71634 108.51166 246.71634 107.702518 247.03781 106.92641 247.60996 106.354259 248.18212 105.782108 248.95822 105.46063 249.76737 105.46063Z" fill="#87b3cb"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#456e87" d="M249.76737 105.46063C250.5765 105.46063 251.35262 105.782108 251.92476 106.354259 252.49692 106.92641 252.81839 107.702518 252.81839 108.51166 252.81839 109.3208 252.49692 110.09691 251.92476 110.66906 251.35262 111.24121 250.5765 111.56268 249.76737 111.56268 248.95822 111.56268 248.18212 111.24121 247.60996 110.66906 247.03781 110.09691 246.71634 109.3208 246.71634 108.51166 246.71634 107.702518 247.03781 106.92641 247.60996 106.354259 248.18212 105.782108 248.95822 105.46063 249.76737 105.46063Z"/>
</g>
<g clip-path="url(#clip_8)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M238.20763 91.03978C239.01134 91.03978 239.78224 91.35909 240.35056 91.92741 240.91887 92.49572 241.23819 93.266628 241.23819 94.070339 241.23819 94.874057 240.91887 95.64496 240.35056 96.213268 239.78224 96.78158 239.01134 97.1009 238.20763 97.1009 237.40392 97.1009 236.63301 96.78158 236.0647 96.213268 235.49639 95.64496 235.17707 94.874057 235.17707 94.070339 235.17707 93.266628 235.49639 92.49572 236.0647 91.92741 236.63301 91.35909 237.40392 91.03978 238.20763 91.03978Z" fill="#87b3cb"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#456e87" d="M238.20763 91.03978C239.01134 91.03978 239.78224 91.35909 240.35056 91.92741 240.91887 92.49572 241.23819 93.266628 241.23819 94.070339 241.23819 94.874057 240.91887 95.64496 240.35056 96.213268 239.78224 96.78158 239.01134 97.1009 238.20763 97.1009 237.40392 97.1009 236.63301 96.78158 236.0647 96.213268 235.49639 95.64496 235.17707 94.874057 235.17707 94.070339 235.17707 93.266628 235.49639 92.49572 236.0647 91.92741 236.63301 91.35909 237.40392 91.03978 238.20763 91.03978Z"/>
</g>
<g clip-path="url(#clip_9)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M288.3398 98.6635C289.0268 98.6635 289.68574 98.93643 290.1715 99.4222 290.65727 99.907978 290.9302 100.56691 290.9302 101.25389 290.9302 101.94087 290.65727 102.59981 290.1715 103.08558 289.68574 103.57135 289.0268 103.84429 288.3398 103.84429 287.65284 103.84429 286.9939 103.57135 286.50813 103.08558 286.02238 102.59981 285.74943 101.94087 285.74943 101.25389 285.74943 100.56691 286.02238 99.907978 286.50813 99.4222 286.9939 98.93643 287.65284 98.6635 288.3398 98.6635Z" fill="#87b3cb"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width="1" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#293844" d="M288.3398 98.6635C289.0268 98.6635 289.68574 98.93643 290.1715 99.4222 290.65727 99.907978 290.9302 100.56691 290.9302 101.25389 290.9302 101.94087 290.65727 102.59981 290.1715 103.08558 289.68574 103.57135 289.0268 103.84429 288.3398 103.84429 287.65284 103.84429 286.9939 103.57135 286.50813 103.08558 286.02238 102.59981 285.74943 101.94087 285.74943 101.25389 285.74943 100.56691 286.02238 99.907978 286.50813 99.4222 286.9939 98.93643 287.65284 98.6635 288.3398 98.6635Z"/>
</g>
<g clip-path="url(#clip_10)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M115.17731 54.870515C115.71685 54.870515 116.23437 55.084878 116.61588 55.466394 116.9974 55.847906 117.21176 56.36542 117.21176 56.904966 117.21176 57.444509 116.9974 57.962026 116.61588 58.34354 116.23437 58.725057 115.71685 58.93942 115.17731 58.93942 114.637767 58.93942 114.12025 58.725057 113.73873 58.34354 113.357219 57.962026 113.14285 57.444509 113.14285 56.904966 113.14285 56.36542 113.357219 55.847906 113.73873 55.466394 114.12025 55.084878 114.637767 54.870515 115.17731 54.870515Z" fill="#ccd6df"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M115.17731 54.870515C115.71685 54.870515 116.23437 55.084878 116.61588 55.466394 116.9974 55.847906 117.21176 56.36542 117.21176 56.904966 117.21176 57.444509 116.9974 57.962026 116.61588 58.34354 116.23437 58.725057 115.71685 58.93942 115.17731 58.93942 114.637767 58.93942 114.12025 58.725057 113.73873 58.34354 113.357219 57.962026 113.14285 57.444509 113.14285 56.904966 113.14285 56.36542 113.357219 55.847906 113.73873 55.466394 114.12025 55.084878 114.637767 54.870515 115.17731 54.870515Z"/>
</g>
<g clip-path="url(#clip_11)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M76.646358 32.941564C77.43747 32.941564 78.19629 33.255876 78.75569 33.815279 79.315097 34.37468 79.62941 35.1335 79.62941 35.924615 79.62941 36.71573 79.315097 37.47455 78.75569 38.03395 78.19629 38.593355 77.43747 38.90767 76.646358 38.90767 75.85524 38.90767 75.09642 38.593355 74.53702 38.03395 73.977619 37.47455 73.6633 36.71573 73.6633 35.924615 73.6633 35.1335 73.977619 34.37468 74.53702 33.815279 75.09642 33.255876 75.85524 32.941564 76.646358 32.941564Z" fill="#ccd6df"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M76.646358 32.941564C77.43747 32.941564 78.19629 33.255876 78.75569 33.815279 79.315097 34.37468 79.62941 35.1335 79.62941 35.924615 79.62941 36.71573 79.315097 37.47455 78.75569 38.03395 78.19629 38.593355 77.43747 38.90767 76.646358 38.90767 75.85524 38.90767 75.09642 38.593355 74.53702 38.03395 73.977619 37.47455 73.6633 36.71573 73.6633 35.924615 73.6633 35.1335 73.977619 34.37468 74.53702 33.815279 75.09642 33.255876 75.85524 32.941564 76.646358 32.941564Z"/>
</g>
<g clip-path="url(#clip_12)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M234.55899 63.05629C235.51313 63.05629 236.4283 63.435369 237.10298 64.11004 237.77765 64.78471 238.15673 65.6999 238.15673 66.65403 238.15673 67.608158 237.77765 68.52334 237.10298 69.19801 236.4283 69.87269 235.51313 70.25176 234.55899 70.25176 233.60486 70.25176 232.68968 69.87269 232.015 69.19801 231.34033 68.52334 230.96126 67.608158 230.96126 66.65403 230.96126 65.6999 231.34033 64.78471 232.015 64.11004 232.68968 63.435369 233.60486 63.05629 234.55899 63.05629Z" fill="#ccd6df"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M234.55899 63.05629C235.51313 63.05629 236.4283 63.435369 237.10298 64.11004 237.77765 64.78471 238.15673 65.6999 238.15673 66.65403 238.15673 67.608158 237.77765 68.52334 237.10298 69.19801 236.4283 69.87269 235.51313 70.25176 234.55899 70.25176 233.60486 70.25176 232.68968 69.87269 232.015 69.19801 231.34033 68.52334 230.96126 67.608158 230.96126 66.65403 230.96126 65.6999 231.34033 64.78471 232.015 64.11004 232.68968 63.435369 233.60486 63.05629 234.55899 63.05629Z"/>
</g>
<g clip-path="url(#clip_13)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M208.17214 94.90285C208.88362 94.90285 209.56606 95.185527 210.06914 95.688617 210.57224 96.1917 210.8549 96.87414 210.8549 97.58561 210.8549 98.29709 210.57224 98.97952 210.06914 99.48261 209.56606 99.9857 208.88362 100.26838 208.17214 100.26838 207.46067 100.26838 206.77823 99.9857 206.27513 99.48261 205.77205 98.97952 205.48938 98.29709 205.48938 97.58561 205.48938 96.87414 205.77205 96.1917 206.27513 95.688617 206.77823 95.185527 207.46067 94.90285 208.17214 94.90285Z" fill="#ccd6df"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M208.17214 94.90285C208.88362 94.90285 209.56606 95.185527 210.06914 95.688617 210.57224 96.1917 210.8549 96.87414 210.8549 97.58561 210.8549 98.29709 210.57224 98.97952 210.06914 99.48261 209.56606 99.9857 208.88362 100.26838 208.17214 100.26838 207.46067 100.26838 206.77823 99.9857 206.27513 99.48261 205.77205 98.97952 205.48938 98.29709 205.48938 97.58561 205.48938 96.87414 205.77205 96.1917 206.27513 95.688617 206.77823 95.185527 207.46067 94.90285 208.17214 94.90285Z"/>
</g>
<g clip-path="url(#clip_14)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M256.95979 59.883405C257.84819 59.883405 258.70033 60.236368 259.32853 60.864564 259.9567 61.492757 260.30967 62.34489 260.30967 63.23329 260.30967 64.12169 259.9567 64.97382 259.32853 65.60202 258.70033 66.23021 257.84819 66.583179 256.95979 66.583179 256.07139 66.583179 255.21926 66.23021 254.59107 65.60202 253.96286 64.97382 253.6099 64.12169 253.6099 63.23329 253.6099 62.34489 253.96286 61.492757 254.59107 60.864564 255.21926 60.236368 256.07139 59.883405 256.95979 59.883405Z" fill="#ccd6df"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M256.95979 59.883405C257.84819 59.883405 258.70033 60.236368 259.32853 60.864564 259.9567 61.492757 260.30967 62.34489 260.30967 63.23329 260.30967 64.12169 259.9567 64.97382 259.32853 65.60202 258.70033 66.23021 257.84819 66.583179 256.95979 66.583179 256.07139 66.583179 255.21926 66.23021 254.59107 65.60202 253.96286 64.97382 253.6099 64.12169 253.6099 63.23329 253.6099 62.34489 253.96286 61.492757 254.59107 60.864564 255.21926 60.236368 256.07139 59.883405 256.95979 59.883405Z"/>
</g>
<g clip-path="url(#clip_15)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M149.00801 75.03411C149.65602 75.03411 150.27758 75.291568 150.73578 75.74978 151.194 76.20799 151.45145 76.829547 151.45145 77.477558 151.45145 78.125568 151.194 78.74712 150.73578 79.20533 150.27758 79.66354 149.65602 79.921 149.00801 79.921 148.36 79.921 147.73844 79.66354 147.28023 79.20533 146.82202 78.74712 146.56456 78.125568 146.56456 77.477558 146.56456 76.829547 146.82202 76.20799 147.28023 75.74978 147.73844 75.291568 148.36 75.03411 149.00801 75.03411Z" fill="#ccd6df"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M149.00801 75.03411C149.65602 75.03411 150.27758 75.291568 150.73578 75.74978 151.194 76.20799 151.45145 76.829547 151.45145 77.477558 151.45145 78.125568 151.194 78.74712 150.73578 79.20533 150.27758 79.66354 149.65602 79.921 149.00801 79.921 148.36 79.921 147.73844 79.66354 147.28023 79.20533 146.82202 78.74712 146.56456 78.125568 146.56456 77.477558 146.56456 76.829547 146.82202 76.20799 147.28023 75.74978 147.73844 75.291568 148.36 75.03411 149.00801 75.03411Z"/>
</g>
<g clip-path="url(#clip_16)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M206.54537 83.58646C207.25676 83.58646 207.93914 83.8691 208.44217 84.37214 208.9452 84.875179 209.22785 85.55754 209.22785 86.26894 209.22785 86.980358 208.9452 87.66271 208.44217 88.16576 207.93914 88.66879 207.25676 88.95144 206.54537 88.95144 205.83396 88.95144 205.1516 88.66879 204.64855 88.16576 204.14551 87.66271 203.86287 86.980358 203.86287 86.26894 203.86287 85.55754 204.14551 84.875179 204.64855 84.37214 205.1516 83.8691 205.83396 83.58646 206.54537 83.58646Z" fill="#ccd6df"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M206.54537 83.58646C207.25676 83.58646 207.93914 83.8691 208.44217 84.37214 208.9452 84.875179 209.22785 85.55754 209.22785 86.26894 209.22785 86.980358 208.9452 87.66271 208.44217 88.16576 207.93914 88.66879 207.25676 88.95144 206.54537 88.95144 205.83396 88.95144 205.1516 88.66879 204.64855 88.16576 204.14551 87.66271 203.86287 86.980358 203.86287 86.26894 203.86287 85.55754 204.14551 84.875179 204.64855 84.37214 205.1516 83.8691 205.83396 83.58646 206.54537 83.58646Z"/>
</g>
<g clip-path="url(#clip_17)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M236.06554 84.95782C236.94675 84.95782 237.79198 85.30792 238.41509 85.93104 239.0382 86.554149 239.3883 87.39938 239.3883 88.280597 239.3883 89.161808 239.0382 90.00704 238.41509 90.63015 237.79198 91.253269 236.94675 91.60337 236.06554 91.60337 235.18433 91.60337 234.33908 91.253269 233.71598 90.63015 233.09287 90.00704 232.74275 89.161808 232.74275 88.280597 232.74275 87.39938 233.09287 86.554149 233.71598 85.93104 234.33908 85.30792 235.18433 84.95782 236.06554 84.95782Z" fill="#ccd6df"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M236.06554 84.95782C236.94675 84.95782 237.79198 85.30792 238.41509 85.93104 239.0382 86.554149 239.3883 87.39938 239.3883 88.280597 239.3883 89.161808 239.0382 90.00704 238.41509 90.63015 237.79198 91.253269 236.94675 91.60337 236.06554 91.60337 235.18433 91.60337 234.33908 91.253269 233.71598 90.63015 233.09287 90.00704 232.74275 89.161808 232.74275 88.280597 232.74275 87.39938 233.09287 86.554149 233.71598 85.93104 234.33908 85.30792 235.18433 84.95782 236.06554 84.95782Z"/>
</g>
<g clip-path="url(#clip_18)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M269.333 104.44925C270.199 104.44925 271.02967 104.79331 271.64204 105.40568 272.2544 106.018039 272.59846 106.848697 272.59846 107.7147 272.59846 108.5807 272.2544 109.41136 271.64204 110.02372 271.02967 110.636089 270.199 110.98015 269.333 110.98015 268.467 110.98015 267.63636 110.636089 267.024 110.02372 266.41163 109.41136 266.06758 108.5807 266.06758 107.7147 266.06758 106.848697 266.41163 106.018039 267.024 105.40568 267.63636 104.79331 268.467 104.44925 269.333 104.44925Z" fill="#e58b65"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#b65332" d="M269.333 104.44925C270.199 104.44925 271.02967 104.79331 271.64204 105.40568 272.2544 106.018039 272.59846 106.848697 272.59846 107.7147 272.59846 108.5807 272.2544 109.41136 271.64204 110.02372 271.02967 110.636089 270.199 110.98015 269.333 110.98015 268.467 110.98015 267.63636 110.636089 267.024 110.02372 266.41163 109.41136 266.06758 108.5807 266.06758 107.7147 266.06758 106.848697 266.41163 106.018039 267.024 105.40568 267.63636 104.79331 268.467 104.44925 269.333 104.44925Z"/>
</g>
<g clip-path="url(#clip_19)">
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M283.95637 103.747409C284.83018 103.747409 285.6683 104.094577 286.2862 104.71245 286.90406 105.33032 287.25123 106.16846 287.25123 107.04227 287.25123 107.91608 286.90406 108.75421 286.2862 109.372089 285.6683 109.98996 284.83018 110.33713 283.95637 110.33713 283.08256 110.33713 282.24443 109.98996 281.62657 109.372089 281.00868 108.75421 280.6615 107.91608 280.6615 107.04227 280.6615 106.16846 281.00868 105.33032 281.62657 104.71245 282.24443 104.094577 283.08256 103.747409 283.95637 103.747409Z" fill="#e58b65"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width="1" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#293844" d="M283.95637 103.747409C284.83018 103.747409 285.6683 104.094577 286.2862 104.71245 286.90406 105.33032 287.25123 106.16846 287.25123 107.04227 287.25123 107.91608 286.90406 108.75421 286.2862 109.372089 285.6683 109.98996 284.83018 110.33713 283.95637 110.33713 283.08256 110.33713 282.24443 109.98996 281.62657 109.372089 281.00868 108.75421 280.6615 107.91608 280.6615 107.04227 280.6615 106.16846 281.00868 105.33032 281.62657 104.71245 282.24443 104.094577 283.08256 103.747409 283.95637 103.747409Z"/>
</g>
<use data-text="S" xlink:href="#font_2_27" transform="matrix(8.2,0,0,-8.2,81.646358,128.85242)" fill="#4e5b65"/>
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,86.20555,128.85242)" fill="#4e5b65"/>
<use data-text="n" xlink:href="#font_2_44" transform="matrix(8.2,0,0,-8.2,90.30556,128.85242)" fill="#4e5b65"/>
<use data-text="g" xlink:href="#font_2_38" transform="matrix(8.2,0,0,-8.2,94.405559,128.85242)" fill="#4e5b65"/>
<use data-text="B" xlink:href="#font_2_17" transform="matrix(8.2,0,0,-8.2,98.505558,128.85242)" fill="#4e5b65"/>
<use data-text="l" xlink:href="#font_2_42" transform="matrix(8.2,0,0,-8.2,103.97496,128.85242)" fill="#4e5b65"/>
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,106.254558,128.85242)" fill="#4e5b65"/>
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,110.35455,128.85242)" fill="#4e5b65"/>
<use data-text="m" xlink:href="#font_2_43" transform="matrix(8.2,0,0,-8.2,114.45456,128.85242)" fill="#4e5b65"/>
<use data-text="Y" xlink:href="#font_2_31" transform="matrix(8.2,0,0,-8.2,105.98981,106.88768)" fill="#4e5b65"/>
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,111.00396,106.88768)" fill="#4e5b65"/>
<use data-text="E" xlink:href="#font_2_20" transform="matrix(8.2,0,0,-8.2,115.10396,106.88768)" fill="#4e5b65"/>
<use data-text="1" xlink:href="#font_2_7" transform="matrix(8.2,0,0,-8.2,120.11416,106.88768)" fill="#4e5b65"/>
<use data-text="D" xlink:href="#font_2_19" transform="matrix(8.2,0,0,-8.2,98.60175,86.346347)" fill="#4e5b65"/>
<use data-text="i" xlink:href="#font_2_40" transform="matrix(8.2,0,0,-8.2,104.522159,86.346347)" fill="#4e5b65"/>
<use data-text="f" xlink:href="#font_2_37" transform="matrix(8.2,0,0,-8.2,106.80176,86.346347)" fill="#4e5b65"/>
<use data-text="f" xlink:href="#font_2_37" transform="matrix(8.2,0,0,-8.2,109.39173,86.346347)" fill="#4e5b65"/>
<use data-text="R" xlink:href="#font_2_26" transform="matrix(8.2,0,0,-8.2,112.12233,86.346347)" fill="#4e5b65"/>
<use data-text="h" xlink:href="#font_2_39" transform="matrix(8.2,0,0,-8.2,117.59173,86.346347)" fill="#4e5b65"/>
<use data-text="y" xlink:href="#font_2_54" transform="matrix(8.2,0,0,-8.2,121.69173,86.346347)" fill="#4e5b65"/>
<use data-text="t" xlink:href="#font_2_50" transform="matrix(8.2,0,0,-8.2,125.79173,86.346347)" fill="#4e5b65"/>
<use data-text="h" xlink:href="#font_2_39" transform="matrix(8.2,0,0,-8.2,128.07134,86.346347)" fill="#4e5b65"/>
<use data-text="m" xlink:href="#font_2_43" transform="matrix(8.2,0,0,-8.2,132.17133,86.346347)" fill="#4e5b65"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,138.55094,86.346347)" fill="#4e5b65"/>
<use data-text="2" xlink:href="#font_2_8" transform="matrix(8.2,0,0,-8.2,140.60092,86.346347)" fill="#4e5b65"/>
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,183.20162,92.523708)" fill="#4e5b65"/>
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,190.49141,92.523708)" fill="#4e5b65"/>
<use data-text="s" xlink:href="#font_2_49" transform="matrix(8.2,0,0,-8.2,194.59142,92.523708)" fill="#4e5b65"/>
<use data-text="e" xlink:href="#font_2_36" transform="matrix(8.2,0,0,-8.2,197.78122,92.523708)" fill="#4e5b65"/>
<use data-text="L" xlink:href="#font_2_22" transform="matrix(8.2,0,0,-8.2,205.71524,112.138629)" fill="#4e5b65"/>
<use data-text="e" xlink:href="#font_2_36" transform="matrix(8.2,0,0,-8.2,210.72544,112.138629)" fill="#4e5b65"/>
<use data-text="V" xlink:href="#font_2_29" transform="matrix(8.2,0,0,-8.2,214.36624,112.138629)" fill="#4e5b65"/>
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,219.22414,112.138629)" fill="#4e5b65"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,223.32415,112.138629)" fill="#4e5b65"/>
<use data-text="2" xlink:href="#font_2_8" transform="matrix(8.2,0,0,-8.2,225.37415,112.138629)" fill="#4e5b65"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#4e5b65" d="M249.29737 75.38554H247.29346L238.77109 85.17332"/>
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,251.29346,97.40711)" fill="#4e5b65"/>
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,258.58326,97.40711)" fill="#4e5b65"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,265.87306,97.40711)" fill="#4e5b65"/>
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,267.92308,97.40711)" fill="#4e5b65"/>
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,275.21287,97.40711)" fill="#4e5b65"/>
<use data-text="s" xlink:href="#font_2_49" transform="matrix(8.2,0,0,-8.2,279.31288,97.40711)" fill="#4e5b65"/>
<use data-text="i" xlink:href="#font_2_40" transform="matrix(8.2,0,0,-8.2,282.50267,97.40711)" fill="#4e5b65"/>
<use data-text="c" xlink:href="#font_2_34" transform="matrix(8.2,0,0,-8.2,284.78227,97.40711)" fill="#4e5b65"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,288.42308,97.40711)" fill="#4e5b65"/>
<use data-text="3" xlink:href="#font_2_9" transform="matrix(8.2,0,0,-8.2,290.47306,97.40711)" fill="#4e5b65"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#456e87" d="M249.29737 85.010158H247.29346L240.9205 91.36511"/>
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,251.29346,87.78249)" fill="#456e87"/>
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,258.58326,87.78249)" fill="#456e87"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,265.87306,87.78249)" fill="#456e87"/>
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,267.92308,87.78249)" fill="#456e87"/>
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,275.21287,87.78249)" fill="#456e87"/>
<use data-text="s" xlink:href="#font_2_49" transform="matrix(8.2,0,0,-8.2,279.31288,87.78249)" fill="#456e87"/>
<use data-text="i" xlink:href="#font_2_40" transform="matrix(8.2,0,0,-8.2,282.50267,87.78249)" fill="#456e87"/>
<use data-text="c" xlink:href="#font_2_34" transform="matrix(8.2,0,0,-8.2,284.78227,87.78249)" fill="#456e87"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,288.42308,87.78249)" fill="#456e87"/>
<use data-text="2" xlink:href="#font_2_8" transform="matrix(8.2,0,0,-8.2,290.47306,87.78249)" fill="#456e87"/>
<use data-text="." xlink:href="#font_2_5" transform="matrix(8.2,0,0,-8.2,294.57307,87.78249)" fill="#456e87"/>
<use data-text="6" xlink:href="#font_2_12" transform="matrix(8.2,0,0,-8.2,296.62306,87.78249)" fill="#456e87"/>
<use data-text="A" xlink:href="#font_2_16" transform="matrix(8.2,0,0,-8.2,154.87526,74.20704)" fill="#4e5b65"/>
<use data-text="C" xlink:href="#font_2_18" transform="matrix(8.2,0,0,-8.2,160.79566,74.20704)" fill="#4e5b65"/>
<use data-text="E" xlink:href="#font_2_20" transform="matrix(8.2,0,0,-8.2,166.26506,74.20704)" fill="#4e5b65"/>
<use data-text="-" xlink:href="#font_2_4" transform="matrix(8.2,0,0,-8.2,171.27527,74.20704)" fill="#4e5b65"/>
<use data-text="S" xlink:href="#font_2_27" transform="matrix(8.2,0,0,-8.2,174.00586,74.20704)" fill="#4e5b65"/>
<use data-text="t" xlink:href="#font_2_50" transform="matrix(8.2,0,0,-8.2,178.56507,74.20704)" fill="#4e5b65"/>
<use data-text="e" xlink:href="#font_2_36" transform="matrix(8.2,0,0,-8.2,180.84467,74.20704)" fill="#4e5b65"/>
<use data-text="p" xlink:href="#font_2_46" transform="matrix(8.2,0,0,-8.2,184.48546,74.20704)" fill="#4e5b65"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,188.58547,74.20704)" fill="#4e5b65"/>
<use data-text="1" xlink:href="#font_2_7" transform="matrix(8.2,0,0,-8.2,190.63547,74.20704)" fill="#4e5b65"/>
<use data-text="." xlink:href="#font_2_5" transform="matrix(8.2,0,0,-8.2,194.73546,74.20704)" fill="#4e5b65"/>
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8.2,0,0,-8.2,196.78546,74.20704)" fill="#4e5b65"/>
<use data-text="H" xlink:href="#font_2_21" transform="matrix(8.2,0,0,-8.2,261.95979,115.55936)" fill="#4e5b65"/>
<use data-text="e" xlink:href="#font_2_36" transform="matrix(8.2,0,0,-8.2,267.8802,115.55936)" fill="#4e5b65"/>
<use data-text="a" xlink:href="#font_2_32" transform="matrix(8.2,0,0,-8.2,271.52098,115.55936)" fill="#4e5b65"/>
<use data-text="r" xlink:href="#font_2_48" transform="matrix(8.2,0,0,-8.2,275.16178,115.55936)" fill="#4e5b65"/>
<use data-text="t" xlink:href="#font_2_50" transform="matrix(8.2,0,0,-8.2,277.89237,115.55936)" fill="#4e5b65"/>
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,280.17198,115.55936)" fill="#4e5b65"/>
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,287.4618,115.55936)" fill="#4e5b65"/>
<use data-text="L" xlink:href="#font_2_22" transform="matrix(8.2,0,0,-8.2,291.56178,115.55936)" fill="#4e5b65"/>
<use data-text="a" xlink:href="#font_2_32" transform="matrix(8.2,0,0,-8.2,296.572,115.55936)" fill="#4e5b65"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#456e87" d="M304.86689 131.20832H302.86299L284.8672 115.44177"/>
<use data-text="S" xlink:href="#font_2_27" transform="matrix(8.2,0,0,-8.2,306.86299,41.584337)" fill="#456e87"/>
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,311.42219,41.584337)" fill="#456e87"/>
<use data-text="n" xlink:href="#font_2_44" transform="matrix(8.2,0,0,-8.2,315.5222,41.584337)" fill="#456e87"/>
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,319.6222,41.584337)" fill="#456e87"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,323.72218,41.584337)" fill="#456e87"/>
<use data-text="v" xlink:href="#font_2_52" transform="matrix(8.2,0,0,-8.2,325.7722,41.584337)" fill="#456e87"/>
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8.2,0,0,-8.2,329.8722,41.584337)" fill="#456e87"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#456e87" d="M247.21591 103.959239H249.21982L266.0198 96.959239V101.064708"/>
<use data-text="S" xlink:href="#font_2_27" transform="matrix(8.2,0,0,-8.2,211.75107,68.83341)" fill="#456e87"/>
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,216.31027,68.83341)" fill="#456e87"/>
<use data-text="n" xlink:href="#font_2_44" transform="matrix(8.2,0,0,-8.2,220.41027,68.83341)" fill="#456e87"/>
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,224.51027,68.83341)" fill="#456e87"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,228.61028,68.83341)" fill="#456e87"/>
<use data-text="v" xlink:href="#font_2_52" transform="matrix(8.2,0,0,-8.2,230.66027,68.83341)" fill="#456e87"/>
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8.2,0,0,-8.2,234.76027,68.83341)" fill="#456e87"/>
<use data-text="." xlink:href="#font_2_5" transform="matrix(8.2,0,0,-8.2,238.86028,68.83341)" fill="#456e87"/>
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8.2,0,0,-8.2,240.91027,68.83341)" fill="#456e87"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#456e87" d="M255.94353 144.68277H257.93965L261.473 111.876918"/>
<use data-text="S" xlink:href="#font_2_27" transform="matrix(8.2,0,0,-8.2,226.65837,28.109879)" fill="#456e87"/>
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,231.21758,28.109879)" fill="#456e87"/>
<use data-text="n" xlink:href="#font_2_44" transform="matrix(8.2,0,0,-8.2,235.31757,28.109879)" fill="#456e87"/>
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,239.41757,28.109879)" fill="#456e87"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,243.51758,28.109879)" fill="#456e87"/>
<use data-text="v" xlink:href="#font_2_52" transform="matrix(8.2,0,0,-8.2,245.56757,28.109879)" fill="#456e87"/>
<use data-text="6" xlink:href="#font_2_12" transform="matrix(8.2,0,0,-8.2,249.66757,28.109879)" fill="#456e87"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#456e87" d="M226.82787 131.20832H228.83177L247.15808 111.34042"/>
<use data-text="S" xlink:href="#font_2_27" transform="matrix(8.2,0,0,-8.2,179.48802,41.584337)" fill="#456e87"/>
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,184.04723,41.584337)" fill="#456e87"/>
<use data-text="n" xlink:href="#font_2_44" transform="matrix(8.2,0,0,-8.2,188.14722,41.584337)" fill="#456e87"/>
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,192.24723,41.584337)" fill="#456e87"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,196.34723,41.584337)" fill="#456e87"/>
<use data-text="v" xlink:href="#font_2_52" transform="matrix(8.2,0,0,-8.2,198.39722,41.584337)" fill="#456e87"/>
<use data-text="6" xlink:href="#font_2_12" transform="matrix(8.2,0,0,-8.2,202.49723,41.584337)" fill="#456e87"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,206.59723,41.584337)" fill="#456e87"/>
<use data-text="W" xlink:href="#font_2_30" transform="matrix(8.2,0,0,-8.2,208.50659,41.584337)" fill="#456e87"/>
<use data-text="i" xlink:href="#font_2_40" transform="matrix(8.2,0,0,-8.2,215.91928,41.584337)" fill="#456e87"/>
<use data-text="l" xlink:href="#font_2_42" transform="matrix(8.2,0,0,-8.2,218.19889,41.584337)" fill="#456e87"/>
<use data-text="d" xlink:href="#font_2_35" transform="matrix(8.2,0,0,-8.2,220.47849,41.584337)" fill="#456e87"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#456e87" d="M304.86689 90.78492H302.86299L276.6306 99.74693"/>
<use data-text="S" xlink:href="#font_2_27" transform="matrix(8.2,0,0,-8.2,306.86299,82.00773)" fill="#456e87"/>
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,311.42219,82.00773)" fill="#456e87"/>
<use data-text="n" xlink:href="#font_2_44" transform="matrix(8.2,0,0,-8.2,315.5222,82.00773)" fill="#456e87"/>
<use data-text="o" xlink:href="#font_2_45" transform="matrix(8.2,0,0,-8.2,319.6222,82.00773)" fill="#456e87"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,323.72218,82.00773)" fill="#456e87"/>
<use data-text="v" xlink:href="#font_2_52" transform="matrix(8.2,0,0,-8.2,325.7722,82.00773)" fill="#456e87"/>
<use data-text="4" xlink:href="#font_2_10" transform="matrix(8.2,0,0,-8.2,329.8722,82.00773)" fill="#456e87"/>
<use data-text="." xlink:href="#font_2_5" transform="matrix(8.2,0,0,-8.2,333.97218,82.00773)" fill="#456e87"/>
<use data-text="5" xlink:href="#font_2_11" transform="matrix(8.2,0,0,-8.2,336.0222,82.00773)" fill="#456e87"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#b65332" d="M279.46369 144.68277H277.45979L270.20503 111.681369"/>
<use data-text="Y" xlink:href="#font_20_10" transform="matrix(9.3,0,0,-9.3,281.45979,28.352066)" fill="#b65332"/>
<use data-text="u" xlink:href="#font_20_12" transform="matrix(9.3,0,0,-9.3,287.33064,28.352066)" fill="#b65332"/>
<use data-text="E" xlink:href="#font_20_9" transform="matrix(9.3,0,0,-9.3,292.50144,28.352066)" fill="#b65332"/>
<use data-text="2" xlink:href="#font_20_6" transform="matrix(9.3,0,0,-9.3,298.70454,28.352066)" fill="#b65332"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#b65332" d="M304.86689 117.73385H302.86299L287.52214 109.058689"/>
<use data-text="Y" xlink:href="#font_20_10" transform="matrix(9.3,0,0,-9.3,306.86299,55.300989)" fill="#b65332"/>
<use data-text="u" xlink:href="#font_20_12" transform="matrix(9.3,0,0,-9.3,312.73384,55.300989)" fill="#b65332"/>
<use data-text="E" xlink:href="#font_20_9" transform="matrix(9.3,0,0,-9.3,317.90464,55.300989)" fill="#b65332"/>
<use data-text="2" xlink:href="#font_20_6" transform="matrix(9.3,0,0,-9.3,324.10774,55.300989)" fill="#b65332"/>
<use data-text=" " xlink:href="#font_20_3" transform="matrix(9.3,0,0,-9.3,328.75773,55.300989)" fill="#b65332"/>
<use data-text="(" xlink:href="#font_20_4" transform="matrix(9.3,0,0,-9.3,331.08274,55.300989)" fill="#b65332"/>
<use data-text="B" xlink:href="#font_20_8" transform="matrix(9.3,0,0,-9.3,334.17964,55.300989)" fill="#b65332"/>
<use data-text="o" xlink:href="#font_20_11" transform="matrix(9.3,0,0,-9.3,340.38273,55.300989)" fill="#b65332"/>
<use data-text="8" xlink:href="#font_20_7" transform="matrix(9.3,0,0,-9.3,345.0327,55.300989)" fill="#b65332"/>
<use data-text=")" xlink:href="#font_20_5" transform="matrix(9.3,0,0,-9.3,349.68275,55.300989)" fill="#b65332"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="round" stroke-linejoin="round" fill="none" stroke="#456e87" d="M304.86689 104.259387H302.86299L291.66215 101.94143"/>
<use data-text="M" xlink:href="#font_2_23" transform="matrix(8.2,0,0,-8.2,306.86299,68.533267)" fill="#456e87"/>
<use data-text="u" xlink:href="#font_2_51" transform="matrix(8.2,0,0,-8.2,314.15278,68.533267)" fill="#456e87"/>
<use data-text="r" xlink:href="#font_2_48" transform="matrix(8.2,0,0,-8.2,318.25279,68.533267)" fill="#456e87"/>
<use data-text="e" xlink:href="#font_2_36" transform="matrix(8.2,0,0,-8.2,320.98338,68.533267)" fill="#456e87"/>
<use data-text="k" xlink:href="#font_2_41" transform="matrix(8.2,0,0,-8.2,324.62419,68.533267)" fill="#456e87"/>
<use data-text="a" xlink:href="#font_2_32" transform="matrix(8.2,0,0,-8.2,328.72419,68.533267)" fill="#456e87"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(8.2,0,0,-8.2,332.365,68.533267)" fill="#456e87"/>
<use data-text="9" xlink:href="#font_2_15" transform="matrix(8.2,0,0,-8.2,334.41499,68.533267)" fill="#456e87"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M277.2 160.63213C277.77293 160.63213 278.32243 160.85974 278.72755 161.26485 279.13264 161.66996 279.36027 162.21947 279.36027 162.79238 279.36027 163.36528 279.13264 163.9148 278.72755 164.3199 278.32243 164.725 277.77293 164.95262 277.2 164.95262 276.6271 164.95262 276.07759 164.725 275.6725 164.3199 275.26737 163.9148 275.03977 163.36528 275.03977 162.79238 275.03977 162.21947 275.26737 161.66996 275.6725 161.26485 276.07759 160.85974 276.6271 160.63213 277.2 160.63213Z" fill="#dde4e9"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".45" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M277.2 160.63213C277.77293 160.63213 278.32243 160.85974 278.72755 161.26485 279.13264 161.66996 279.36027 162.21947 279.36027 162.79238 279.36027 163.36528 279.13264 163.9148 278.72755 164.3199 278.32243 164.725 277.77293 164.95262 277.2 164.95262 276.6271 164.95262 276.07759 164.725 275.6725 164.3199 275.26737 163.9148 275.03977 163.36528 275.03977 162.79238 275.03977 162.21947 275.26737 161.66996 275.6725 161.26485 276.07759 160.85974 276.6271 160.63213 277.2 160.63213Z"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M314.82 159.96395C315.5701 159.96395 316.28959 160.26197 316.82 160.79238 317.3504 161.32277 317.64845 162.04227 317.64845 162.79238 317.64845 163.54248 317.3504 164.26197 316.82 164.79238 316.28959 165.32277 315.5701 165.62079 314.82 165.62079 314.0699 165.62079 313.3504 165.32277 312.82 164.79238 312.28959 164.26197 311.99159 163.54248 311.99159 162.79238 311.99159 162.04227 312.28959 161.32277 312.82 160.79238 313.3504 160.26197 314.0699 159.96395 314.82 159.96395Z" fill="#dde4e9"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".45" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M314.82 159.96395C315.5701 159.96395 316.28959 160.26197 316.82 160.79238 317.3504 161.32277 317.64845 162.04227 317.64845 162.79238 317.64845 163.54248 317.3504 164.26197 316.82 164.79238 316.28959 165.32277 315.5701 165.62079 314.82 165.62079 314.0699 165.62079 313.3504 165.32277 312.82 164.79238 312.28959 164.26197 311.99159 163.54248 311.99159 162.79238 311.99159 162.04227 312.28959 161.32277 312.82 160.79238 313.3504 160.26197 314.0699 159.96395 314.82 159.96395Z"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M352.44 159.42588C353.3328 159.42588 354.18919 159.78058 354.82048 160.4119 355.45179 161.0432 355.8065 161.89957 355.8065 162.79238 355.8065 163.68518 355.45179 164.54154 354.82048 165.17285 354.18919 165.80416 353.3328 166.15888 352.44 166.15888 351.54719 166.15888 350.69084 165.80416 350.0595 165.17285 349.42823 164.54154 349.0735 163.68518 349.0735 162.79238 349.0735 161.89957 349.42823 161.0432 350.0595 160.4119 350.69084 159.78058 351.54719 159.42588 352.44 159.42588Z" fill="#dde4e9"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".45" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M352.44 159.42588C353.3328 159.42588 354.18919 159.78058 354.82048 160.4119 355.45179 161.0432 355.8065 161.89957 355.8065 162.79238 355.8065 163.68518 355.45179 164.54154 354.82048 165.17285 354.18919 165.80416 353.3328 166.15888 352.44 166.15888 351.54719 166.15888 350.69084 165.80416 350.0595 165.17285 349.42823 164.54154 349.0735 163.68518 349.0735 162.79238 349.0735 161.89957 349.42823 161.0432 350.0595 160.4119 350.69084 159.78058 351.54719 159.42588 352.44 159.42588Z"/>
<use data-text="A" xlink:href="#font_2_16" transform="matrix(7.4,0,0,-7.4,223.74,9.7659)" fill="#4e5b65"/>
<use data-text="u" xlink:href="#font_2_51" transform="matrix(7.4,0,0,-7.4,229.08281,9.7659)" fill="#4e5b65"/>
<use data-text="d" xlink:href="#font_2_35" transform="matrix(7.4,0,0,-7.4,232.7828,9.7659)" fill="#4e5b65"/>
<use data-text="i" xlink:href="#font_2_40" transform="matrix(7.4,0,0,-7.4,236.4828,9.7659)" fill="#4e5b65"/>
<use data-text="o" xlink:href="#font_2_45" transform="matrix(7.4,0,0,-7.4,238.54001,9.7659)" fill="#4e5b65"/>
<use data-text="B" xlink:href="#font_2_17" transform="matrix(7.4,0,0,-7.4,242.24,9.7659)" fill="#4e5b65"/>
<use data-text="o" xlink:href="#font_2_45" transform="matrix(7.4,0,0,-7.4,247.17581,9.7659)" fill="#4e5b65"/>
<use data-text="x" xlink:href="#font_2_53" transform="matrix(7.4,0,0,-7.4,250.87581,9.7659)" fill="#4e5b65"/>
<use data-text=" " xlink:href="#font_2_3" transform="matrix(7.4,0,0,-7.4,254.5758,9.7659)" fill="#4e5b65"/>
<use data-text="P" xlink:href="#font_2_24" transform="matrix(7.4,0,0,-7.4,256.4258,9.7659)" fill="#4e5b65"/>
<use data-text="Q" xlink:href="#font_2_25" transform="matrix(7.4,0,0,-7.4,260.54023,9.7659)" fill="#4e5b65"/>
<use data-text="7" xlink:href="#font_2_13" transform="matrix(7.4,0,0,-7.4,284.328,9.7659)" fill="#4e5b65"/>
<use data-text="." xlink:href="#font_2_5" transform="matrix(7.4,0,0,-7.4,288.028,9.7659)" fill="#4e5b65"/>
<use data-text="9" xlink:href="#font_2_15" transform="matrix(7.4,0,0,-7.4,289.878,9.7659)" fill="#4e5b65"/>
<use data-text="8" xlink:href="#font_2_14" transform="matrix(7.4,0,0,-7.4,321.948,9.7659)" fill="#4e5b65"/>
<use data-text="." xlink:href="#font_2_5" transform="matrix(7.4,0,0,-7.4,325.648,9.7659)" fill="#4e5b65"/>
<use data-text="1" xlink:href="#font_2_7" transform="matrix(7.4,0,0,-7.4,327.498,9.7659)" fill="#4e5b65"/>
<use data-text="8" xlink:href="#font_2_14" transform="matrix(7.4,0,0,-7.4,359.568,9.7659)" fill="#4e5b65"/>
<use data-text="." xlink:href="#font_2_5" transform="matrix(7.4,0,0,-7.4,363.268,9.7659)" fill="#4e5b65"/>
<use data-text="3" xlink:href="#font_2_9" transform="matrix(7.4,0,0,-7.4,365.11799,9.7659)" fill="#4e5b65"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M43.3 163.02677C43.96301 163.02677 44.59895 163.29019 45.06777 163.759 45.536584 164.22782 45.8 164.86376 45.8 165.52677 45.8 166.18978 45.536584 166.82572 45.06777 167.29454 44.59895 167.76335 43.96301 168.02677 43.3 168.02677 42.636995 168.02677 42.00105 167.76335 41.532236 167.29454 41.063417 166.82572 40.8 166.18978 40.8 165.52677 40.8 164.86376 41.063417 164.22782 41.532236 163.759 42.00105 163.29019 42.636995 163.02677 43.3 163.02677Z" fill="#ccd6df"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#8292a0" d="M43.3 163.02677C43.96301 163.02677 44.59895 163.29019 45.06777 163.759 45.536584 164.22782 45.8 164.86376 45.8 165.52677 45.8 166.18978 45.536584 166.82572 45.06777 167.29454 44.59895 167.76335 43.96301 168.02677 43.3 168.02677 42.636995 168.02677 42.00105 167.76335 41.532236 167.29454 41.063417 166.82572 40.8 166.18978 40.8 165.52677 40.8 164.86376 41.063417 164.22782 41.532236 163.759 42.00105 163.29019 42.636995 163.02677 43.3 163.02677Z"/>
<use data-text="P" xlink:href="#font_2_24" transform="matrix(7.4,0,0,-7.4,50.33,7.9340059)" fill="#293844"/>
<use data-text="u" xlink:href="#font_2_51" transform="matrix(7.4,0,0,-7.4,54.4444,7.9340059)" fill="#293844"/>
<use data-text="b" xlink:href="#font_2_33" transform="matrix(7.4,0,0,-7.4,58.1444,7.9340059)" fill="#293844"/>
<use data-text="l" xlink:href="#font_2_42" transform="matrix(7.4,0,0,-7.4,61.844404,7.9340059)" fill="#293844"/>
<use data-text="i" xlink:href="#font_2_40" transform="matrix(7.4,0,0,-7.4,63.901605,7.9340059)" fill="#293844"/>
<use data-text="c" xlink:href="#font_2_34" transform="matrix(7.4,0,0,-7.4,65.9588,7.9340059)" fill="#293844"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M80.65938 163.02677C81.32238 163.02677 81.95833 163.29019 82.42714 163.759 82.89596 164.22782 83.15938 164.86376 83.15938 165.52677 83.15938 166.18978 82.89596 166.82572 82.42714 167.29454 81.95833 167.76335 81.32238 168.02677 80.65938 168.02677 79.99637 168.02677 79.36043 167.76335 78.89161 167.29454 78.42279 166.82572 78.15938 166.18978 78.15938 165.52677 78.15938 164.86376 78.42279 164.22782 78.89161 163.759 79.36043 163.29019 79.99637 163.02677 80.65938 163.02677Z" fill="#87b3cb"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width=".5" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#456e87" d="M80.65938 163.02677C81.32238 163.02677 81.95833 163.29019 82.42714 163.759 82.89596 164.22782 83.15938 164.86376 83.15938 165.52677 83.15938 166.18978 82.89596 166.82572 82.42714 167.29454 81.95833 167.76335 81.32238 168.02677 80.65938 168.02677 79.99637 168.02677 79.36043 167.76335 78.89161 167.29454 78.42279 166.82572 78.15938 166.18978 78.15938 165.52677 78.15938 164.86376 78.42279 164.22782 78.89161 163.759 79.36043 163.29019 79.99637 163.02677 80.65938 163.02677Z"/>
<use data-text="P" xlink:href="#font_2_24" transform="matrix(7.4,0,0,-7.4,87.68938,7.9340059)" fill="#293844"/>
<use data-text="r" xlink:href="#font_2_48" transform="matrix(7.4,0,0,-7.4,91.80378,7.9340059)" fill="#293844"/>
<use data-text="o" xlink:href="#font_2_45" transform="matrix(7.4,0,0,-7.4,94.267978,7.9340059)" fill="#293844"/>
<use data-text="p" xlink:href="#font_2_46" transform="matrix(7.4,0,0,-7.4,97.96798,7.9340059)" fill="#293844"/>
<use data-text="r" xlink:href="#font_2_48" transform="matrix(7.4,0,0,-7.4,101.66798,7.9340059)" fill="#293844"/>
<use data-text="i" xlink:href="#font_2_40" transform="matrix(7.4,0,0,-7.4,104.13218,7.9340059)" fill="#293844"/>
<use data-text="e" xlink:href="#font_2_36" transform="matrix(7.4,0,0,-7.4,106.18938,7.9340059)" fill="#293844"/>
<use data-text="t" xlink:href="#font_2_50" transform="matrix(7.4,0,0,-7.4,109.474979,7.9340059)" fill="#293844"/>
<use data-text="a" xlink:href="#font_2_32" transform="matrix(7.4,0,0,-7.4,111.53218,7.9340059)" fill="#293844"/>
<use data-text="r" xlink:href="#font_2_48" transform="matrix(7.4,0,0,-7.4,114.81778,7.9340059)" fill="#293844"/>
<use data-text="y" xlink:href="#font_2_54" transform="matrix(7.4,0,0,-7.4,117.28198,7.9340059)" fill="#293844"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" d="M132.34688 162.67678C133.1027 162.67678 133.82769 162.97707 134.36212 163.51152 134.89658 164.04596 135.19687 164.77094 135.19687 165.52677 135.19687 166.2826 134.89658 167.00757 134.36212 167.54203 133.82769 168.07648 133.1027 168.37677 132.34688 168.37677 131.59105 168.37677 130.86608 168.07648 130.33162 167.54203 129.79717 167.00757 129.49687 166.2826 129.49687 165.52677 129.49687 164.77094 129.79717 164.04596 130.33162 163.51152 130.86608 162.97707 131.59105 162.67678 132.34688 162.67678Z" fill="#ffffff"/>
<path transform="matrix(1,0,0,-1,0,170.87078)" stroke-width="1" stroke-linecap="butt" stroke-linejoin="round" fill="none" stroke="#293844" d="M132.34688 162.67678C133.1027 162.67678 133.82769 162.97707 134.36212 163.51152 134.89658 164.04596 135.19687 164.77094 135.19687 165.52677 135.19687 166.2826 134.89658 167.00757 134.36212 167.54203 133.82769 168.07648 133.1027 168.37677 132.34688 168.37677 131.59105 168.37677 130.86608 168.07648 130.33162 167.54203 129.79717 167.00757 129.49687 166.2826 129.49687 165.52677 129.49687 164.77094 129.79717 164.04596 130.33162 163.51152 130.86608 162.97707 131.59105 162.67678 132.34688 162.67678Z"/>
<use data-text="P" xlink:href="#font_2_24" transform="matrix(7.4,0,0,-7.4,139.37688,7.9340059)" fill="#293844"/>
<use data-text="a" xlink:href="#font_2_32" transform="matrix(7.4,0,0,-7.4,143.49127,7.9340059)" fill="#293844"/>
<use data-text="r" xlink:href="#font_2_48" transform="matrix(7.4,0,0,-7.4,146.77687,7.9340059)" fill="#293844"/>
<use data-text="e" xlink:href="#font_2_36" transform="matrix(7.4,0,0,-7.4,149.24108,7.9340059)" fill="#293844"/>
<use data-text="t" xlink:href="#font_2_50" transform="matrix(7.4,0,0,-7.4,152.52667,7.9340059)" fill="#293844"/>
<use data-text="o" xlink:href="#font_2_45" transform="matrix(7.4,0,0,-7.4,154.58388,7.9340059)" fill="#293844"/>
<use data-text="-" xlink:href="#font_2_4" transform="matrix(7.4,0,0,-7.4,158.28388,7.9340059)" fill="#293844"/>
<use data-text="o" xlink:href="#font_2_45" transform="matrix(7.4,0,0,-7.4,160.74808,7.9340059)" fill="#293844"/>
<use data-text="p" xlink:href="#font_2_46" transform="matrix(7.4,0,0,-7.4,164.44808,7.9340059)" fill="#293844"/>
<use data-text="t" xlink:href="#font_2_50" transform="matrix(7.4,0,0,-7.4,168.14807,7.9340059)" fill="#293844"/>
<use data-text="i" xlink:href="#font_2_40" transform="matrix(7.4,0,0,-7.4,170.20528,7.9340059)" fill="#293844"/>
<use data-text="m" xlink:href="#font_2_43" transform="matrix(7.4,0,0,-7.4,172.26248,7.9340059)" fill="#293844"/>
<use data-text="a" xlink:href="#font_2_32" transform="matrix(7.4,0,0,-7.4,178.01969,7.9340059)" fill="#293844"/>
<use data-text="l" xlink:href="#font_2_42" transform="matrix(7.4,0,0,-7.4,181.30529,7.9340059)" fill="#293844"/>
</svg>

After

Width:  |  Height:  |  Size: 141 KiB

File diff suppressed because one or more lines are too long

After

Width:  |  Height:  |  Size: 170 KiB

File diff suppressed because one or more lines are too long

After

Width:  |  Height:  |  Size: 170 KiB

Before

Width:  |  Height:  |  Size: 14 KiB

After

Width:  |  Height:  |  Size: 14 KiB

Binary file not shown.
Binary file not shown.

Before

Width:  |  Height:  |  Size: 11 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 14 KiB

+18
View File
@@ -0,0 +1,18 @@
dataset,system,n,songbench_global_avg,songbench_musicality,songeval_global_avg,songeval_musicality,audiobox_pq,mulan,allmusiccaps,control_overall,per
WSB,YuE2,192,6.731599404761909,5.907463020833336,4.262452916666666,4.2151218749999995,8.25979095697403,0.5068482221880307,0.40536119728009606,4.681896986601725,0.08443630690279864
WSB,YuE2 (best-of-8),192,6.963229315476184,6.266602604166664,4.296048333333334,4.250610416666664,8.271365446348986,0.5050805467180908,0.398045457637636,4.70087253751977,0.09786539961704843
WSB,ACE-Step 1.5,192,6.011793824404761,5.158814583333332,3.846484166666666,3.805107291666667,8.051833018660545,0.43723272854307044,0.38685302749703016,4.580894501177339,0.07457884998484957
WSB,DiffRhythm 2,192,5.242845982142859,4.477539583333331,3.5244677083333333,3.471093229166668,7.978225273390611,0.37817035724098486,0.32549789230688475,4.087044602340498,0.18413265339015286
WSB,HeartMuLa,192,6.248347544642855,5.496267708333332,4.551859166666667,4.5329015625,8.293305036922296,0.3823466453177389,0.2786190857962841,3.490655976705673,0.10705881594471056
WSB,LeVo 2,192,6.324671800595234,5.4589927083333345,4.023369479166667,3.981881250000001,8.396623315910498,0.3542381529114209,0.2679880544101252,3.9457878565807953,0.26116136186343974
WSB,MiniMax Music 2.6,192,6.3221788690476215,5.443674479166666,4.098453333333334,4.055989583333334,8.171058195332686,0.42510899043797207,0.3669951342102043,4.568809841873251,0.24545647955335212
WSB,MiniMax Music 3,192,6.282988392857145,5.348229166666669,4.099352187500001,4.0430729166666675,8.28245102117459,0.39279851930526394,0.36085424468425725,4.43616720498247,0.06268414232388418
WSB,Mureka 9,192,6.937748809523812,6.048820833333333,4.411131979166667,4.361940104166667,8.022609589000544,0.43942587031051517,0.41017753163275,4.636807564214055,0.11690337887061523
WSB,Muse,192,6.034903943452382,5.169220312499998,3.788667708333335,3.736199999999999,8.051745663086573,0.3936510290513979,0.3465821466803997,4.403758449771435,0.33418648580693827
WSB,SongBloom,192,4.235043750000001,3.4492635416666655,3.2051130208333345,3.204842187499999,8.153916046023369,0.2697471845106823,0.19264957021005102,3.0286910546158463,0.19193351857701904
WSB,Suno v4.5,192,6.699528199404766,5.831729166666669,4.366563958333332,4.319757291666664,8.254062737027803,0.5021767822715143,0.38734816436772235,4.41490242329515,0.05798428222369787
WSB,Suno v5,192,6.872074330357141,5.9918354166666665,4.357936874999999,4.305118229166664,8.169838989774385,0.5428044088184834,0.43525151138116297,4.5907186541713685,0.080993611026942
WSB,Suno v5.5,192,6.714983407738093,5.808658854166672,4.2151607291666675,4.149727604166669,8.195489337046942,0.5088621161606474,0.3917035845806822,4.591446755445953,0.05958313723720433
WSB,Suno v6,192,6.556171130952383,5.6557864583333375,4.308628645833337,4.263533854166667,8.129587392012278,0.4916386870124067,0.4304573973834825,4.625818883206533,0.07580377922906821
WSB,Suno v6 Wild,192,6.419507440476191,5.564441145833332,4.219912500000002,4.171640104166664,8.178526304662228,0.49989257991546765,0.43164352465343353,4.589767688464477,0.07454661500473854
WSB,YuE 1,192,4.916463244047622,4.084726562499998,3.214967395833334,3.1524088541666675,7.868339642882347,0.262323199029197,0.2881557388876293,3.7300764989993715,0.3637846445448743
1 dataset system n songbench_global_avg songbench_musicality songeval_global_avg songeval_musicality audiobox_pq mulan allmusiccaps control_overall per
2 WSB YuE2 192 6.731599404761909 5.907463020833336 4.262452916666666 4.2151218749999995 8.25979095697403 0.5068482221880307 0.40536119728009606 4.681896986601725 0.08443630690279864
3 WSB YuE2 (best-of-8) 192 6.963229315476184 6.266602604166664 4.296048333333334 4.250610416666664 8.271365446348986 0.5050805467180908 0.398045457637636 4.70087253751977 0.09786539961704843
4 WSB ACE-Step 1.5 192 6.011793824404761 5.158814583333332 3.846484166666666 3.805107291666667 8.051833018660545 0.43723272854307044 0.38685302749703016 4.580894501177339 0.07457884998484957
5 WSB DiffRhythm 2 192 5.242845982142859 4.477539583333331 3.5244677083333333 3.471093229166668 7.978225273390611 0.37817035724098486 0.32549789230688475 4.087044602340498 0.18413265339015286
6 WSB HeartMuLa 192 6.248347544642855 5.496267708333332 4.551859166666667 4.5329015625 8.293305036922296 0.3823466453177389 0.2786190857962841 3.490655976705673 0.10705881594471056
7 WSB LeVo 2 192 6.324671800595234 5.4589927083333345 4.023369479166667 3.981881250000001 8.396623315910498 0.3542381529114209 0.2679880544101252 3.9457878565807953 0.26116136186343974
8 WSB MiniMax Music 2.6 192 6.3221788690476215 5.443674479166666 4.098453333333334 4.055989583333334 8.171058195332686 0.42510899043797207 0.3669951342102043 4.568809841873251 0.24545647955335212
9 WSB MiniMax Music 3 192 6.282988392857145 5.348229166666669 4.099352187500001 4.0430729166666675 8.28245102117459 0.39279851930526394 0.36085424468425725 4.43616720498247 0.06268414232388418
10 WSB Mureka 9 192 6.937748809523812 6.048820833333333 4.411131979166667 4.361940104166667 8.022609589000544 0.43942587031051517 0.41017753163275 4.636807564214055 0.11690337887061523
11 WSB Muse 192 6.034903943452382 5.169220312499998 3.788667708333335 3.736199999999999 8.051745663086573 0.3936510290513979 0.3465821466803997 4.403758449771435 0.33418648580693827
12 WSB SongBloom 192 4.235043750000001 3.4492635416666655 3.2051130208333345 3.204842187499999 8.153916046023369 0.2697471845106823 0.19264957021005102 3.0286910546158463 0.19193351857701904
13 WSB Suno v4.5 192 6.699528199404766 5.831729166666669 4.366563958333332 4.319757291666664 8.254062737027803 0.5021767822715143 0.38734816436772235 4.41490242329515 0.05798428222369787
14 WSB Suno v5 192 6.872074330357141 5.9918354166666665 4.357936874999999 4.305118229166664 8.169838989774385 0.5428044088184834 0.43525151138116297 4.5907186541713685 0.080993611026942
15 WSB Suno v5.5 192 6.714983407738093 5.808658854166672 4.2151607291666675 4.149727604166669 8.195489337046942 0.5088621161606474 0.3917035845806822 4.591446755445953 0.05958313723720433
16 WSB Suno v6 192 6.556171130952383 5.6557864583333375 4.308628645833337 4.263533854166667 8.129587392012278 0.4916386870124067 0.4304573973834825 4.625818883206533 0.07580377922906821
17 WSB Suno v6 Wild 192 6.419507440476191 5.564441145833332 4.219912500000002 4.171640104166664 8.178526304662228 0.49989257991546765 0.43164352465343353 4.589767688464477 0.07454661500473854
18 WSB YuE 1 192 4.916463244047622 4.084726562499998 3.214967395833334 3.1524088541666675 7.868339642882347 0.262323199029197 0.2881557388876293 3.7300764989993715 0.3637846445448743
+316
View File
@@ -0,0 +1,316 @@
{
"snapshot": "2026-09-12",
"scope": "15 comparator systems and 2 YuE2 settings on WildSongBench; recorded generation and selection protocols, not matched compute.",
"datasets": [
{
"id": "WSB",
"label": "WildSongBench",
"prompts": 192
}
],
"metrics": [
{
"key": "songbench_global_avg",
"label": "SongBench",
"description": "Seven-dimension global average",
"maximum": 10,
"lowerIsBetter": false
},
{
"key": "songbench_musicality",
"label": "SongBench Musicality",
"description": "SongBench musicality dimension",
"maximum": 10,
"lowerIsBetter": false
},
{
"key": "songeval_global_avg",
"label": "SongEval",
"description": "Five-dimension global average",
"maximum": 5,
"lowerIsBetter": false
},
{
"key": "songeval_musicality",
"label": "SongEval Musicality",
"description": "SongEval musicality dimension",
"maximum": 5,
"lowerIsBetter": false
},
{
"key": "audiobox_pq",
"label": "AudioBox PQ",
"description": "Production quality",
"maximum": 10,
"lowerIsBetter": false
},
{
"key": "mulan",
"label": "MuLan",
"description": "Style–audio similarity",
"maximum": 1,
"lowerIsBetter": false
},
{
"key": "allmusiccaps",
"label": "AllMusicCaps",
"description": "Style–audio similarity",
"maximum": 1,
"lowerIsBetter": false
},
{
"key": "control_overall",
"label": "Q3O Control",
"description": "Prompt-weighted overall control",
"maximum": 5,
"lowerIsBetter": false
},
{
"key": "per",
"label": "PER",
"description": "Phoneme error rate",
"maximum": 1,
"lowerIsBetter": true
}
],
"rows": [
{
"dataset": "WSB",
"system": "YuE2",
"n": 192,
"songbench_global_avg": 6.731599404761909,
"songbench_musicality": 5.907463020833336,
"songeval_global_avg": 4.262452916666666,
"songeval_musicality": 4.2151218749999995,
"audiobox_pq": 8.25979095697403,
"mulan": 0.5068482221880307,
"allmusiccaps": 0.40536119728009606,
"control_overall": 4.681896986601725,
"per": 0.08443630690279864
},
{
"dataset": "WSB",
"system": "YuE2 (best-of-8)",
"n": 192,
"songbench_global_avg": 6.963229315476184,
"songbench_musicality": 6.266602604166664,
"songeval_global_avg": 4.296048333333334,
"songeval_musicality": 4.250610416666664,
"audiobox_pq": 8.271365446348986,
"mulan": 0.5050805467180908,
"allmusiccaps": 0.398045457637636,
"control_overall": 4.70087253751977,
"per": 0.09786539961704843
},
{
"dataset": "WSB",
"system": "ACE-Step 1.5",
"n": 192,
"songbench_global_avg": 6.011793824404761,
"songbench_musicality": 5.158814583333332,
"songeval_global_avg": 3.846484166666666,
"songeval_musicality": 3.805107291666667,
"audiobox_pq": 8.051833018660545,
"mulan": 0.43723272854307044,
"allmusiccaps": 0.38685302749703016,
"control_overall": 4.580894501177339,
"per": 0.07457884998484957
},
{
"dataset": "WSB",
"system": "DiffRhythm 2",
"n": 192,
"songbench_global_avg": 5.242845982142859,
"songbench_musicality": 4.477539583333331,
"songeval_global_avg": 3.5244677083333333,
"songeval_musicality": 3.471093229166668,
"audiobox_pq": 7.978225273390611,
"mulan": 0.37817035724098486,
"allmusiccaps": 0.32549789230688475,
"control_overall": 4.087044602340498,
"per": 0.18413265339015286
},
{
"dataset": "WSB",
"system": "HeartMuLa",
"n": 192,
"songbench_global_avg": 6.248347544642855,
"songbench_musicality": 5.496267708333332,
"songeval_global_avg": 4.551859166666667,
"songeval_musicality": 4.5329015625,
"audiobox_pq": 8.293305036922296,
"mulan": 0.3823466453177389,
"allmusiccaps": 0.2786190857962841,
"control_overall": 3.490655976705673,
"per": 0.10705881594471056
},
{
"dataset": "WSB",
"system": "LeVo 2",
"n": 192,
"songbench_global_avg": 6.324671800595234,
"songbench_musicality": 5.4589927083333345,
"songeval_global_avg": 4.023369479166667,
"songeval_musicality": 3.981881250000001,
"audiobox_pq": 8.396623315910498,
"mulan": 0.3542381529114209,
"allmusiccaps": 0.2679880544101252,
"control_overall": 3.9457878565807953,
"per": 0.26116136186343974
},
{
"dataset": "WSB",
"system": "MiniMax Music 2.6",
"n": 192,
"songbench_global_avg": 6.3221788690476215,
"songbench_musicality": 5.443674479166666,
"songeval_global_avg": 4.098453333333334,
"songeval_musicality": 4.055989583333334,
"audiobox_pq": 8.171058195332686,
"mulan": 0.42510899043797207,
"allmusiccaps": 0.3669951342102043,
"control_overall": 4.568809841873251,
"per": 0.24545647955335212
},
{
"dataset": "WSB",
"system": "MiniMax Music 3",
"n": 192,
"songbench_global_avg": 6.282988392857145,
"songbench_musicality": 5.348229166666669,
"songeval_global_avg": 4.099352187500001,
"songeval_musicality": 4.0430729166666675,
"audiobox_pq": 8.28245102117459,
"mulan": 0.39279851930526394,
"allmusiccaps": 0.36085424468425725,
"control_overall": 4.43616720498247,
"per": 0.06268414232388418
},
{
"dataset": "WSB",
"system": "Mureka 9",
"n": 192,
"songbench_global_avg": 6.937748809523812,
"songbench_musicality": 6.048820833333333,
"songeval_global_avg": 4.411131979166667,
"songeval_musicality": 4.361940104166667,
"audiobox_pq": 8.022609589000544,
"mulan": 0.43942587031051517,
"allmusiccaps": 0.41017753163275,
"control_overall": 4.636807564214055,
"per": 0.11690337887061523
},
{
"dataset": "WSB",
"system": "Muse",
"n": 192,
"songbench_global_avg": 6.034903943452382,
"songbench_musicality": 5.169220312499998,
"songeval_global_avg": 3.788667708333335,
"songeval_musicality": 3.736199999999999,
"audiobox_pq": 8.051745663086573,
"mulan": 0.3936510290513979,
"allmusiccaps": 0.3465821466803997,
"control_overall": 4.403758449771435,
"per": 0.33418648580693827
},
{
"dataset": "WSB",
"system": "SongBloom",
"n": 192,
"songbench_global_avg": 4.235043750000001,
"songbench_musicality": 3.4492635416666655,
"songeval_global_avg": 3.2051130208333345,
"songeval_musicality": 3.204842187499999,
"audiobox_pq": 8.153916046023369,
"mulan": 0.2697471845106823,
"allmusiccaps": 0.19264957021005102,
"control_overall": 3.0286910546158463,
"per": 0.19193351857701904
},
{
"dataset": "WSB",
"system": "Suno v4.5",
"n": 192,
"songbench_global_avg": 6.699528199404766,
"songbench_musicality": 5.831729166666669,
"songeval_global_avg": 4.366563958333332,
"songeval_musicality": 4.319757291666664,
"audiobox_pq": 8.254062737027803,
"mulan": 0.5021767822715143,
"allmusiccaps": 0.38734816436772235,
"control_overall": 4.41490242329515,
"per": 0.05798428222369787
},
{
"dataset": "WSB",
"system": "Suno v5",
"n": 192,
"songbench_global_avg": 6.872074330357141,
"songbench_musicality": 5.9918354166666665,
"songeval_global_avg": 4.357936874999999,
"songeval_musicality": 4.305118229166664,
"audiobox_pq": 8.169838989774385,
"mulan": 0.5428044088184834,
"allmusiccaps": 0.43525151138116297,
"control_overall": 4.5907186541713685,
"per": 0.080993611026942
},
{
"dataset": "WSB",
"system": "Suno v5.5",
"n": 192,
"songbench_global_avg": 6.714983407738093,
"songbench_musicality": 5.808658854166672,
"songeval_global_avg": 4.2151607291666675,
"songeval_musicality": 4.149727604166669,
"audiobox_pq": 8.195489337046942,
"mulan": 0.5088621161606474,
"allmusiccaps": 0.3917035845806822,
"control_overall": 4.591446755445953,
"per": 0.05958313723720433
},
{
"dataset": "WSB",
"system": "Suno v6",
"n": 192,
"songbench_global_avg": 6.556171130952383,
"songbench_musicality": 5.6557864583333375,
"songeval_global_avg": 4.308628645833337,
"songeval_musicality": 4.263533854166667,
"audiobox_pq": 8.129587392012278,
"mulan": 0.4916386870124067,
"allmusiccaps": 0.4304573973834825,
"control_overall": 4.625818883206533,
"per": 0.07580377922906821
},
{
"dataset": "WSB",
"system": "Suno v6 Wild",
"n": 192,
"songbench_global_avg": 6.419507440476191,
"songbench_musicality": 5.564441145833332,
"songeval_global_avg": 4.219912500000002,
"songeval_musicality": 4.171640104166664,
"audiobox_pq": 8.178526304662228,
"mulan": 0.49989257991546765,
"allmusiccaps": 0.43164352465343353,
"control_overall": 4.589767688464477,
"per": 0.07454661500473854
},
{
"dataset": "WSB",
"system": "YuE 1",
"n": 192,
"songbench_global_avg": 4.916463244047622,
"songbench_musicality": 4.084726562499998,
"songeval_global_avg": 3.214967395833334,
"songeval_musicality": 3.1524088541666675,
"audiobox_pq": 7.868339642882347,
"mulan": 0.262323199029197,
"allmusiccaps": 0.2881557388876293,
"control_overall": 3.7300764989993715,
"per": 0.3637846445448743
}
]
}
+63
View File
@@ -0,0 +1,63 @@
# Benchmark results and evaluation scope
The reported results describe generation with symbolic planning under the recorded automatic-evaluation protocols. They support **frontier quality: YuE2 is competitive with Suno v5/v6**. They do not establish universal metric superiority or human preference.
The [interactive demo results](https://map-yue2.github.io/#model-overview), [full WSB results CSV](benchmark-results.csv), and [WildSongBench dataset](https://huggingface.co/datasets/m-a-p/WildSongBench) provide the public result and benchmark resources.
## WildSongBench
The September 12, 2026 comparison covers **192 prompts and 17 settings**: eight public comparison systems, seven proprietary systems, YuE2, and YuE2 (best-of-8). All displayed aggregate metrics cover the same 192 selected outputs per setting.
| Setting | SongBench Avg ↑ | SB Musicality ↑ | SongEval Avg ↑ | SE Musicality ↑ | AudioBox PQ ↑ | MuLan ↑ | AllMusicCaps ↑ | Q3O ↑ | PER ↓ |
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|
| **YuE2 (best-of-8)** | 6.9632 | 6.2666 | 4.2960 | 4.2506 | 8.2714 | 0.5051 | 0.3980 | 4.7009 | 9.79% |
| Mureka 9 | 6.9377 | 6.0488 | 4.4111 | 4.3619 | 8.0226 | 0.4394 | 0.4102 | 4.6368 | 11.69% |
| Suno v5 | 6.8721 | 5.9918 | 4.3579 | 4.3051 | 8.1698 | 0.5428 | 0.4353 | 4.5907 | 8.10% |
| **YuE2** | 6.7316 | 5.9075 | 4.2625 | 4.2151 | 8.2598 | 0.5068 | 0.4054 | 4.6819 | 8.44% |
| Suno v5.5 | 6.7150 | 5.8087 | 4.2152 | 4.1497 | 8.1955 | 0.5089 | 0.3917 | 4.5914 | 5.96% |
| Suno v4.5 | 6.6995 | 5.8317 | 4.3666 | 4.3198 | 8.2541 | 0.5022 | 0.3873 | 4.4149 | 5.80% |
| Suno v6 | 6.5562 | 5.6558 | 4.3086 | 4.2635 | 8.1296 | 0.4916 | 0.4305 | 4.6258 | 7.58% |
| Suno v6 Wild | 6.4195 | 5.5644 | 4.2199 | 4.1716 | 8.1785 | 0.4999 | 0.4316 | 4.5898 | 7.45% |
| LeVo 2 | 6.3247 | 5.4590 | 4.0234 | 3.9819 | 8.3966 | 0.3542 | 0.2680 | 3.9458 | 26.12% |
| MiniMax Music 2.6 | 6.3222 | 5.4437 | 4.0985 | 4.0560 | 8.1711 | 0.4251 | 0.3670 | 4.5688 | 24.55% |
| MiniMax Music 3 | 6.2830 | 5.3482 | 4.0994 | 4.0431 | 8.2825 | 0.3928 | 0.3609 | 4.4362 | 6.27% |
| HeartMuLa | 6.2483 | 5.4963 | 4.5519 | 4.5329 | 8.2933 | 0.3823 | 0.2786 | 3.4907 | 10.71% |
| Muse | 6.0349 | 5.1692 | 3.7887 | 3.7362 | 8.0517 | 0.3937 | 0.3466 | 4.4038 | 33.42% |
| ACE-Step 1.5 | 6.0118 | 5.1588 | 3.8465 | 3.8051 | 8.0518 | 0.4372 | 0.3869 | 4.5809 | 7.46% |
| DiffRhythm 2 | 5.2428 | 4.4775 | 3.5245 | 3.4711 | 7.9782 | 0.3782 | 0.3255 | 4.0870 | 18.41% |
| YuE 1 | 4.9165 | 4.0847 | 3.2150 | 3.1524 | 7.8683 | 0.2623 | 0.2882 | 3.7301 | 36.38% |
| SongBloom | 4.2350 | 3.4493 | 3.2051 | 3.2048 | 8.1539 | 0.2697 | 0.1926 | 3.0287 | 19.19% |
SongBench Avg is the mean of seven dimensions. SongEval Avg is the mean of five dimensions from a separate automatic quality evaluator. SB Musicality and SE Musicality report the respective musicality dimensions. AudioBox PQ measures production quality; MuLan and AllMusicCaps assess text/audio alignment; prompt control uses Q3O on a 0–5 scale. PER is phoneme error rate, for which lower is better.
**Candidate selection.** Standard YuE2 selects the lower-PER candidate from two generations. Best-of-8 selects by SongBench Musicality, then prompt control, then PER. Each candidate's PER uses the lowest PER from four ASR passes. This selection uses a SongBench dimension and is not equivalent to one unselected pipeline call. Prompt-control weights differ on 10 of 192 prompts between the two selected YuE2 settings.
Public baselines, Suno v6, and Suno v6 Wild use two candidates and four ASR passes per candidate, followed by lower-PER selection; earlier proprietary systems retain their delivered-candidate protocols. MiniMax Music 3 uses its official caption rewriter and is classified as a public model because its weights are released and the evaluated campaign uses them locally. SongBloom uses a fixed audio prompt. These are documented system comparisons, not matched-compute experiments.
Both YuE2 settings use melody-and-chord planning and the verified evaluation decoder distributed as **[YuE2-Vae-legacy](https://huggingface.co/m-a-p/YuE2-Vae-legacy)**. Default listening uses **[YuE2-Vae](https://huggingface.co/m-a-p/YuE2-Vae)**. Keep their audio separate; use the same cached acoustic latents for decoder comparisons. The release names determine these roles, not the everyday meaning of “legacy.”
**What the lead means.** YuE2 (best-of-8) has the highest observed SongBench Avg, 6.9632; the highest proprietary mean is Mureka 9 at 6.9377, while Suno v5 scores 6.8721, Suno v6 scores 6.5562, and Suno v6 Wild scores 6.4195. The small gaps are descriptive, not claims of statistical significance. Different systems lead other metrics: Suno v6 has higher SongEval scores than both YuE2 settings, and both v6 variants have lower PER and higher AllMusicCaps scores. The unqualified YuE2 setting scores 6.7316 and is already within the proprietary quality range.
The overview figure combines SongBench and SongEval into a normalized song-quality index and MuLan, AllMusicCaps, and prompt control into a normalized text-alignment index. These axes are comparison indices, not percentages of correct output. Bubble area indicates AudioBox PQ; outlined points are Pareto-optimal on the two plotted axes. The 15 external baselines define the z-score reference; quality weights SongBench Avg and SongEval Avg at 2:1, while alignment weights its three metrics equally. Each axis maps the observed extrema across all 17 settings to 10 and 90. [Figure data and normalization](frontier-composites.json) · [Full-precision CSV](benchmark-results.csv) · [JSON](benchmark-results.json).
## Zero-shot cover generation
The cover evaluation uses **948 SHS100K works**, two requested styles and two seeds per work: **3,792 outputs per method**, without candidate selection. All YuE2 conditions use the general song-generation checkpoint and the benchmark decoder. The generator received no original–cover paired supervision or cover-specific fine-tuning; exclusion of the evaluated works from generator training was confirmed by the authors. This claim does not assert training-data exclusion for external analyzers, evaluators, or comparison models.
| Method | CLEWS mAP ↑ | CLEWS Hit@1 ↑ | Discogs-VINet mAP ↑ | MuLan ↑ | SongBench Musicality ↑ |
|---|---:|---:|---:|---:|---:|
| SongEcho | 0.419 | 48.4% | 0.122 | 0.366 | 3.286 |
| ACE-Step 1.5 | 0.024 | 2.4% | 0.006 | 0.166 | 3.689 |
| YuE2 (full score) | 0.647 | 71.3% | 0.288 | 0.382 | 5.104 |
| YuE2 (without chords) | 0.598 | 67.3% | 0.179 | 0.417 | 5.490 |
| YuE2 (without score) | 0.006 | 0.3% | 0.004 | 0.474 | 5.691 |
CLEWS and Discogs-VINet assess preserved work identity against a source-excluded retrieval gallery of 10,545 recordings per query. MuLan assesses alignment with the requested target style; SongBench measures musicality. Every metric displayed here covers all 3,792 outputs per method. Incomplete Q3O results are omitted.
The full score best preserves work identity in this comparison. Relaxing the supplied score improves target-style alignment and quality in the current cover configuration. These are distinct outcomes: a fixed transcription from a source performance can constrain adaptation to a contrasting style. This does not show that generating a fresh symbolic plan from the current prompt reduces quality. Product guidance recommends melody-only covers to leave accompaniment freer to adapt.
## Editing evidence
A separate paired editing study uses ten original works, two seeds, and 380 full-song recordings. Local changed-note melody attainment increases from 0.0083 to 0.9375; changed-duration harmony attainment increases from 0 to 0.8313. Their units differ and should not be combined into one score. Content outside melody/harmony edits remains close to unedited regeneration on the automatic measurements.
This is evidence for selective control through the score, not identical waveform preservation. The measurements cover ten works and were developed on that cohort. The public agentic demo illustrates a multi-step workflow; it is not a separate statistically controlled human-preference study.
+73
View File
@@ -0,0 +1,73 @@
# Cover a recording
A cover starts with a readable composition: transcribe the source recording, review the melody, then ask YuE2 to realize it in a new style. The general YuE2 checkpoint supports this workflow without cover-specific fine-tuning.
## 1. Set up SheetSage2 separately
SheetSage2 and YuE2 use different dependency versions. Keep separate environments and exchange ABC and audio files. Run transcription and generation sequentially so they can share one GPU.
SheetSage2 uses Python 3.10 or 3.11 and requires FFmpeg 6.1 and its shared libraries. Follow the platform setup in its [model card](https://huggingface.co/m-a-p/SheetSage2), then run from the YuE repository root:
```bash
python3.11 -m venv .venv-sheetsage2
.venv-sheetsage2/bin/python -m pip install huggingface-hub==0.36.0
.venv-sheetsage2/bin/huggingface-cli download m-a-p/SheetSage2 \
--local-dir models/SheetSage2
.venv-sheetsage2/bin/python -m pip install \
torch==2.8.0 torchaudio==2.8.0 \
--index-url https://download.pytorch.org/whl/cu126
.venv-sheetsage2/bin/python -m pip install -r models/SheetSage2/requirements.txt
```
Loading SheetSage2 automatically loads the MERT-v2-FullSong encoder selected by its configuration. A separate MERT2 feature-extraction step is unnecessary.
## 2. Transcribe and review the melody
```bash
.venv-sheetsage2/bin/python models/SheetSage2/infer.py source.wav \
--output cover-score --melody-only
```
`cover-score/score.abc` retains vocal and instrumental melodies while omitting chord symbols. Check the command's exit status and transcription warnings, then listen to the source while reviewing notes, meter, and section order. Transcription errors can carry into the cover.
The same operation is available through Transformers:
```python
from transformers import AutoModel
model = AutoModel.from_pretrained(
"models/SheetSage2", trust_remote_code=True,
).eval().to("cuda")
result = model.transcribe(
"source.wav", output_dir="cover-score", melody_only=True,
)
if not result.get("abc") or result.get("abc_error"):
raise RuntimeError("Transcription did not produce a usable melody score")
print(result.get("warnings", []))
```
For repeatable runs, select reviewed model revisions and retain the downloaded configuration. The model's Python implementation is executed by `trust_remote_code=True`.
## 3. Supply lyrics and a target style
Prepare `cover-request.json` with `style`, `lyrics`, `cot`, and `seed`, following [the original example](../examples/song.json). Set `cot` to `melody`. Use source lyrics you have available, or transcribe and correct the words; align section tags and lyric order with the score. When translating lyrics, match phrasing and syllable counts to the melody.
Switch back to the YuE2 environment:
```bash
.venv/bin/python examples/generate.py --request cover-request.json \
--abc-file cover-score/score.abc --cot melody --output outputs/cover
```
For a complete score with fixed harmony, use `cot="full"`. Melody-only is recommended for changing styles because it gives the accompaniment more freedom. The [benchmark comparison](benchmarks.md#zero-shot-cover-generation) reports both identity retention and target-style/quality trade-offs.
## Try the included original melody
This example requires only YuE2:
```bash
.venv/bin/python examples/generate.py --request examples/song.json \
--abc-file examples/melody.abc --cot melody --output outputs/original-melody
```
It tests the score-conditioned interface with original material; it is not a transcription or a benchmark result. The [public cover demos](https://map-yue2.github.io/#cover) illustrate the complete workflow.
+57
View File
@@ -0,0 +1,57 @@
# Edit with a score and an agent
YuE2 makes the intended composition available as ABC notation. An agent can change specific notes, chord symbols, tempo, or section order, then use the same generator to render the revised composition. This is the editable interface meant by **white-box music generation**.
## A complete harmony-edit example
The original fixtures keep the exact melody, lyric order, meter, tempo, style, and seed fixed. `score-jazz.abc` changes the chord symbols in `score.abc` to seventh chords.
```bash
python skills/yue2-music/scripts/abc_tools.py inspect examples/score.abc
python skills/yue2-music/scripts/abc_tools.py inspect examples/score-jazz.abc
python skills/yue2-music/scripts/abc_tools.py compare \
examples/score.abc examples/score-jazz.abc
python examples/generate.py --abc-file examples/score.abc \
--cot full --output outputs/harmony-before
python examples/generate.py --abc-file examples/score-jazz.abc \
--cot full --output outputs/harmony-after
```
The symbolic comparison should report `match: true`: notes, timing, meter, and tempo are unchanged. That check deliberately allows different harmony. Listen to both outputs to judge whether the revised harmony is realized and whether you prefer it.
## Start from a generated composition
Generate a song with `save_artifacts()` to retain both its audio and score, or export only the plan:
```bash
python skills/yue2-music/scripts/run_yue2.py plan \
--request examples/song.json --output outputs/plan
cp outputs/plan/score.abc edited.abc
```
Give the agent the ABC, lyrics, style, and a bounded brief. For example:
> Reharmonize this song with jazz-influenced chords. Preserve every vocal and instrumental melody note, note duration, bar boundary, section order, and tempo. Keep the lyrics and style unchanged for this comparison. Save a new ABC file and explain the chord changes.
Then inspect the edited score and verify the requested invariants:
```bash
python skills/yue2-music/scripts/abc_tools.py inspect edited.abc
python skills/yue2-music/scripts/abc_tools.py compare outputs/plan/score.abc edited.abc
python examples/generate.py --request examples/song.json \
--abc-file edited.abc --cot full --output outputs/edited
```
The helper checks a limited native ABC dialect. A rejected score may need notation conversion or correction; do not treat an unsupported construct as a demonstrated musical error. See the [ABC reference](../skills/yue2-music/references/abc-editing.md).
## Edit lyrics, style, tempo, or form
Copy the request JSON and update only the intended fields. Align revised lyrics with the melody's phrasing and syllable counts. For section deletion, duplication, or reordering, update both the score and corresponding lyric sections. To change tempo while preserving notes, use the comparison helper's `--allow-tempo-change` option; listen to verify the actual response.
Editing renders a new complete recording. Matching the unchanged score does not guarantee identical singing, timbre, or waveform outside the edit. Preserve the original and use fresh output directories for every version. Check truncation and compare complete outputs rather than choosing examples solely by the desired result.
## Use the packaged skill
Load [yue2-music](../skills/yue2-music/SKILL.md) in an agent supporting `SKILL.md` packages. Its helpers cover generation, transcription, ABC checks, cached decoding, and listening comparisons. The skill guides the agent between existing APIs; no separate agentic-generation checkpoint is required.
[Explore the 9-step, 14-version editing demo](https://map-yue2.github.io/#agentic-music-editing), where each version includes its audio, score, prompts, lyrics, and conversation.
+639
View File
@@ -0,0 +1,639 @@
{
"dataset": "WSB",
"n_prompts": 192,
"source_sha256": "579ffb6b9b069f6d3cdd32a76788998ec633a12426a0ebb7c9f83e6fb0d5b1fc",
"reference_systems": [
"Suno v5",
"Suno v4.5",
"Suno v5.5",
"Suno v6",
"Suno v6 Wild",
"MiniMax Music 2.6",
"Mureka 9",
"YuE 1",
"SongBloom",
"LeVo 2",
"ACE-Step 1.5",
"HeartMuLa",
"DiffRhythm 2",
"Muse",
"MiniMax Music 3"
],
"normalization": "z = (system mean - reference mean) / reference population std",
"reference_statistics": {
"songbench_global_avg": {
"mean": 6.121283377976191,
"population_std": 0.7357457614725659
},
"songeval_global_avg": {
"mean": 4.015471256944445,
"population_std": 0.40701956006795303
},
"songbench_musicality": {
"mean": 5.265866701388889,
"population_std": 0.7053243399537521
},
"songeval_musicality": {
"mean": 3.9703476041666663,
"population_std": 0.4051162346821886
},
"mulan": {
"mean": 0.4186944833891175,
"population_std": 0.08094623772340859
},
"allmusiccaps": {
"mean": 0.3533851072454733,
"population_std": 0.06820180734302435
},
"control_overall": {
"mean": 4.248089863722948,
"population_std": 0.4756164261685128
}
},
"quality_weights": {
"songbench": 0.6666666666666666,
"songeval": 0.33333333333333337
},
"alignment_weights": {
"mulan": 0.3333333333333333,
"allmusiccaps": 0.3333333333333333,
"control_overall": 0.3333333333333333
},
"values": {
"YuE2": {
"quality_global_avg": 0.7552819777715396,
"quality_musicality": 0.8078339775660601,
"alignment": 0.9210758825216073,
"audiobox_pq": 8.25979095697403
},
"YuE2 (best-of-8)": {
"quality_global_avg": 0.9926775311951548,
"quality_musicality": 1.1764900036325572,
"alignment": 0.8913402282808791,
"audiobox_pq": 8.271365446348986
},
"ACE-Step 1.5": {
"quality_global_avg": -0.2376035047550664,
"quality_musicality": -0.23714600301672967,
"alignment": 0.47315715541685155,
"audiobox_pq": 8.051833018660545
},
"DiffRhythm 2": {
"quality_global_avg": -1.1980739812149017,
"quality_musicality": -1.1559112521857333,
"alignment": -0.41604199452568785,
"audiobox_pq": 7.978225273390611
},
"HeartMuLa": {
"quality_global_avg": 0.5544151508643054,
"quality_musicality": 0.6806476331364909,
"alignment": -1.0459382238546173,
"audiobox_pq": 8.293305036922296
},
"LeVo 2": {
"quality_global_avg": 0.19076064388585245,
"quality_musicality": 0.19203107542441739,
"alignment": -0.8946697129809009,
"audiobox_pq": 8.396623315910498
},
"MiniMax Music 2.6": {
"quality_global_avg": 0.24999255629041012,
"quality_musicality": 0.2385294260172236,
"alignment": 0.317708040770442,
"audiobox_pq": 8.171058195332686
},
"MiniMax Music 3": {
"quality_global_avg": 0.21521779684943682,
"quality_musicality": 0.13768736380561403,
"alignment": 0.06167958802908772,
"audiobox_pq": 8.28245102117459
},
"Mureka 9": {
"quality_global_avg": 1.063838458020569,
"quality_musicality": 1.0622475778213865,
"alignment": 0.6353722959962361,
"audiobox_pq": 8.022609589000544
},
"Muse": {
"quality_global_avg": -0.2640126435903063,
"quality_musicality": -0.28400814299371513,
"alignment": -0.027277572405822206,
"audiobox_pq": 8.051745663086573
},
"SongBloom": {
"quality_global_avg": -2.3727929454241776,
"quality_musicality": -2.3469029518895956,
"alignment": -2.25355622224482,
"audiobox_pq": 8.153916046023369
},
"Suno v4.5": {
"quality_global_avg": 0.8114848653568392,
"quality_musicality": 0.822345947368047,
"alignment": 0.6266794053102317,
"audiobox_pq": 8.254062737027803
},
"Suno v5": {
"quality_global_avg": 0.9607654088358881,
"quality_musicality": 0.9616318815497118,
"alignment": 1.1513277335650094,
"audiobox_pq": 8.169838989774385
},
"Suno v5.5": {
"quality_global_avg": 0.7014955760788759,
"quality_musicality": 0.6606381030854666,
"alignment": 0.7992264544775121,
"audiobox_pq": 8.195489337046942
},
"Suno v6": {
"quality_global_avg": 0.6341407893639396,
"quality_musicality": 0.6097852138267854,
"alignment": 0.9417981546473502,
"audiobox_pq": 8.129587392012278
},
"Suno v6 Wild": {
"quality_global_avg": 0.43765333454931554,
"quality_musicality": 0.44783537273294216,
"alignment": 0.9563182017109547,
"audiobox_pq": 8.178526304662228
},
"YuE 1": {
"quality_global_avg": -1.7472815051109758,
"quality_musicality": -1.7894112446823107,
"alignment": -1.3257833039118254,
"audiobox_pq": 7.868339642882347
}
},
"sensitivity": [
{
"variant": "global_avg",
"songbench_weight": 0.5,
"best_public": "HeartMuLa",
"yue2_minus_best_public": -0.02710930368588904,
"rank_yue2": 6,
"rank_bo8": 3,
"ranking": [
"Mureka 9",
"Suno v5",
"YuE2 (best-of-8)",
"Suno v4.5",
"HeartMuLa",
"YuE2",
"Suno v6",
"Suno v5.5",
"Suno v6 Wild",
"MiniMax Music 2.6",
"MiniMax Music 3",
"LeVo 2",
"ACE-Step 1.5",
"Muse",
"DiffRhythm 2",
"YuE 1",
"SongBloom"
]
},
{
"variant": "global_avg",
"songbench_weight": 0.6,
"best_public": "HeartMuLa",
"yue2_minus_best_public": 0.10967637466998481,
"rank_yue2": 5,
"rank_bo8": 2,
"ranking": [
"Mureka 9",
"YuE2 (best-of-8)",
"Suno v5",
"Suno v4.5",
"YuE2",
"Suno v5.5",
"Suno v6",
"HeartMuLa",
"Suno v6 Wild",
"MiniMax Music 2.6",
"MiniMax Music 3",
"LeVo 2",
"ACE-Step 1.5",
"Muse",
"DiffRhythm 2",
"YuE 1",
"SongBloom"
]
},
{
"variant": "global_avg",
"songbench_weight": 0.6666666666666666,
"best_public": "HeartMuLa",
"yue2_minus_best_public": 0.20086682690723423,
"rank_yue2": 5,
"rank_bo8": 2,
"ranking": [
"Mureka 9",
"YuE2 (best-of-8)",
"Suno v5",
"Suno v4.5",
"YuE2",
"Suno v5.5",
"Suno v6",
"HeartMuLa",
"Suno v6 Wild",
"MiniMax Music 2.6",
"MiniMax Music 3",
"LeVo 2",
"ACE-Step 1.5",
"Muse",
"DiffRhythm 2",
"YuE 1",
"SongBloom"
]
},
{
"variant": "global_avg",
"songbench_weight": 0.7,
"best_public": "HeartMuLa",
"yue2_minus_best_public": 0.24646205302585877,
"rank_yue2": 5,
"rank_bo8": 2,
"ranking": [
"Mureka 9",
"YuE2 (best-of-8)",
"Suno v5",
"Suno v4.5",
"YuE2",
"Suno v5.5",
"Suno v6",
"HeartMuLa",
"Suno v6 Wild",
"MiniMax Music 2.6",
"MiniMax Music 3",
"LeVo 2",
"ACE-Step 1.5",
"Muse",
"DiffRhythm 2",
"YuE 1",
"SongBloom"
]
},
{
"variant": "global_avg",
"songbench_weight": 0.75,
"best_public": "HeartMuLa",
"yue2_minus_best_public": 0.3148548922037958,
"rank_yue2": 5,
"rank_bo8": 2,
"ranking": [
"Mureka 9",
"YuE2 (best-of-8)",
"Suno v5",
"Suno v4.5",
"YuE2",
"Suno v5.5",
"Suno v6",
"HeartMuLa",
"Suno v6 Wild",
"MiniMax Music 2.6",
"MiniMax Music 3",
"LeVo 2",
"ACE-Step 1.5",
"Muse",
"DiffRhythm 2",
"YuE 1",
"SongBloom"
]
},
{
"variant": "global_avg",
"songbench_weight": 0.8,
"best_public": "HeartMuLa",
"yue2_minus_best_public": 0.3832477313817329,
"rank_yue2": 5,
"rank_bo8": 2,
"ranking": [
"Mureka 9",
"YuE2 (best-of-8)",
"Suno v5",
"Suno v4.5",
"YuE2",
"Suno v5.5",
"Suno v6",
"Suno v6 Wild",
"HeartMuLa",
"MiniMax Music 2.6",
"LeVo 2",
"MiniMax Music 3",
"ACE-Step 1.5",
"Muse",
"DiffRhythm 2",
"YuE 1",
"SongBloom"
]
},
{
"variant": "musicality",
"songbench_weight": 0.5,
"best_public": "HeartMuLa",
"yue2_minus_best_public": -0.10071426090411806,
"rank_yue2": 6,
"rank_bo8": 1,
"ranking": [
"YuE2 (best-of-8)",
"Mureka 9",
"Suno v5",
"HeartMuLa",
"Suno v4.5",
"YuE2",
"Suno v6",
"Suno v5.5",
"Suno v6 Wild",
"MiniMax Music 2.6",
"LeVo 2",
"MiniMax Music 3",
"ACE-Step 1.5",
"Muse",
"DiffRhythm 2",
"YuE 1",
"SongBloom"
]
},
{
"variant": "musicality",
"songbench_weight": 0.6,
"best_public": "HeartMuLa",
"yue2_minus_best_public": 0.036026102296094376,
"rank_yue2": 5,
"rank_bo8": 1,
"ranking": [
"YuE2 (best-of-8)",
"Mureka 9",
"Suno v5",
"Suno v4.5",
"YuE2",
"HeartMuLa",
"Suno v5.5",
"Suno v6",
"Suno v6 Wild",
"MiniMax Music 2.6",
"LeVo 2",
"MiniMax Music 3",
"ACE-Step 1.5",
"Muse",
"DiffRhythm 2",
"YuE 1",
"SongBloom"
]
},
{
"variant": "musicality",
"songbench_weight": 0.6666666666666666,
"best_public": "HeartMuLa",
"yue2_minus_best_public": 0.12718634442956922,
"rank_yue2": 5,
"rank_bo8": 1,
"ranking": [
"YuE2 (best-of-8)",
"Mureka 9",
"Suno v5",
"Suno v4.5",
"YuE2",
"HeartMuLa",
"Suno v5.5",
"Suno v6",
"Suno v6 Wild",
"MiniMax Music 2.6",
"LeVo 2",
"MiniMax Music 3",
"ACE-Step 1.5",
"Muse",
"DiffRhythm 2",
"YuE 1",
"SongBloom"
]
},
{
"variant": "musicality",
"songbench_weight": 0.7,
"best_public": "HeartMuLa",
"yue2_minus_best_public": 0.1727664654963067,
"rank_yue2": 5,
"rank_bo8": 1,
"ranking": [
"YuE2 (best-of-8)",
"Mureka 9",
"Suno v5",
"Suno v4.5",
"YuE2",
"Suno v5.5",
"HeartMuLa",
"Suno v6",
"Suno v6 Wild",
"MiniMax Music 2.6",
"LeVo 2",
"MiniMax Music 3",
"ACE-Step 1.5",
"Muse",
"DiffRhythm 2",
"YuE 1",
"SongBloom"
]
},
{
"variant": "musicality",
"songbench_weight": 0.75,
"best_public": "HeartMuLa",
"yue2_minus_best_public": 0.24113664709641314,
"rank_yue2": 4,
"rank_bo8": 1,
"ranking": [
"YuE2 (best-of-8)",
"Mureka 9",
"Suno v5",
"YuE2",
"Suno v4.5",
"Suno v5.5",
"Suno v6",
"HeartMuLa",
"Suno v6 Wild",
"MiniMax Music 2.6",
"LeVo 2",
"MiniMax Music 3",
"ACE-Step 1.5",
"Muse",
"DiffRhythm 2",
"YuE 1",
"SongBloom"
]
},
{
"variant": "musicality",
"songbench_weight": 0.8,
"best_public": "HeartMuLa",
"yue2_minus_best_public": 0.30950682869651935,
"rank_yue2": 4,
"rank_bo8": 1,
"ranking": [
"YuE2 (best-of-8)",
"Mureka 9",
"Suno v5",
"YuE2",
"Suno v4.5",
"Suno v5.5",
"Suno v6",
"HeartMuLa",
"Suno v6 Wild",
"MiniMax Music 2.6",
"LeVo 2",
"MiniMax Music 3",
"ACE-Step 1.5",
"Muse",
"DiffRhythm 2",
"YuE 1",
"SongBloom"
]
}
],
"display_normalization": {
"formula": "index = 10 + 80 * (composite - min) / (max - min)",
"display_range": [
10,
90
],
"axis_range": [
0,
100
],
"endpoint_systems": [
"YuE2",
"YuE2 (best-of-8)",
"ACE-Step 1.5",
"DiffRhythm 2",
"HeartMuLa",
"LeVo 2",
"MiniMax Music 2.6",
"MiniMax Music 3",
"Mureka 9",
"Muse",
"SongBloom",
"Suno v4.5",
"Suno v5",
"Suno v5.5",
"Suno v6",
"Suno v6 Wild",
"YuE 1"
],
"bounds": {
"quality_global_avg": {
"min": -2.3727929454241776,
"max": 1.063838458020569
},
"quality_musicality": {
"min": -2.3469029518895956,
"max": 1.1764900036325572
},
"alignment": {
"min": -2.25355622224482,
"max": 1.1513277335650094
}
},
"scope": "Observed extrema across the frozen 17-setting WSB comparison."
},
"display_values": {
"YuE2": {
"quality_global_avg": 82.81723422675485,
"quality_musicality": 81.6295223219151,
"alignment": 84.59008050713697
},
"YuE2 (best-of-8)": {
"quality_global_avg": 88.34347258180588,
"quality_musicality": 90.0,
"alignment": 83.89142164823544
},
"ACE-Step 1.5": {
"quality_global_avg": 59.70424092683039,
"quality_musicality": 57.90284763591369,
"alignment": 74.06593383035025
},
"DiffRhythm 2": {
"quality_global_avg": 37.34582389095981,
"quality_musicality": 37.04192724997628,
"alignment": 53.17361182506653
},
"HeartMuLa": {
"quality_global_avg": 78.1413338271742,
"quality_musicality": 78.7417071724812,
"alignment": 38.373783402036175
},
"LeVo 2": {
"quality_global_avg": 69.67596261246808,
"quality_musicality": 67.6474792193652,
"alignment": 41.9279370903721
},
"MiniMax Music 2.6": {
"quality_global_avg": 71.0547991637534,
"quality_musicality": 68.70324225641005,
"alignment": 70.41355409197678
},
"MiniMax Music 3": {
"quality_global_avg": 70.24529112268468,
"quality_musicality": 66.41358422542463,
"alignment": 64.3979962976035
},
"Mureka 9": {
"quality_global_avg": 90.0,
"quality_musicality": 87.40608152985898,
"alignment": 77.87728582201137
},
"Muse": {
"quality_global_avg": 59.08947290000577,
"quality_musicality": 56.83882462017735,
"alignment": 62.30788899081859
},
"SongBloom": {
"quality_global_avg": 10.0,
"quality_musicality": 10.0,
"alignment": 10.0
},
"Suno v4.5": {
"quality_global_avg": 84.12555929249136,
"quality_musicality": 81.95902220989649,
"alignment": 77.67304060722402
},
"Suno v5": {
"quality_global_avg": 87.60060275113905,
"quality_musicality": 85.12156322510432,
"alignment": 90.0
},
"Suno v5.5": {
"quality_global_avg": 81.56516159216855,
"quality_musicality": 78.28738305243859,
"alignment": 81.72714762306776
},
"Suno v6": {
"quality_global_avg": 79.9972358228254,
"quality_musicality": 77.13274853053025,
"alignment": 85.0769639932043
},
"Suno v6 Wild": {
"quality_global_avg": 75.42328111548792,
"quality_musicality": 73.45561474186167,
"alignment": 85.41812210025412
},
"YuE 1": {
"quality_global_avg": 24.561036477434577,
"quality_musicality": 22.65806486519848,
"alignment": 31.798638200280863
}
},
"per_song_validation": {
"systems": 17,
"records": 3264,
"absolute_tolerance": 1e-10
},
"interpretation": [
"Descriptive standardized composites, not percentages or validated new evaluators.",
"Observed extrema map to 10 and 90 on 0–100 axes, leaving symmetric display margins.",
"All 15 external baselines, including HeartMuLa and both Suno v6 variants, define one frozen WSB reference.",
"YuE2 and Bo8 are excluded from the 15-system z-score standardization reference.",
"Q3O retains recorded prompt weights, which differ on some selected records.",
"Evaluator reuse and candidate-selection protocols remain as documented in Tables 1–2."
],
"source": "benchmark-results.csv",
"snapshot": "2026-09-12"
}
+74
View File
@@ -0,0 +1,74 @@
# Generate songs with YuE2
Install from the repository root with Python 3.12 and `python -m pip install .`. The supported starting point is a BF16-capable NVIDIA GPU with 24 GB VRAM, one request at a time. The default output is 48 kHz stereo with full symbolic planning and the listening decoder, [YuE2-Vae](https://huggingface.co/m-a-p/YuE2-Vae).
## Style, lyrics, and planning
Put genre, instruments, vocal character, language, and tempo in `style`. Put the words to sing in `lyrics`, with section tags such as `[Verse]` and `[Chorus]`. Start with the [original request](../examples/song.json).
```python
import json
from pathlib import Path
from yue2 import YuE2Pipeline
request = json.loads(Path("examples/song.json").read_text(encoding="utf-8"))
with YuE2Pipeline.from_pretrained("m-a-p/YuE2-3B", device="cuda") as pipe:
song = pipe(**request)
song.save_artifacts("outputs/song")
print(song.truncated)
```
`full` plans melody and chords, `melody` plans only melody, and `off` generates without a symbolic plan. Supplying `abc` uses that composition directly; it requires `full` or `melody`. The native melody input has `Vocal` and `Ins` voices and no chord symbols. Arbitrary ABC dialects may need conversion; the [skill's ABC reference](../skills/yue2-music/references/abc-editing.md) describes the supported notation.
One pipeline call produces one candidate. The benchmark's candidate selection is a separate evaluation step. `cfg_scale` controls text guidance; defaults are ready to use, while changing sampling or guidance may change quality.
The command-line equivalent is:
```bash
yue2 generate --request examples/song.json --output outputs/song-cli
```
## Save and inspect a plan before synthesis
```python
import json
from pathlib import Path
from yue2 import YuE2Pipeline, SymbolicPlan
request = json.loads(Path("examples/song.json").read_text(encoding="utf-8"))
with YuE2Pipeline.from_pretrained("m-a-p/YuE2-3B", device="cuda") as pipe:
plan = pipe.plan(**request)
plan.save("outputs/plan")
restored = SymbolicPlan.load("outputs/plan")
semantic = pipe.generate_semantic(restored)
latents = pipe.synthesize(semantic)
audio = pipe.decode(latents)
import soundfile as sf
sf.write("outputs/plan/audio.flac", audio, 48000)
```
`SymbolicPlan.load` restores exact token IDs and checks the saved files. Use it for an unchanged plan. To change the composition, copy the ABC, edit the copy, and submit it as a new `abc` input; do not change a saved plan in place. The end-to-end `save_artifacts()` path is the simplest way to retain a complete generation record.
## Outputs and reproducibility
`save_artifacts()` retains `audio.flac`, `score.abc` when applicable, `plan.json`, semantic tokens, `latent.npy`, effective configuration, timings, model identities, and integrity records. Inspect `result.json` and its `truncated` flags. Saved audio can be playable even when the model hit a token limit. Use a fresh directory for each changed request and retain failures in comparisons.
`from_pretrained` accepts local model directories, `revision`, `vae_revision`, `cache_dir`, and `local_files_only=True`. Pin model and VAE revisions for comparisons. Separate GPUs, runtime versions, or sampling settings can change a seeded generation.
## Listening and evaluation decoders
| Decoder | Use |
|---|---|
| [YuE2-Vae](https://huggingface.co/m-a-p/YuE2-Vae) | Default generation and listening |
| [YuE2-Vae-legacy](https://huggingface.co/m-a-p/YuE2-Vae-legacy) | Reproducing the recorded benchmark protocol |
Choose explicitly with `YuE2Pipeline.from_pretrained("m-a-p/YuE2-3B", vae="m-a-p/YuE2-Vae-legacy")`. Keep listening and evaluation audio in separate directories. When comparing decoders, decode the same cached latents instead of generating a new song:
```bash
python skills/yue2-music/scripts/run_yue2.py decode \
--source outputs/song --output outputs/song-benchmark \
--vae m-a-p/YuE2-Vae-legacy
```
The helper verifies the source artifacts and preserves the latent identity. Benchmark claims refer to the stated evaluation model and decoder; a new local generation is not itself a reproduction of the published aggregate.
+18
View File
@@ -0,0 +1,18 @@
# Public documentation sources
The README's frontier figure and architecture figure are reproduced from the [YuE2 demo page](https://map-yue2.github.io/#model-overview). The logo is the original project's logo, preserved from [YuE-v1](https://github.com/multimodal-art-projection/YuE/tree/YuE-v1/assets/logo). PNG text, timestamp, and EXIF metadata were removed where present; image pixels were preserved. These figures and original example inputs are documentation assets, not generated evaluation audio.
The institution panels reuse the eight marks and their order from the [demo page](https://map-yue2.github.io/), with layouts for desktop and mobile. The SVGs contain the original vector artwork and embedded raster artwork; EXIF metadata was removed without recompressing image pixels. Institution names and marks belong to their respective organizations.
The [WSB CSV](benchmark-results.csv) is the September 12, 2026 result set displayed on the demo page. The README presents all 17 settings for four metrics; the CSV retains the complete metric set. Benchmark scope and selection are recorded in [benchmarks.md](benchmarks.md).
## Citation verification
The two README BibTeX entries use arXiv versions consistently. Titles, full ordered author lists, identifiers, and first-submission years were manually checked against the primary records on September 10, 2026:
| Work | Primary record | Selected version metadata |
|---|---|---|
| MERT | [arXiv:2306.00107](https://arxiv.org/abs/2306.00107) | 20 authors; first submitted May 31, 2023; current record v5 |
| YuE | [arXiv:2503.08638](https://arxiv.org/abs/2503.08638) | 58 authors; first submitted March 11, 2025; current record v2 |
MERT was also accepted at ICLR 2024. The README deliberately cites the arXiv version with year 2023, rather than mixing the conference year with preprint metadata. These entries cite the preceding MERT and YuE papers; they are not presented as a published YuE2 paper.
BIN
View File
Binary file not shown.

Before

Width:  |  Height:  |  Size: 424 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 292 KiB

+485
View File
@@ -0,0 +1,485 @@
{
"id": "7c9d78d2-6b53-43f8-9173-4b6748d1af04",
"revision": 0,
"last_node_id": 22,
"last_link_id": 24,
"nodes": [
{
"id": 4,
"type": "SaveAudioAdvanced",
"pos": [
1696.2669811564128,
5221.529241119489
],
"size": [
414.5625,
165.984375
],
"flags": {},
"order": 10,
"mode": 0,
"inputs": [
{
"name": "audio",
"type": "AUDIO",
"link": 24
}
],
"outputs": [
{
"name": "audio",
"type": "AUDIO",
"links": null
}
],
"properties": {},
"widgets_values": [
"audio/yue2",
"flac"
],
"widgets_values_named": {
"filename_prefix": "audio/yue2",
"format": "flac"
}
},
{
"id": 9,
"type": "Note",
"pos": [
2211.75764837062,
4732.53074774176
],
"size": [
391.3125,
1272.1875
],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {},
"widgets_values": [
"[Intro]\n\n[Verse1]\n挡风玻璃 凝结 冰霜视线\n钢铁血管 泵送 液压熔流\n方向盘 是 锻打的 钢铁誓约\n涡轮增压 撕裂 寒夜 的 氧\n导航删除 坐标 植入 芯片\n柏油路 震颤 传送带 的 脉动\n后视镜里 昨日 碾成 铁屑\n这 电流脉冲 即 我的 通行 令\n\n[Pre-Chorus]\n音量阀 旋至 红色 警戒区\n底鼓 如 锻锤 砸击 脊髓 Yeah\n肾上腺素 是 劣质 硝化燃料\n爆\n\n[Chorus]\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 生锈铁板\n让 电极火花 焚烧 这 牢笼 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 铁渣 界限 全崩塌\n这 工业圣歌 是 终极 点火栓\n一路 碾压 向 废土 行军 Yeah\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 生锈铁板\n让 电极火花 焚烧 这 牢笼 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 铁渣 界限 全崩塌\n这 工业圣歌 是 终极 点火栓\n一路 碾压 向 废土 行军 Yeah\n\n[Interlude]\n\n[Verse2]\n钢铁战友 轰鸣 是 唯一 祷文\n转速指针 在 红区 跳着 死亡舞\n风 如 砂轮 打磨 脸庞 至 白骨\n要的 就是 这 齿轮 卡死的 快感\n霓虹 是 熔炉 泄露的 毒光\n像 驾驭 一台 暴动的 锻压机\n排气管 喷吐 硫磺 的 赞美诗\n每一声 回火 都 震荡 灵魂 的 铸铁砧\n\n[Pre-Chorus]\n音波 如 冲压机 重塑 肋骨\n汗水 是 冷却液 腐蚀 沥青\n转速指针 与 熔炉 一同 过载\n再 爆\n\n[Chorus]\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 润滑脂\n让 电弧 鞭笞 这 苍穹 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 蒸汽 界限 是 尘埃\n这 工业圣歌 是 淬火的 撞针\n一路 碾压 向 虚无 进军 Yeah\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 润滑脂\n让 电弧 鞭笞 这 苍穹 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 蒸汽 界限 是 尘埃\n这 工业圣歌 是 淬火的 撞针\n一路 碾压 向 虚无 进军 Yeah\n\n[Bridge]\nUNTERBRECHUNG\nDIESER KANAL GEHÖRT UNSEREM STAHL\nSCHLACKE EINSPEISEN DAMPFDRUCK STEIGT\nWIR SIND DIE MASCHINE WIR SIND DIE EISENBAHN\nVOLLE FAHRT VORWÄRTS ZERSTÖRUNG IST PROZESS\n熔炉警报 参数异常\n\n[Chorus]\nHIER KOMMT DIE KETTE ZERSTÖREN\nHIER KOMMT DER STAHL VERBRENNEN\n油门 焊死 终点 是 熔渣之海\n让 铁锈尘埃 覆盖 这 纪元 Woo\nMASCHINE BRENNT HELL\nMASCHINE STIRBT NICHT METALLGOTT\n这 工业圣歌 是 末日的 黑匣\n一路 碾压 直至 时间 崩解 Yeah\nHIER KOMMT DIE KETTE ZERSTÖREN\nHIER KOMMT DER STAHL VERBRENNEN\n油门 焊死 终点 是 熔渣之海\n让 铁锈尘埃 覆盖 这 纪元 Woo\nMASCHINE BRENNT HELL\nMASCHINE STIRBT NICHT METALLGOTT\n这 工业圣歌 是 末日的 黑匣\n一路 碾压 直至 时间 崩解 Yeah\n\n[Outro]"
],
"widgets_values_named": {
"text": "[Intro]\n\n[Verse1]\n挡风玻璃 凝结 冰霜视线\n钢铁血管 泵送 液压熔流\n方向盘 是 锻打的 钢铁誓约\n涡轮增压 撕裂 寒夜 的 氧\n导航删除 坐标 植入 芯片\n柏油路 震颤 传送带 的 脉动\n后视镜里 昨日 碾成 铁屑\n这 电流脉冲 即 我的 通行 令\n\n[Pre-Chorus]\n音量阀 旋至 红色 警戒区\n底鼓 如 锻锤 砸击 脊髓 Yeah\n肾上腺素 是 劣质 硝化燃料\n爆\n\n[Chorus]\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 生锈铁板\n让 电极火花 焚烧 这 牢笼 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 铁渣 界限 全崩塌\n这 工业圣歌 是 终极 点火栓\n一路 碾压 向 废土 行军 Yeah\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 生锈铁板\n让 电极火花 焚烧 这 牢笼 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 铁渣 界限 全崩塌\n这 工业圣歌 是 终极 点火栓\n一路 碾压 向 废土 行军 Yeah\n\n[Interlude]\n\n[Verse2]\n钢铁战友 轰鸣 是 唯一 祷文\n转速指针 在 红区 跳着 死亡舞\n风 如 砂轮 打磨 脸庞 至 白骨\n要的 就是 这 齿轮 卡死的 快感\n霓虹 是 熔炉 泄露的 毒光\n像 驾驭 一台 暴动的 锻压机\n排气管 喷吐 硫磺 的 赞美诗\n每一声 回火 都 震荡 灵魂 的 铸铁砧\n\n[Pre-Chorus]\n音波 如 冲压机 重塑 肋骨\n汗水 是 冷却液 腐蚀 沥青\n转速指针 与 熔炉 一同 过载\n再 爆\n\n[Chorus]\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 润滑脂\n让 电弧 鞭笞 这 苍穹 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 蒸汽 界限 是 尘埃\n这 工业圣歌 是 淬火的 撞针\n一路 碾压 向 虚无 进军 Yeah\nHIER KOMMT DIE KETTE Brich sie\nHIER KOMMT DER STAHL Fress ihn\n油门 焊死 地平线 是 润滑脂\n让 电弧 鞭笞 这 苍穹 Woo\nMASCHINE BRENNT Lodernd\nMASCHINE STIRBT NICHT Ewig\n规则 成 蒸汽 界限 是 尘埃\n这 工业圣歌 是 淬火的 撞针\n一路 碾压 向 虚无 进军 Yeah\n\n[Bridge]\nUNTERBRECHUNG\nDIESER KANAL GEHÖRT UNSEREM STAHL\nSCHLACKE EINSPEISEN DAMPFDRUCK STEIGT\nWIR SIND DIE MASCHINE WIR SIND DIE EISENBAHN\nVOLLE FAHRT VORWÄRTS ZERSTÖRUNG IST PROZESS\n熔炉警报 参数异常\n\n[Chorus]\nHIER KOMMT DIE KETTE ZERSTÖREN\nHIER KOMMT DER STAHL VERBRENNEN\n油门 焊死 终点 是 熔渣之海\n让 铁锈尘埃 覆盖 这 纪元 Woo\nMASCHINE BRENNT HELL\nMASCHINE STIRBT NICHT METALLGOTT\n这 工业圣歌 是 末日的 黑匣\n一路 碾压 直至 时间 崩解 Yeah\nHIER KOMMT DIE KETTE ZERSTÖREN\nHIER KOMMT DER STAHL VERBRENNEN\n油门 焊死 终点 是 熔渣之海\n让 铁锈尘埃 覆盖 这 纪元 Woo\nMASCHINE BRENNT HELL\nMASCHINE STIRBT NICHT METALLGOTT\n这 工业圣歌 是 末日的 黑匣\n一路 碾压 直至 时间 崩解 Yeah\n\n[Outro]"
},
"color": "#432",
"bgcolor": "#653"
},
{
"id": 11,
"type": "Note",
"pos": [
1757.9660466423838,
4683.239887527698
],
"size": [
406.0625,
106.5625
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {},
"widgets_values": [
"powerful belting, autotune, Electric guitar, symphony, bass, Electric Piano, excited, bold, Heavy Metal"
],
"widgets_values_named": {
"text": "powerful belting, autotune, Electric guitar, symphony, bass, Electric Piano, excited, bold, Heavy Metal"
},
"color": "#432",
"bgcolor": "#653"
},
{
"id": 12,
"type": "YUE_SM_Clip",
"pos": [
728.8559830396886,
4753.938340897657
],
"size": [
308.6875,
114
],
"flags": {},
"order": 2,
"mode": 2,
"inputs": [],
"outputs": [
{
"name": "clip",
"type": "CLIP",
"links": [
13
]
}
],
"properties": {
"Node name for S&R": "YUE_SM_Clip"
},
"widgets_values": [
"sheetsage2.safetensors",
"mert2_model.safetensors"
],
"widgets_values_named": {
"clip": "sheetsage2.safetensors",
"mert2": "mert2_model.safetensors"
}
},
{
"id": 13,
"type": "YUE_SM_Cond",
"pos": [
1103.2156851446443,
4788.3021930569785
],
"size": [
225,
102
],
"flags": {},
"order": 7,
"mode": 2,
"inputs": [
{
"name": "clip",
"type": "CLIP",
"link": 13
},
{
"name": "audio",
"type": "AUDIO",
"link": 14
}
],
"outputs": [
{
"name": "abc_file",
"type": "STRING",
"links": [
15
]
}
],
"properties": {
"Node name for S&R": "YUE_SM_Cond"
}
},
{
"id": 14,
"type": "LoadAudio",
"pos": [
731.6274159175886,
4930.8322269567225
],
"size": [
270,
138
],
"flags": {},
"order": 3,
"mode": 2,
"inputs": [],
"outputs": [
{
"name": "AUDIO",
"type": "AUDIO",
"links": [
14
]
}
],
"properties": {
"Node name for S&R": "LoadAudio"
},
"widgets_values": [
"002.WAV",
null,
null
],
"widgets_values_named": {
"audio": "002.WAV",
"upload": null
}
},
{
"id": 15,
"type": "PreviewAny",
"pos": [
1387.4152739138706,
4792.112042935652
],
"size": [
274.0625,
138
],
"flags": {},
"order": 9,
"mode": 2,
"inputs": [
{
"name": "source",
"type": "*",
"link": 15
}
],
"outputs": [
{
"name": "STRING",
"type": "STRING",
"links": null
}
],
"properties": {
"Node name for S&R": "PreviewAny"
},
"widgets_values": [],
"widgets_values_named": {}
},
{
"id": 16,
"type": "Note",
"pos": [
1760.2972422274647,
4822.118468642673
],
"size": [
409.9375,
275.75
],
"flags": {},
"order": 4,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {},
"widgets_values": [
"##### 三种基本功能 ######\n\n1.只写风格和歌词,cot选 off关闭,abc files留空,最基本的歌曲生成\n\n2.cover 复刻能力,要先用 clip和cond 节点 复刻原歌曲的abc,然后填写你写的新对应的风格和歌词,选择melody格式。注意有abc文件时, 只能选melody 或者full(if use cover mode use melody only)\n\n3.改谱精修,你或者agent来修改一个已有的abc谱子,而谱子,可以是你推理后生成的,也可以是开启plan模式生成的,这个谱子你编辑后,再给模型生成,如果环境变量不变,就可以达到自由的自定义生成你想要的歌曲(专业级),当然如果你没有音乐天赋,直接让大模型的agent帮你去改最好,官方也提供了skill"
],
"widgets_values_named": {
"text": "##### 三种基本功能 ######\n\n1.只写风格和歌词,cot选 off关闭,abc files留空,最基本的歌曲生成\n\n2.cover 复刻能力,要先用 clip和cond 节点 复刻原歌曲的abc,然后填写你写的新对应的风格和歌词,选择melody格式。注意有abc文件时, 只能选melody 或者full(if use cover mode use melody only)\n\n3.改谱精修,你或者agent来修改一个已有的abc谱子,而谱子,可以是你推理后生成的,也可以是开启plan模式生成的,这个谱子你编辑后,再给模型生成,如果环境变量不变,就可以达到自由的自定义生成你想要的歌曲(专业级),当然如果你没有音乐天赋,直接让大模型的agent帮你去改最好,官方也提供了skill"
},
"color": "#432",
"bgcolor": "#653"
},
{
"id": 1,
"type": "YUE_SM_Model",
"pos": [
768.8408106201794,
5182.267051817941
],
"size": [
328.3125,
110
],
"flags": {},
"order": 5,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "model",
"type": "MODEL",
"links": [
22
]
}
],
"properties": {
"Node name for S&R": "YUE_SM_Model"
},
"widgets_values": [
"yue2_model.safetensors"
],
"widgets_values_named": {
"diffusion_models": "yue2_model.safetensors"
}
},
{
"id": 3,
"type": "YUE_SM_Vae",
"pos": [
755.3975836890024,
5443.775479913051
],
"size": [
270,
86
],
"flags": {},
"order": 6,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "vae",
"type": "VAE",
"links": [
23
]
}
],
"properties": {
"Node name for S&R": "YUE_SM_Vae"
},
"widgets_values": [
"yue_vae.safetensors"
],
"widgets_values_named": {
"vae": "yue_vae.safetensors"
}
},
{
"id": 22,
"type": "YUE_SM_Sampler",
"pos": [
1184.2803775744214,
5196.501931925436
],
"size": [
394.296875,
361.984375
],
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 22
},
{
"name": "vae",
"type": "VAE",
"link": 23
}
],
"outputs": [
{
"name": "audio",
"type": "AUDIO",
"links": [
24
]
}
],
"properties": {
"Node name for S&R": "YUE_SM_Sampler"
},
"widgets_values": [
"English, warm piano pop, expressive female voice, acoustic piano, rounded bass and light drums, lyrical memorable melody, unhurried phrasing, 88 BPM",
"[Verse]\nNeon fades along the lane\nFootsteps keep the time of rain\nFold the night and leave it here\nMorning has a sky to clear\n\n[Chorus]\nLet the day come into view\nEvery road begins with you\nHold a little room for light\nWe will sing beyond the night",
"full",
1452144020,
"randomize",
true,
"E:\\ComfyUI313\\ComfyUI\\output\\yue_sm_temp\\score-jazz.abc",
""
],
"widgets_values_named": {
"style": "English, warm piano pop, expressive female voice, acoustic piano, rounded bass and light drums, lyrical memorable melody, unhurried phrasing, 88 BPM",
"lyrics": "[Verse]\nNeon fades along the lane\nFootsteps keep the time of rain\nFold the night and leave it here\nMorning has a sky to clear\n\n[Chorus]\nLet the day come into view\nEvery road begins with you\nHold a little room for light\nWe will sing beyond the night",
"cot": "full",
"seed": 1452144020,
"control_after_generate": "randomize",
"only_plan": true,
"abc_file": "E:\\ComfyUI313\\ComfyUI\\output\\yue_sm_temp\\score-jazz.abc",
"path-button": ""
}
}
],
"links": [
[
13,
12,
0,
13,
0,
"CLIP"
],
[
14,
14,
0,
13,
1,
"AUDIO"
],
[
15,
13,
0,
15,
0,
"STRING"
],
[
22,
1,
0,
22,
0,
"MODEL"
],
[
23,
3,
0,
22,
1,
"VAE"
],
[
24,
22,
0,
4,
0,
"AUDIO"
]
],
"groups": [
{
"id": 1,
"title": "infer / 推理",
"bounding": [
722.6544477597142,
5103.8516245671,
1449.5114652695938,
502.3017131428023
],
"color": "#3f789e",
"flags": {}
},
{
"id": 2,
"title": "get-abc 获取abc",
"bounding": [
718.8559830396886,
4683.938340897657,
948.4655408741819,
394.89388605906515
],
"color": "#3f789e",
"flags": {}
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.6492965217391748,
"offset": [
-9.916996159523507,
-4402.273552215773
]
},
"frontendVersion": "1.52.7"
},
"version": 0.4
}
+32
View File
@@ -0,0 +1,32 @@
# Original examples
`song.json` contains the original **City Lights** lyrics and style request used by the agent skill. `melody.abc` is a short, original eight-bar melody for those words. Each lyric line has seven syllables and each bar has seven note onsets. `score.abc` adds harmony; `score-jazz.abc` changes only the chord symbols. These are runnable inputs, not recorded benchmark or quality results.
From the repository root, after `pip install .`:
```bash
# Create a song and its score from lyrics and style.
python examples/generate.py --output outputs/create
# Realize the included melody with free accompaniment.
python examples/generate.py --abc-file examples/melody.abc \
--cot melody --output outputs/melody
# Compare two harmonizations with the same melody, lyrics, style, and seed.
python examples/generate.py --abc-file examples/score.abc \
--cot full --output outputs/harmony-original
python examples/generate.py --abc-file examples/score-jazz.abc \
--cot full --output outputs/harmony-edited
```
The script preserves all native artifacts, refuses an existing output directory, and exits nonzero if generation reports truncation. The seed is fixed in `song.json`; seeds aid comparison but do not guarantee identical output across devices or software versions. Use `--revision` and `--vae-revision` to pin model versions.
Check the symbolic intervention without a GPU:
```bash
python skills/yue2-music/scripts/abc_tools.py inspect examples/score.abc
python skills/yue2-music/scripts/abc_tools.py compare \
examples/score.abc examples/score-jazz.abc
```
The comparison verifies unchanged note pitches, timing, meter, and tempo. Listening or audio transcription is still needed to assess the realized music.
+41
View File
@@ -0,0 +1,41 @@
#!/usr/bin/env python3
"""Generate from original example lyrics, optionally with a supplied ABC score."""
import argparse
import json
from pathlib import Path
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--request", type=Path, default=Path(__file__).with_name("song.json"))
parser.add_argument("--abc-file", type=Path)
parser.add_argument("--cot", choices=("full", "melody", "off"))
parser.add_argument("--output", type=Path, required=True)
parser.add_argument("--model", default="m-a-p/YuE2-3B")
parser.add_argument("--vae", default="m-a-p/YuE2-Vae")
parser.add_argument("--revision")
parser.add_argument("--vae-revision")
args = parser.parse_args()
if args.output.exists():
parser.error("Choose a fresh output directory to retain each version.")
request = json.loads(args.request.read_text(encoding="utf-8"))
if args.abc_file:
request["abc"] = args.abc_file.read_text(encoding="utf-8")
if args.cot:
request["cot"] = args.cot
if request.get("abc") is not None and request.get("cot", "full") == "off":
parser.error("A supplied score requires full or melody mode.")
from yue2 import YuE2Pipeline
with YuE2Pipeline.from_pretrained(
args.model, vae=args.vae, revision=args.revision,
vae_revision=args.vae_revision, device="cuda",
) as pipe:
song = pipe(**request)
song.save_artifacts(args.output)
print(json.dumps({"audio": str(args.output / "audio.flac"), "truncated": song.truncated}))
return 1 if any(song.truncated.values()) else 0
# if __name__ == "__main__":
# raise SystemExit(main())
+18
View File
@@ -0,0 +1,18 @@
X:1
T:
M:4/4
L:1/16
Q:1/4=88
V: Vocal clef=treble name="Vocal Melody" snm="Vocal"
V: Ins clef=treble name="Ins Melody" snm="Inst."
K:C
% verse
V: Vocal
E2G2A2G2E2D2C4|D2E2G2E2D2C2D4|E2G2A2c2B2A2G4|F2E2D2E2G2E2C4|
V: Ins
Z4|
% chorus
V: Vocal
G2A2c2B2A2G2E4|F2A2G2E2D2E2G4|A2c2B2A2G2E2D4|E2G2A2G2E2D2C4|
V: Ins
Z4|
+18
View File
@@ -0,0 +1,18 @@
X:1
T:
M:4/4
L:1/16
Q:1/4=88
V: Vocal clef=treble name="Vocal Melody" snm="Vocal"
V: Ins clef=treble name="Ins Melody" snm="Inst."
K:C
% verse
V: Vocal
"Cmaj7"E2G2A2G2E2D2C4|"G7"D2E2G2E2D2C2D4|"Am7"E2G2A2c2B2A2G4|"Fmaj7"F2E2D2E2G2E2C4|
V: Ins
Z4|
% chorus
V: Vocal
"Cmaj7"G2A2c2B2A2G2E4|"Fmaj7"F2A2G2E2D2E2G4|"G7"A2c2B2A2G2E2D4|"Cmaj7"E2G2A2G2E2D2C4|
V: Ins
Z4|
+18
View File
@@ -0,0 +1,18 @@
X:1
T:
M:4/4
L:1/16
Q:1/4=88
V: Vocal clef=treble name="Vocal Melody" snm="Vocal"
V: Ins clef=treble name="Ins Melody" snm="Inst."
K:C
% verse
V: Vocal
"C"E2G2A2G2E2D2C4|"G"D2E2G2E2D2C2D4|"Am"E2G2A2c2B2A2G4|"F"F2E2D2E2G2E2C4|
V: Ins
Z4|
% chorus
V: Vocal
"C"G2A2c2B2A2G2E4|"F"F2A2G2E2D2E2G4|"G"A2c2B2A2G2E2D4|"C"E2G2A2G2E2D2C4|
V: Ins
Z4|
+135
View File
@@ -0,0 +1,135 @@
X:1
T:
M:4/4
L:1/16
Q:1/4=89
V: Vocal clef=treble name="Vocal Melody" snm="Vocal"
V: Ins clef=treble name="Ins Melody" snm="Inst."
K:D#m
% intro
V: Vocal
z12"D#m"z4|"D#m"z16|"D#m"z16|"D#m"z16|
V: Ins
Z|D,2D2A2D^^G2DG2^G^^G^GF|D2D2A2D^^G2D^G^^G^G4|D,2D2A2D^^G2DG2^G^^G^GF|
V: Vocal
"D#m"z16|"D#m"z16|"D#m"z16|"D#m"z16|
V: Ins
D2D2A2D^^G2z^G^^G^G4|D,2D2A2D^^G2DG2^G^^G^GF|D2D2A2D^^G2D^G^^G^G4|D2D2A2D^^G2DG2^G^^G^GF|
V: Vocal
"D#m"z16|
V: Ins
D2D2A2D^^Gz8|
% verse
V: Vocal
"D#m"z2AAAGGFG2G2A2D2|"D#m"z2AAAGGFG2G2A2D2|"Bmaj7"z2AAAGGFG2GFA2D2|"C#"FDFDFDFDFDD4z2|
V: Ins
Z4|
V: Vocal
"D#m"z2AAAGGFG2GFA2z2|"D#m"z2AAAGGFGGGFA4|"B"z2AAAGGFG2GFA2D2|"C#"FDFDFDFDFDD2z4|
V: Ins
Z4|
% pre-chorus
V: Vocal
"D#m"d2d2a2aagggfa2z2|"D#m"d2dda2aagggfa3z|"B"d2d2a2aagggfa2d2|"C#"z8zdddfddc|
V: Ins
Z4|
% chorus
V: Vocal
"D#m"d2z2f2z3dddfddc|"F#"d2f2f2z4ddffdc|"C#"d2a2g2fg2f3dcdc|"G#"d2a2g2fg2f3"F#6"f4|
V: Ins
Z4|
V: Vocal
"D#m"z2a2d2a2d2a2d4|"F#"z2a2d2a2d2a2d3c|"C#"dcdcdcdcdcdcdcdc|"G#"dcdcdcdcdcdc"F#6"d2f2|
V: Ins
Z4|
V: Vocal
"D#m"z8zdddfddc|"D#m"d2z6zdddfddc|"F#"d2z8ddffdc|"C#"d2a2g2fg2f3dcdc|
V: Ins
Z4|
V: Vocal
"G#"d2a2g2fg2f3"F#6"f4|"D#m"z2a2d2a2d2a2d4|"F#"z2a2d2a2d2a2d3c|"C#"dcdcdcdcdcdcdcdc|
V: Ins
Z4|
V: Vocal
"G#"dcdcdcdcdcdc"F#6"d2f2|
V: Ins
Z|
% interlude
V: Vocal
"D#m"z16|"D#m"z16|"D#m"z16|"D#m"z16|
V: Ins
D,2D2A2D^^G2DG2^G^^G^GF|D,2D2A2D^^G2DG2^G4|D,2D2A2D^^G2DG2^G^^G^GF|D,2D2A2D^^G2z^G^^G^G4|
% verse
V: Vocal
"D#m"z2AAAGGFGGGFA2DD|"D#m"AAAAAGGFGFD2z4|"B"z2AAAGGFGGGFA2DD|"C#"FDFDFDFDFDD4z2|
V: Ins
Z4|
V: Vocal
"D#m"z2AAA2AAGGGFA3D|"D#m"AAAAA2AAGFD2z4|"B"z2AAA2AAGGGFA2DD|"C#"FDFDFDFDFDD2D2z2|
V: Ins
Z4|
% pre-chorus
V: Vocal
"D#m"d2dda2aagggfa3z|"D#m"d2dda2aagggfa3z|"B"d2dda2aagggfa4|"C#"d6f3dddfddc|
V: Ins
Z4|
% chorus
V: Vocal
"D#m"d2z2f2z3dddfddc|"F#"d2z8ddffdc|"C#"d2a2g2fg3f2dcdc|"G#"d2a2g2f4a2"F#6"f4|
V: Ins
Z4|
V: Vocal
"D#m"z2a2d2a2d2a2d4|"F#"z2a2d2a2d2a2d3c|"C#"dcdcdcdcdcdcdcdc|"G#"dcdcdcdcdcdc"F#6"d2f2|
V: Ins
Z4|
V: Vocal
"D#m"z8zdddfddc|"D#m"d2z6zdddfddc|"F#"d2z8ddffdc|"C#"d2a2g2fg3f2dcdc|
V: Ins
Z4|
V: Vocal
"G#"d2a2g2f4a2"F#6"f4|"D#m"z2a2d2a2d2a2d4|"F#"z2a2d2a2d2a2d3c|"C#"dcdcdcdcdcdcdcdc|
V: Ins
Z4|
V: Vocal
"G#"dcdcdcdcdcdc"F#6"d2f2|"D#m"z16|
V: Ins
Z|D2D2A2D^^G2DG2^G^^G^GF|
% interlude
V: Vocal
"D#m"z16|"D#m"z16|"D#m"z16|"D#m"z16|
V: Ins
D2D2A2D^^G2DG2^G^^G^GF|D,2D,2A,2D,^^G,2D,^G,^^G,^G,4|D,2D,2A,2D,^^G,2D,G,2^G,^^G,^G,F,|D,2D,2A,2D,^^G,2D,^G,^^G,^G,4|
V: Vocal
"D#m"z16|"D#m"z16|"D#m"z16|"D#m"z16|
V: Ins
D,2D,2A,2D,^^G,2D,G,2^G,^^G,^G,F,|D,2D,2A,2D,^^G,2D,^G,^^G,^G,4|D,2D,2A,2D,^^G,2D,G,2^G,^^G,^G,F,|D,2D,2A,2D,^^G,2D,^G,^^G,^G,4|
V: Vocal
"D#m"z16|"F#"z16|"C#"z16|"G#"z12"F#6"z4|
V: Ins
d12d2cA|c12d2cA|G12z4|FEDFEDFEDFEDFEDC|
V: Vocal
"D#m"z16|"F#"z16|"C#"z16|"G#"z8zdddfddc|
V: Ins
D4a8z2gf|gfd12z2|D,F,G,A,CDFGAcdff4|g8z8|
% chorus
V: Vocal
"D#m"d2a2f2z3dddfddc|"F#"d2a2f2z4ddffdc|"C#"d2a2g2fg3f2dcdc|"G#"d2a2g2fg3f2"F#"f4|
V: Ins
Z4|
V: Vocal
"D#m"z2a2d2a2d2a2d4|"F#"z2a2d2a2d2a2d3c|"C#"dcdcdcdcdcdcdcdc|"G#"dcdcdcdcdcdc"F#6"d2f2|
V: Ins
Z4|
V: Vocal
"D#m"d2a2f2z3dddfddc|"F#"d2a2f2z4ddffdc|"C#"d2a2g2fg3f2dcdc|"G#"d2a2g2fg3f2"F#"f4|
V: Ins
Z4|
V: Vocal
"D#m"z2a2d2a2d2a2d4|"F#"z2a2d2a2d2a2d3c|"C#"dcdcdcdcdcdcdcdc|"G#"dcdcdcdcdcdc"F#6"d2f2|
V: Ins
Z4|
% outro
V: Vocal
"D#m"z16|"D#m"z16|"D#m"z16|
V: Ins
D,2D,2A,2D,^^G,2D,G,2^G,^^G,^G,F,|D,2D,2A,2D,^^G,2D,G,2^G,4|Z|
+7
View File
@@ -0,0 +1,7 @@
{
"id": "city_lights",
"style": "English, warm piano pop, expressive female voice, acoustic piano, rounded bass and light drums, lyrical memorable melody, unhurried phrasing, 88 BPM",
"lyrics": "[Verse]\nNeon fades along the lane\nFootsteps keep the time of rain\nFold the night and leave it here\nMorning has a sky to clear\n\n[Chorus]\nLet the day come into view\nEvery road begins with you\nHold a little room for light\nWe will sing beyond the night",
"cot": "full",
"seed": 831001
}
-204
View File
@@ -1,204 +0,0 @@
import json
import numpy as np
import einops
class CodecManipulator(object):
r"""
**mm tokenizer v0.1**
see codeclm/hf/mm_tokenizer_v0.1_hf/id2vocab.json
text tokens:
llama tokenizer 0~31999
special tokens: "32000": "<EOD>", "32001": "<SOA>", "32002": "<EOA>", "32003": "<SOI>", "32004": "<EOI>", "32005": "<SOV>", "32006": "<EOV>", "32007": "<s_local>", "32008": "<e_local>", "32009": "<s_global>", "32010": "<e_global>", "32011": "<semantic>", "32012": "<acoustic>", "32013": "<low_level>", "32014": "<dac_16k>", "32015": "<dac_44k>", "32016": "<xcodec>", "32017": "<placeholder>", "32018": "<semantic_mert>", "32019": "<semantic_hubert>", "32020": "<visual>", "32021": "<semanticodec>"
mm tokens:
dac_16k: 4 codebook, 1024 vocab, 32022 - 36117
dac_44k: 9 codebook, 1024 vocab, 36118 - 45333
xcodec: 12 codebook, 1024 vocab, 45334 - 57621
semantic mert: 1024, 57622 - 58645
semantic hubert: 512, 58646 - 59157
visual: 64000, not included in v0.1
semanticodec 100tps 16384: semantic=16384, 59158 - 75541, acoustic=8192, 75542 - 83733
"""
def __init__(self, codec_type, quantizer_begin=None, n_quantizer=None, teacher_forcing=False, data_feature="codec"):
self.codec_type = codec_type
self.mm_v0_2_cfg = {
"dac16k": {"codebook_size": 1024, "num_codebooks": 4, "global_offset": 32022, "sep": ["<dac_16k>"], "fps": 50},
"dac44k": {"codebook_size": 1024, "num_codebooks": 9, "global_offset": 36118, "sep": ["<dac_44k>"]},
"xcodec": {"codebook_size": 1024, "num_codebooks": 12, "global_offset": 45334, "sep": ["<xcodec>"], "fps": 50},
"mert": {"codebook_size": 1024, "global_offset": 57622, "sep": ["<semantic_mert>"]},
"hubert": {"codebook_size": 512, "global_offset": 58646, "sep": ["<semantic_hubert>"]},
"semantic/s": {"codebook_size": 16384, "num_codebooks": 1, "global_offset": 59158, "sep": ["<semanticodec>", "<semantic>"]},
"semantic/a": {"codebook_size": 8192, "num_codebooks": 1, "global_offset": 75542, "sep": ["<semanticodec>", "<acoustic>"]},
"semanticodec": {"codebook_size": [16384, 8192], "num_codebooks": 2, "global_offset": 59158, "sep": ["<semanticodec>"], "fps": 50},
"special_tokens": {
'<EOD>': 32000, '<SOA>': 32001, '<EOA>': 32002, '<SOI>': 32003, '<EOI>': 32004, '<SOV>': 32005, '<EOV>': 32006, '<s_local>': 32007, '<e_local>': 32008, '<s_global>': 32009, '<e_global>': 32010, '<semantic>': 32011, '<acoustic>': 32012, '<stage_1>': 32013, '<dac_16k>': 32014, '<dac_44k>': 32015, '<xcodec>': 32016, '<stage_2>': 32017, '<semantic_mert>': 32018, '<semantic_hubert>': 32019, '<visual>': 32020, '<semanticodec>': 32021
},
"metadata": {
"len": 83734,
"text_range": [0, 31999],
"special_range": [32000, 32021],
"mm_range": [32022, 83733]
},
"codec_range": {
"dac16k": [32022, 36117],
"dac44k": [36118, 45333],
"xcodec": [45334, 57621],
# "hifi16k": [53526, 57621],
"mert": [57622, 58645],
"hubert": [58646, 59157],
"semantic/s": [59158, 75541],
"semantic/a": [75542, 83733],
"semanticodec": [59158, 83733]
}
}
self.sep = self.mm_v0_2_cfg[self.codec_type]["sep"]
self.sep_ids = [self.mm_v0_2_cfg["special_tokens"][s] for s in self.sep]
self.codebook_size = self.mm_v0_2_cfg[self.codec_type]["codebook_size"]
self.num_codebooks = self.mm_v0_2_cfg[self.codec_type]["num_codebooks"]
self.global_offset = self.mm_v0_2_cfg[self.codec_type]["global_offset"]
self.fps = self.mm_v0_2_cfg[self.codec_type]["fps"] if "fps" in self.mm_v0_2_cfg[self.codec_type] else None
self.quantizer_begin = quantizer_begin if quantizer_begin is not None else 0
self.n_quantizer = n_quantizer if n_quantizer is not None else self.num_codebooks
self.teacher_forcing = teacher_forcing
self.data_feature = data_feature
def offset_tok_ids(self, x, global_offset=0, codebook_size=2048, num_codebooks=4):
"""
x: (K, T)
"""
if isinstance(codebook_size, int):
assert x.max() < codebook_size, f"max(x)={x.max()}, codebook_size={codebook_size}"
elif isinstance(codebook_size, list):
for i, cs in enumerate(codebook_size):
assert x[i].max() < cs, f"max(x)={x[i].max()}, codebook_size={cs}, layer_id={i}"
else:
raise ValueError(f"codebook_size={codebook_size}")
assert x.min() >= 0, f"min(x)={x.min()}"
assert x.shape[0] == num_codebooks or x.shape[0] == self.n_quantizer, \
f"x.shape[0]={x.shape[0]}, num_codebooks={num_codebooks}, n_quantizer={self.n_quantizer}"
_x = x.copy()
_x = _x.astype(np.uint32)
cum_offset = 0
quantizer_begin = self.quantizer_begin
quantizer_end = quantizer_begin+self.n_quantizer
for k in range(self.quantizer_begin, quantizer_end): # k: quantizer_begin to quantizer_end - 1
if isinstance(codebook_size, int):
_x[k] += global_offset + k * codebook_size
elif isinstance(codebook_size, list):
_x[k] += global_offset + cum_offset
cum_offset += codebook_size[k]
else:
raise ValueError(f"codebook_size={codebook_size}")
return _x[quantizer_begin:quantizer_end]
def unoffset_tok_ids(self, x, global_offset=0, codebook_size=2048, num_codebooks=4):
"""
x: (K, T)
"""
if isinstance(codebook_size, int):
assert x.max() < global_offset + codebook_size * num_codebooks, f"max(x)={x.max()}, codebook_size={codebook_size}"
elif isinstance(codebook_size, list):
assert x.max() < global_offset + sum(codebook_size), f"max(x)={x.max()}, codebook_size={codebook_size}"
assert x.min() >= global_offset, f"min(x)={x.min()}, global_offset={global_offset}"
assert x.shape[0] == num_codebooks or x.shape[0] == self.n_quantizer, \
f"x.shape[0]={x.shape[0]}, num_codebooks={num_codebooks}, n_quantizer={self.n_quantizer}"
_x = x.copy()
_x = _x.astype(np.uint32)
cum_offset = 0
quantizer_begin = self.quantizer_begin
quantizer_end = quantizer_begin+self.n_quantizer
for k in range(quantizer_begin, quantizer_end):
if isinstance(codebook_size, int):
_x[k-quantizer_begin] -= global_offset + k * codebook_size
elif isinstance(codebook_size, list):
_x[k-quantizer_begin] -= global_offset + cum_offset
cum_offset += codebook_size[k]
else:
raise ValueError(f"codebook_size={codebook_size}")
return _x
def flatten(self, x):
if len(x.shape) > 2:
x = x.squeeze()
assert x.shape[0] == self.num_codebooks or x.shape[0] == self.n_quantizer, \
f"x.shape[0]={x.shape[0]}, num_codebooks={self.num_codebooks}, n_quantizer={self.n_quantizer}"
return einops.rearrange(x, 'K T -> (T K)')
def unflatten(self, x, n_quantizer=None):
if x.ndim > 1 and x.shape[0] == 1:
x = x.squeeze(0)
assert len(x.shape) == 1
assert x.shape[0] % self.num_codebooks == 0 or x.shape[0] % self.n_quantizer == 0, \
f"x.shape[0]={x.shape[0]}, num_codebooks={self.num_codebooks}, n_quantizer={self.n_quantizer}"
if n_quantizer!=self.num_codebooks:
return einops.rearrange(x, '(T K) -> K T', K=n_quantizer)
return einops.rearrange(x, '(T K) -> K T', K=self.num_codebooks)
# def check_codec_type_from_path(self, path):
# if self.codec_type == "hifi16k":
# assert "academicodec_hifi_16k_320d_large_uni" in path
def get_codec_type_from_range(self, ids):
ids_range = [ids.min(), ids.max()]
codec_range = self.mm_v0_2_cfg["codec_range"]
for codec_type, r in codec_range.items():
if ids_range[0] >= r[0] and ids_range[1] <= r[1]:
return codec_type
raise ValueError(f"ids_range={ids_range}, codec_range={codec_range}")
def npy2ids(self, npy):
if isinstance(npy, str):
data = np.load(npy)
elif isinstance(npy, np.ndarray):
data = npy
else:
raise ValueError(f"not supported type: {type(npy)}")
# data = data.squeeze()
assert len(data.shape)==2, f'data shape: {data.shape} is not (n_codebook, seq_len)'
data = self.offset_tok_ids(
data,
global_offset=self.global_offset,
codebook_size=self.codebook_size,
num_codebooks=self.num_codebooks,
)
data = self.flatten(data)
codec_range = self.get_codec_type_from_range(data)
assert codec_range == self.codec_type, f"get_codec_type_from_range(data)={codec_range}, self.codec_type={self.codec_type}"
data = data.tolist()
return data
def ids2npy(self, token_ids):
# make sure token_ids starts with codebook 0
if isinstance(self.codebook_size, int):
codebook_0_range = (self.global_offset + self.quantizer_begin*self.codebook_size, self.global_offset + (self.quantizer_begin+1)*self.codebook_size)
elif isinstance(self.codebook_size, list):
codebook_0_range = (self.global_offset, self.global_offset + self.codebook_size[0])
assert token_ids[0] >= codebook_0_range[0] \
and token_ids[0] < codebook_0_range[1], f"token_ids[0]={token_ids[self.quantizer_begin]}, codebook_0_range={codebook_0_range}"
data = np.array(token_ids)
data = self.unflatten(data, n_quantizer=self.n_quantizer)
data = self.unoffset_tok_ids(
data,
global_offset=self.global_offset,
codebook_size=self.codebook_size,
num_codebooks=self.num_codebooks,
)
return data
def npy_to_json_str(self, npy_path):
data = self.npy2ids(npy_path)
return json.dumps({"text": data, "src": npy_path, "codec": self.codec_type})
def sep(self):
return ''.join(self.sep)
def sep_ids(self):
return self.sep_ids
-111
View File
@@ -1,111 +0,0 @@
# import argparse
from exllamav2 import (
ExLlamaV2Cache,
ExLlamaV2Cache_Q4,
ExLlamaV2Cache_Q6,
ExLlamaV2Cache_Q8,
)
from transformers import LogitsProcessor
import torch
import random
import numpy as np
# parser = argparse.ArgumentParser()
# # Model Configuration:
# parser.add_argument("--stage1_model", type=str, default="m-a-p/YuE-s1-7B-anneal-en-cot", help="The model checkpoint path or identifier for the Stage 1 model.")
# parser.add_argument("--stage2_model", type=str, default="m-a-p/YuE-s2-1B-general", help="The model checkpoint path or identifier for the Stage 2 model.")
# parser.add_argument("--max_new_tokens", type=int, default=3000, help="The maximum number of new tokens to generate in one pass during text generation.")
# parser.add_argument("--repetition_penalty", type=float, default=1.1, help="repetition_penalty ranges from 1.0 to 2.0 (or higher in some cases). It controls the diversity and coherence of the audio tokens generated. The higher the value, the greater the discouragement of repetition. Setting value to 1.0 means no penalty.")
# parser.add_argument("--run_n_segments", type=int, default=2, help="The number of segments to process during the generation.")
# parser.add_argument("--stage1_use_exl2", action="store_true", help="Use exllamav2 to load and run stage 1 model.")
# parser.add_argument("--stage2_use_exl2", action="store_true", help="Use exllamav2 to load and run stage 2 model.")
# parser.add_argument("--stage2_batch_size", type=int, default=4, help="The non-exl2 batch size used in Stage 2 inference.")
# parser.add_argument("--stage1_cache_size", type=int, default=16384, help="The cache size used in Stage 1 inference.")
# parser.add_argument("--stage2_cache_size", type=int, default=8192, help="The exl2 cache size used in Stage 2 inference.")
# parser.add_argument("--stage1_cache_mode", type=str, default="FP16", help="The cache mode used in Stage 1 inference (FP16, Q8, Q6, Q4). Quantized k/v cache will save VRAM at the cost of some speed and precision.")
# parser.add_argument("--stage2_cache_mode", type=str, default="FP16", help="The cache mode used in Stage 2 inference (FP16, Q8, Q6, Q4). Quantized k/v cache will save VRAM at the cost of some speed and precision.")
# parser.add_argument("--stage1_no_guidance", action="store_true", help="Disable classifier-free guidance for stage 1")
# # Prompt
# parser.add_argument(
# "--genre_txt",
# type=str,
# required=True,
# help="The file path to a text file containing genre tags that describe the musical style or characteristics (e.g., instrumental, genre, mood, vocal timbre, vocal gender). This is used as part of the generation prompt.",
# )
# parser.add_argument(
# "--lyrics_txt",
# type=str,
# required=True,
# help="The file path to a text file containing the lyrics for the music generation. These lyrics will be processed and split into structured segments to guide the generation process.",
# )
# parser.add_argument(
# "--use_audio_prompt",
# action="store_true",
# help="If set, the model will use an audio file as a prompt during generation. The audio file should be specified using --audio_prompt_path.",
# )
# parser.add_argument(
# "--audio_prompt_path", type=str, default="", help="The file path to an audio file to use as a reference prompt when --use_audio_prompt is enabled."
# )
# parser.add_argument("--prompt_start_time", type=float, default=0.0, help="The start time in seconds to extract the audio prompt from the given audio file.")
# parser.add_argument("--prompt_end_time", type=float, default=30.0, help="The end time in seconds to extract the audio prompt from the given audio file.")
# parser.add_argument(
# "--use_dual_tracks_prompt",
# action="store_true",
# help="If set, the model will use dual tracks as a prompt during generation. The vocal and instrumental files should be specified using --vocal_track_prompt_path and --instrumental_track_prompt_path.",
# )
# parser.add_argument(
# "--vocal_track_prompt_path",
# type=str,
# default="",
# help="The file path to a vocal track file to use as a reference prompt when --use_dual_tracks_prompt is enabled.",
# )
# parser.add_argument(
# "--instrumental_track_prompt_path",
# type=str,
# default="",
# help="The file path to an instrumental track file to use as a reference prompt when --use_dual_tracks_prompt is enabled.",
# )
# # Output
# parser.add_argument("--output_dir", type=str, default="./output", help="The directory where generated outputs will be saved.")
# parser.add_argument("--keep_intermediate", action="store_true", help="If set, intermediate outputs will be saved during processing.")
# parser.add_argument("--disable_offload_model", action="store_true", help="If set, the model will not be offloaded from the GPU to CPU after Stage 1 inference.")
# parser.add_argument("--cuda_idx", type=int, default=0)
# parser.add_argument("--seed", type=int, default=None, help="An integer value to reproduce generation.")
# # Config for xcodec and upsampler
# parser.add_argument("--basic_model_config", default="./xcodec_mini_infer/final_ckpt/config.yaml", help="YAML files for xcodec configurations.")
# parser.add_argument("--resume_path", default="./xcodec_mini_infer/final_ckpt/ckpt_00360000.pth", help="Path to the xcodec checkpoint.")
# parser.add_argument("--config_path", type=str, default="./xcodec_mini_infer/decoders/config.yaml", help="Path to Vocos config file.")
# parser.add_argument("--vocal_decoder_path", type=str, default="./xcodec_mini_infer/decoders/decoder_131000.pth", help="Path to Vocos decoder weights.")
# parser.add_argument("--inst_decoder_path", type=str, default="./xcodec_mini_infer/decoders/decoder_151000.pth", help="Path to Vocos decoder weights.")
# parser.add_argument("-r", "--rescale", action="store_true", help="Rescale output to avoid clipping.")
def seed_everything(seed: int = 42):
random.seed(seed)
np.random.seed(seed)
torch.manual_seed(seed)
torch.cuda.manual_seed_all(seed)
torch.backends.cudnn.deterministic = True
torch.backends.cudnn.benchmark = False
def get_cache_class(cache_mode: str):
match cache_mode:
case "Q4":
return ExLlamaV2Cache_Q4
case "Q6":
return ExLlamaV2Cache_Q6
case "Q8":
return ExLlamaV2Cache_Q8
case _:
return ExLlamaV2Cache
class BlockTokenRangeProcessor(LogitsProcessor):
def __init__(self, start_id, end_id):
self.blocked_token_ids = list(range(start_id, end_id))
def __call__(self, input_ids, scores):
scores[:, self.blocked_token_ids] = -float("inf")
return scores
-492
View File
@@ -1,492 +0,0 @@
import os
import sys
sys.path.append(os.path.join(os.path.dirname(os.path.abspath(__file__)), 'xcodec_mini_infer'))
sys.path.append(os.path.join(os.path.dirname(os.path.abspath(__file__)), 'xcodec_mini_infer', 'descriptaudiocodec'))
import re
import random
# import uuid
import copy
from tqdm import tqdm
from collections import Counter
#import argparse
import numpy as np
import torch
import torchaudio
from torchaudio.transforms import Resample
# import soundfile as sf
# from einops import rearrange
from transformers import AutoTokenizer, AutoModelForCausalLM, LogitsProcessor, LogitsProcessorList
# from omegaconf import OmegaConf
# from .codecmanipulator import CodecManipulator
# from .mmtokenizer import _MMSentencePieceTokenizer
# from .xcodec_mini_infer.models.soundstream_hubert_new import SoundStream
# from .xcodec_mini_infer.vocoder import build_codec_model, process_audio
# from .xcodec_mini_infer.post_process_audio import replace_low_freq_with_energy_matched
# parser = argparse.ArgumentParser()
# # Model Configuration:
# parser.add_argument("--stage1_model", type=str, default="m-a-p/YuE-s1-7B-anneal-en-cot", help="The model checkpoint path or identifier for the Stage 1 model.")
# parser.add_argument("--stage2_model", type=str, default="m-a-p/YuE-s2-1B-general", help="The model checkpoint path or identifier for the Stage 2 model.")
# parser.add_argument("--max_new_tokens", type=int, default=3000, help="The maximum number of new tokens to generate in one pass during text generation.")
# parser.add_argument("--repetition_penalty", type=float, default=1.1, help="repetition_penalty ranges from 1.0 to 2.0 (or higher in some cases). It controls the diversity and coherence of the audio tokens generated. The higher the value, the greater the discouragement of repetition. Setting value to 1.0 means no penalty.")
# parser.add_argument("--run_n_segments", type=int, default=2, help="The number of segments to process during the generation.")
# parser.add_argument("--stage2_batch_size", type=int, default=4, help="The batch size used in Stage 2 inference.")
# # Prompt
# parser.add_argument("--genre_txt", type=str, required=True, help="The file path to a text file containing genre tags that describe the musical style or characteristics (e.g., instrumental, genre, mood, vocal timbre, vocal gender). This is used as part of the generation prompt.")
# parser.add_argument("--lyrics_txt", type=str, required=True, help="The file path to a text file containing the lyrics for the music generation. These lyrics will be processed and split into structured segments to guide the generation process.")
# parser.add_argument("--use_audio_prompt", action="store_true", help="If set, the model will use an audio file as a prompt during generation. The audio file should be specified using --audio_prompt_path.")
# parser.add_argument("--audio_prompt_path", type=str, default="", help="The file path to an audio file to use as a reference prompt when --use_audio_prompt is enabled.")
# parser.add_argument("--prompt_start_time", type=float, default=0.0, help="The start time in seconds to extract the audio prompt from the given audio file.")
# parser.add_argument("--prompt_end_time", type=float, default=30.0, help="The end time in seconds to extract the audio prompt from the given audio file.")
# parser.add_argument("--use_dual_tracks_prompt", action="store_true", help="If set, the model will use dual tracks as a prompt during generation. The vocal and instrumental files should be specified using --vocal_track_prompt_path and --instrumental_track_prompt_path.")
# parser.add_argument("--vocal_track_prompt_path", type=str, default="", help="The file path to a vocal track file to use as a reference prompt when --use_dual_tracks_prompt is enabled.")
# parser.add_argument("--instrumental_track_prompt_path", type=str, default="", help="The file path to an instrumental track file to use as a reference prompt when --use_dual_tracks_prompt is enabled.")
# # Output
# parser.add_argument("--output_dir", type=str, default="./output", help="The directory where generated outputs will be saved.")
# parser.add_argument("--keep_intermediate", action="store_true", help="If set, intermediate outputs will be saved during processing.")
# parser.add_argument("--disable_offload_model", action="store_true", help="If set, the model will not be offloaded from the GPU to CPU after Stage 1 inference.")
# parser.add_argument("--cuda_idx", type=int, default=0)
# parser.add_argument("--seed", type=int, default=42, help="An integer value to reproduce generation.")
# # Config for xcodec and upsampler
# parser.add_argument('--basic_model_config', default='./xcodec_mini_infer/final_ckpt/config.yaml', help='YAML files for xcodec configurations.')
# parser.add_argument('--resume_path', default='./xcodec_mini_infer/final_ckpt/ckpt_00360000.pth', help='Path to the xcodec checkpoint.')
# parser.add_argument('--config_path', type=str, default='./xcodec_mini_infer/decoders/config.yaml', help='Path to Vocos config file.')
# parser.add_argument('--vocal_decoder_path', type=str, default='./xcodec_mini_infer/decoders/decoder_131000.pth', help='Path to Vocos decoder weights.')
# parser.add_argument('--inst_decoder_path', type=str, default='./xcodec_mini_infer/decoders/decoder_151000.pth', help='Path to Vocos decoder weights.')
# parser.add_argument('-r', '--rescale', action='store_true', help='Rescale output to avoid clipping.')
# args = parser.parse_args()
# if args.use_audio_prompt and not args.audio_prompt_path:
# raise FileNotFoundError("Please offer audio prompt filepath using '--audio_prompt_path', when you enable 'use_audio_prompt'!")
# if args.use_dual_tracks_prompt and not args.vocal_track_prompt_path and not args.instrumental_track_prompt_path:
# raise FileNotFoundError("Please offer dual tracks prompt filepath using '--vocal_track_prompt_path' and '--inst_decoder_path', when you enable '--use_dual_tracks_prompt'!")
# stage1_model = args.stage1_model
# stage2_model = args.stage2_model
# cuda_idx = args.cuda_idx
# max_new_tokens = args.max_new_tokens
# stage1_output_dir = os.path.join(args.output_dir, f"stage1")
# stage2_output_dir = stage1_output_dir.replace('stage1', 'stage2')
# os.makedirs(stage1_output_dir, exist_ok=True)
# os.makedirs(stage2_output_dir, exist_ok=True)
def seed_everything(seed=42):
random.seed(seed)
np.random.seed(seed)
torch.manual_seed(seed)
torch.cuda.manual_seed_all(seed)
torch.backends.cudnn.deterministic = True
torch.backends.cudnn.benchmark = False
# seed_everything(args.seed)
# # load tokenizer and model
# device = torch.device(f"cuda:{cuda_idx}" if torch.cuda.is_available() else "cpu")
# mmtokenizer = _MMSentencePieceTokenizer("./mm_tokenizer_v0.2_hf/tokenizer.model")
# model = AutoModelForCausalLM.from_pretrained(
# stage1_model,
# torch_dtype=torch.bfloat16,
# attn_implementation="flash_attention_2", # To enable flashattn, you have to install flash-attn
# # device_map="auto",
# )
# # to device, if gpu is available
# model.to(device)
# model.eval()
# if torch.__version__ >= "2.0.0":
# model = torch.compile(model)
# codectool = CodecManipulator("xcodec", 0, 1)
# codectool_stage2 = CodecManipulator("xcodec", 0, 8)
# model_config = OmegaConf.load(args.basic_model_config)
# codec_model = eval(model_config.generator.name)(**model_config.generator.config).to(device)
# parameter_dict = torch.load(args.resume_path, map_location='cpu', weights_only=False)
# codec_model.load_state_dict(parameter_dict['codec_model'])
# codec_model.to(device)
# codec_model.eval()
class BlockTokenRangeProcessor(LogitsProcessor):
def __init__(self, start_id, end_id):
self.blocked_token_ids = list(range(start_id, end_id))
def __call__(self, input_ids, scores):
scores[:, self.blocked_token_ids] = -float("inf")
return scores
def load_audio_mono(filepath, sampling_rate=16000):
audio, sr = torchaudio.load(filepath)
# Convert to mono
audio = torch.mean(audio, dim=0, keepdim=True)
# Resample if needed
if sr != sampling_rate:
resampler = Resample(orig_freq=sr, new_freq=sampling_rate)
audio = resampler(audio)
return audio
def encode_audio(codec_model, audio_prompt, device, target_bw=0.5):
if len(audio_prompt.shape) < 3:
audio_prompt.unsqueeze_(0)
with torch.no_grad():
raw_codes = codec_model.encode(audio_prompt.to(device), target_bw=target_bw)
raw_codes = raw_codes.transpose(0, 1)
raw_codes = raw_codes.cpu().numpy().astype(np.int16)
return raw_codes
def split_lyrics(lyrics):
pattern = r"\[(\w+)\](.*?)(?=\[|\Z)"
segments = re.findall(pattern, lyrics, re.DOTALL)
structured_lyrics = [f"[{seg[0]}]\n{seg[1].strip()}\n\n" for seg in segments]
return structured_lyrics
# Call the function and print the result
# stage1_output_set = []
# # Tips:
# # genre tags support instrumental,genre,mood,vocal timbr and vocal gender
# # all kinds of tags are needed
# with open(args.genre_txt) as f:
# genres = f.read().strip()
# with open(args.lyrics_txt) as f:
# lyrics = split_lyrics(f.read())
# # intruction
# full_lyrics = "\n".join(lyrics)
# prompt_texts = [f"Generate music from the given lyrics segment by segment.\n[Genre] {genres}\n{full_lyrics}"]
# prompt_texts += lyrics
# random_id = uuid.uuid4()
# output_seq = None
# # Here is suggested decoding config
# top_p = 0.93
# temperature = 1.0
# repetition_penalty = args.repetition_penalty
# # special tokens
# start_of_segment = mmtokenizer.tokenize('[start_of_segment]')
# end_of_segment = mmtokenizer.tokenize('[end_of_segment]')
# # Format text prompt
# run_n_segments = min(args.run_n_segments+1, len(lyrics))
# for i, p in enumerate(tqdm(prompt_texts[:run_n_segments], desc="Stage1 inference...")):
# section_text = p.replace('[start_of_segment]', '').replace('[end_of_segment]', '')
# guidance_scale = 1.5 if i <=1 else 1.2
# if i==0:
# continue
# if i==1:
# if args.use_dual_tracks_prompt or args.use_audio_prompt:
# if args.use_dual_tracks_prompt:
# vocals_ids = load_audio_mono(args.vocal_track_prompt_path)
# instrumental_ids = load_audio_mono(args.instrumental_track_prompt_path)
# vocals_ids = encode_audio(codec_model, vocals_ids, device, target_bw=0.5)
# instrumental_ids = encode_audio(codec_model, instrumental_ids, device, target_bw=0.5)
# vocals_ids = codectool.npy2ids(vocals_ids[0])
# instrumental_ids = codectool.npy2ids(instrumental_ids[0])
# ids_segment_interleaved = rearrange([np.array(vocals_ids), np.array(instrumental_ids)], 'b n -> (n b)')
# audio_prompt_codec = ids_segment_interleaved[int(args.prompt_start_time*50*2): int(args.prompt_end_time*50*2)]
# audio_prompt_codec = audio_prompt_codec.tolist()
# elif args.use_audio_prompt:
# audio_prompt = load_audio_mono(args.audio_prompt_path)
# raw_codes = encode_audio(codec_model, audio_prompt, device, target_bw=0.5)
# # Format audio prompt
# code_ids = codectool.npy2ids(raw_codes[0])
# audio_prompt_codec = code_ids[int(args.prompt_start_time *50): int(args.prompt_end_time *50)] # 50 is tps of xcodec
# audio_prompt_codec_ids = [mmtokenizer.soa] + codectool.sep_ids + audio_prompt_codec + [mmtokenizer.eoa]
# sentence_ids = mmtokenizer.tokenize("[start_of_reference]") + audio_prompt_codec_ids + mmtokenizer.tokenize("[end_of_reference]")
# head_id = mmtokenizer.tokenize(prompt_texts[0]) + sentence_ids
# else:
# head_id = mmtokenizer.tokenize(prompt_texts[0])
# prompt_ids = head_id + start_of_segment + mmtokenizer.tokenize(section_text) + [mmtokenizer.soa] + codectool.sep_ids
# else:
# prompt_ids = end_of_segment + start_of_segment + mmtokenizer.tokenize(section_text) + [mmtokenizer.soa] + codectool.sep_ids
# prompt_ids = torch.as_tensor(prompt_ids).unsqueeze(0).to(device)
# input_ids = torch.cat([raw_output, prompt_ids], dim=1) if i > 1 else prompt_ids
# # Use window slicing in case output sequence exceeds the context of model
# max_context = 16384-max_new_tokens-1
# if input_ids.shape[-1] > max_context:
# print(f'Section {i}: output length {input_ids.shape[-1]} exceeding context length {max_context}, now using the last {max_context} tokens.')
# input_ids = input_ids[:, -(max_context):]
# with torch.no_grad():
# output_seq = model.generate(
# input_ids=input_ids,
# max_new_tokens=max_new_tokens,
# min_new_tokens=100,
# do_sample=True,
# top_p=top_p,
# temperature=temperature,
# repetition_penalty=repetition_penalty,
# eos_token_id=mmtokenizer.eoa,
# pad_token_id=mmtokenizer.eoa,
# logits_processor=LogitsProcessorList([BlockTokenRangeProcessor(0, 32002), BlockTokenRangeProcessor(32016, 32016)]),
# guidance_scale=guidance_scale,
# )
# if output_seq[0][-1].item() != mmtokenizer.eoa:
# tensor_eoa = torch.as_tensor([[mmtokenizer.eoa]]).to(model.device)
# output_seq = torch.cat((output_seq, tensor_eoa), dim=1)
# if i > 1:
# raw_output = torch.cat([raw_output, prompt_ids, output_seq[:, input_ids.shape[-1]:]], dim=1)
# else:
# raw_output = output_seq
# # save raw output and check sanity
# ids = raw_output[0].cpu().numpy()
# soa_idx = np.where(ids == mmtokenizer.soa)[0].tolist()
# eoa_idx = np.where(ids == mmtokenizer.eoa)[0].tolist()
# if len(soa_idx)!=len(eoa_idx):
# raise ValueError(f'invalid pairs of soa and eoa, Num of soa: {len(soa_idx)}, Num of eoa: {len(eoa_idx)}')
# vocals = []
# instrumentals = []
# range_begin = 1 if args.use_audio_prompt or args.use_dual_tracks_prompt else 0
# for i in range(range_begin, len(soa_idx)):
# codec_ids = ids[soa_idx[i]+1:eoa_idx[i]]
# if codec_ids[0] == 32016:
# codec_ids = codec_ids[1:]
# codec_ids = codec_ids[:2 * (codec_ids.shape[0] // 2)]
# vocals_ids = codectool.ids2npy(rearrange(codec_ids,"(n b) -> b n", b=2)[0])
# vocals.append(vocals_ids)
# instrumentals_ids = codectool.ids2npy(rearrange(codec_ids,"(n b) -> b n", b=2)[1])
# instrumentals.append(instrumentals_ids)
# vocals = np.concatenate(vocals, axis=1)
# instrumentals = np.concatenate(instrumentals, axis=1)
# vocal_save_path = os.path.join(stage1_output_dir, f"{genres.replace(' ', '-')}_tp{top_p}_T{temperature}_rp{repetition_penalty}_maxtk{max_new_tokens}_{random_id}_vtrack".replace('.', '@')+'.npy')
# inst_save_path = os.path.join(stage1_output_dir, f"{genres.replace(' ', '-')}_tp{top_p}_T{temperature}_rp{repetition_penalty}_maxtk{max_new_tokens}_{random_id}_itrack".replace('.', '@')+'.npy')
# np.save(vocal_save_path, vocals)
# np.save(inst_save_path, instrumentals)
# stage1_output_set.append(vocal_save_path)
# stage1_output_set.append(inst_save_path)
# # offload model
# if not args.disable_offload_model:
# model.cpu()
# del model
# torch.cuda.empty_cache()
# print("Stage 2 inference...")
# model_stage2 = AutoModelForCausalLM.from_pretrained(
# stage2_model,
# torch_dtype=torch.bfloat16,
# attn_implementation="flash_attention_2",
# # device_map="auto",
# )
# model_stage2.to(device)
# model_stage2.eval()
# if torch.__version__ >= "2.0.0":
# model_stage2 = torch.compile(model_stage2)
def stage2_generate(model,mmtokenizer,codectool,device, prompt, batch_size=16):
codec_ids = codectool.unflatten(prompt, n_quantizer=1)
codec_ids = codectool.offset_tok_ids(
codec_ids,
global_offset=codectool.global_offset,
codebook_size=codectool.codebook_size,
num_codebooks=codectool.num_codebooks,
).astype(np.int32)
# Prepare prompt_ids based on batch size or single input
if batch_size > 1:
codec_list = []
for i in range(batch_size):
idx_begin = i * 300
idx_end = (i + 1) * 300
codec_list.append(codec_ids[:, idx_begin:idx_end])
codec_ids = np.concatenate(codec_list, axis=0)
prompt_ids = np.concatenate(
[
np.tile([mmtokenizer.soa, mmtokenizer.stage_1], (batch_size, 1)),
codec_ids,
np.tile([mmtokenizer.stage_2], (batch_size, 1)),
],
axis=1
)
else:
prompt_ids = np.concatenate([
np.array([mmtokenizer.soa, mmtokenizer.stage_1]),
codec_ids.flatten(), # Flatten the 2D array to 1D
np.array([mmtokenizer.stage_2])
]).astype(np.int32)
prompt_ids = prompt_ids[np.newaxis, ...]
codec_ids = torch.as_tensor(codec_ids).to(device)
prompt_ids = torch.as_tensor(prompt_ids).to(device)
len_prompt = prompt_ids.shape[-1]
block_list = LogitsProcessorList([BlockTokenRangeProcessor(0, 46358), BlockTokenRangeProcessor(53526, mmtokenizer.vocab_size)])
# Teacher forcing generate loop
for frames_idx in range(codec_ids.shape[1]):
cb0 = codec_ids[:, frames_idx:frames_idx+1]
prompt_ids = torch.cat([prompt_ids, cb0], dim=1)
input_ids = prompt_ids
with torch.no_grad():
stage2_output = model.generate(input_ids=input_ids,
min_new_tokens=7,
max_new_tokens=7,
eos_token_id=mmtokenizer.eoa,
pad_token_id=mmtokenizer.eoa,
logits_processor=block_list,
)
assert stage2_output.shape[1] - prompt_ids.shape[1] == 7, f"output new tokens={stage2_output.shape[1]-prompt_ids.shape[1]}"
prompt_ids = stage2_output
# Return output based on batch size
if batch_size > 1:
output = prompt_ids.cpu().numpy()[:, len_prompt:]
output_list = [output[i] for i in range(batch_size)]
output = np.concatenate(output_list, axis=0)
else:
output = prompt_ids[0].cpu().numpy()[len_prompt:]
return output
def stage2_inference(model, stage1_output_set, stage2_output_dir,codectool_stage2,mmtokenizer,codectool,device, batch_size=4):
stage2_result = []
for i in tqdm(range(len(stage1_output_set))):
output_filename = os.path.join(stage2_output_dir, os.path.basename(stage1_output_set[i]))
if os.path.exists(output_filename):
print(f'{output_filename} stage2 has done.')
continue
# Load the prompt
prompt = np.load(stage1_output_set[i]).astype(np.int32)
# Only accept 6s segments
output_duration = prompt.shape[-1] // 50 // 6 * 6
num_batch = output_duration // 6
if num_batch <= batch_size:
# If num_batch is less than or equal to batch_size, we can infer the entire prompt at once
output = stage2_generate(model,mmtokenizer,codectool,device, prompt[:, :output_duration*50], batch_size=num_batch)
else:
# If num_batch is greater than batch_size, process in chunks of batch_size
segments = []
num_segments = (num_batch // batch_size) + (1 if num_batch % batch_size != 0 else 0)
for seg in range(num_segments):
start_idx = seg * batch_size * 300
# Ensure the end_idx does not exceed the available length
end_idx = min((seg + 1) * batch_size * 300, output_duration*50) # Adjust the last segment
current_batch_size = batch_size if seg != num_segments-1 or num_batch % batch_size == 0 else num_batch % batch_size
segment = stage2_generate(
model,mmtokenizer,codectool,device,
prompt[:, start_idx:end_idx],
batch_size=current_batch_size
)
segments.append(segment)
# Concatenate all the segments
output = np.concatenate(segments, axis=0)
# Process the ending part of the prompt
if output_duration*50 != prompt.shape[-1]:
ending = stage2_generate(model,mmtokenizer,codectool,device, prompt[:, output_duration*50:], batch_size=1)
output = np.concatenate([output, ending], axis=0)
output = codectool_stage2.ids2npy(output)
# Fix invalid codes (a dirty solution, which may harm the quality of audio)
# We are trying to find better one
fixed_output = copy.deepcopy(output)
for i, line in enumerate(output):
for j, element in enumerate(line):
if element < 0 or element > 1023:
counter = Counter(line)
most_frequant = sorted(counter.items(), key=lambda x: x[1], reverse=True)[0][0]
fixed_output[i, j] = most_frequant
# save output
np.save(output_filename, fixed_output)
stage2_result.append(output_filename)
return stage2_result
# stage2_result = stage2_inference(model_stage2, stage1_output_set, stage2_output_dir, batch_size=args.stage2_batch_size)
# print(stage2_result)
# print('Stage 2 DONE.\n')
# # convert audio tokens to audio
def save_audio(wav: torch.Tensor, path, sample_rate: int, rescale: bool = False):
folder_path = os.path.dirname(path)
if not os.path.exists(folder_path):
os.makedirs(folder_path)
limit = 0.99
max_val = wav.abs().max()
wav = wav * min(limit / max_val, 1) if rescale else wav.clamp(-limit, limit)
wav = wav.cpu()
torchaudio.save(str(path), wav, sample_rate=sample_rate, encoding='PCM_S', bits_per_sample=16)
# # reconstruct tracks
# recons_output_dir = os.path.join(args.output_dir, "recons")
# recons_mix_dir = os.path.join(recons_output_dir, 'mix')
# os.makedirs(recons_mix_dir, exist_ok=True)
# tracks = []
# for npy in stage2_result:
# codec_result = np.load(npy)
# decodec_rlt=[]
# with torch.no_grad():
# decoded_waveform = codec_model.decode(torch.as_tensor(codec_result.astype(np.int16), dtype=torch.long).unsqueeze(0).permute(1, 0, 2).to(device))
# decoded_waveform = decoded_waveform.cpu().squeeze(0)
# decodec_rlt.append(torch.as_tensor(decoded_waveform))
# decodec_rlt = torch.cat(decodec_rlt, dim=-1)
# save_path = os.path.join(recons_output_dir, os.path.splitext(os.path.basename(npy))[0] + ".mp3")
# tracks.append(save_path)
# save_audio(decodec_rlt, save_path, 16000)
# # mix tracks
# for inst_path in tracks:
# try:
# if (inst_path.endswith('.wav') or inst_path.endswith('.mp3')) \
# and '_itrack' in inst_path:
# # find pair
# vocal_path = inst_path.replace('_itrack', '_vtrack')
# if not os.path.exists(vocal_path):
# continue
# # mix
# recons_mix = os.path.join(recons_mix_dir, os.path.basename(inst_path).replace('_itrack', '_mixed'))
# vocal_stem, sr = sf.read(inst_path)
# instrumental_stem, _ = sf.read(vocal_path)
# mix_stem = (vocal_stem + instrumental_stem) / 1
# sf.write(recons_mix, mix_stem, sr)
# except Exception as e:
# print(e)
# # vocoder to upsample audios
# vocal_decoder, inst_decoder = build_codec_model(args.config_path, args.vocal_decoder_path, args.inst_decoder_path)
# vocoder_output_dir = os.path.join(args.output_dir, 'vocoder')
# vocoder_stems_dir = os.path.join(vocoder_output_dir, 'stems')
# vocoder_mix_dir = os.path.join(vocoder_output_dir, 'mix')
# os.makedirs(vocoder_mix_dir, exist_ok=True)
# os.makedirs(vocoder_stems_dir, exist_ok=True)
# for npy in stage2_result:
# if '_itrack' in npy:
# # Process instrumental
# instrumental_output = process_audio(
# npy,
# os.path.join(vocoder_stems_dir, 'itrack.mp3'),
# args.rescale,
# args,
# inst_decoder,
# codec_model
# )
# else:
# # Process vocal
# vocal_output = process_audio(
# npy,
# os.path.join(vocoder_stems_dir, 'vtrack.mp3'),
# args.rescale,
# args,
# vocal_decoder,
# codec_model
# )
# # mix tracks
# try:
# mix_output = instrumental_output + vocal_output
# vocoder_mix = os.path.join(vocoder_mix_dir, os.path.basename(recons_mix))
# save_audio(mix_output, vocoder_mix, 44100, args.rescale)
# print(f"Created mix: {vocoder_mix}")
# except RuntimeError as e:
# print(e)
# print(f"mix {vocoder_mix} failed! inst: {instrumental_output.shape}, vocal: {vocal_output.shape}")
# # Post process
# replace_low_freq_with_energy_matched(
# a_file=recons_mix, # 16kHz
# b_file=vocoder_mix, # 48kHz
# c_file=os.path.join(args.output_dir, os.path.basename(recons_mix)),
# cutoff_freq=5500.0
# )
-112
View File
@@ -1,112 +0,0 @@
import os
import numpy as np
import soundfile as sf
import torch
import torchaudio
from .common import seed_everything
from .xcodec_mini_infer.models.soundstream_hubert_new import SoundStream
from omegaconf import OmegaConf
from .xcodec_mini_infer.post_process_audio import replace_low_freq_with_energy_matched
from .xcodec_mini_infer.vocoder import build_codec_model, process_audio
# convert audio tokens to audio
def save_audio(wav: torch.Tensor, path, sample_rate: int, rescale: bool = False):
folder_path = os.path.dirname(path)
if not os.path.exists(folder_path):
os.makedirs(folder_path)
limit = 0.99
max_val = wav.abs().max()
wav = wav * min(limit / max_val, 1) if rescale else wav.clamp(-limit, limit)
torchaudio.save(str(path), wav, sample_rate=sample_rate, encoding="PCM_S", bits_per_sample=16)
def post_process(
codec_model: SoundStream, device: torch.device, output_dir: str, config_path: str, vocal_decoder_path: str, inst_decoder_path: str, rescale: bool,
file_prefix,):
# reconstruct tracks
recons_output_dir = os.path.join(output_dir, "recons")
recons_mix_dir = os.path.join(recons_output_dir, "mix")
os.makedirs(recons_mix_dir, exist_ok=True)
stage2_result = [os.path.join(output_dir, "stage2", filename) for filename in ["vtrack.npy", "itrack.npy"]]
tracks = []
for npy in stage2_result:
codec_result = np.load(npy)
decodec_rlt = []
decoded_waveform = codec_model.decode(torch.as_tensor(codec_result.astype(np.int16), dtype=torch.long).unsqueeze(0).permute(1, 0, 2).to(device))
decoded_waveform = decoded_waveform.cpu().squeeze(0)
decodec_rlt.append(torch.as_tensor(decoded_waveform))
decodec_rlt = torch.cat(decodec_rlt, dim=-1)
save_path = os.path.join(recons_output_dir, os.path.splitext(os.path.basename(npy))[0] + ".mp3")
tracks.append(save_path)
save_audio(decodec_rlt, save_path, 16000)
# mix tracks
for inst_path in tracks:
try:
if (inst_path.endswith(".wav") or inst_path.endswith(".mp3")) and "itrack" in inst_path:
# find pair
vocal_path = inst_path.replace("itrack", "vtrack")
if not os.path.exists(vocal_path):
continue
# mix
recons_mix = os.path.join(recons_mix_dir, os.path.basename(inst_path).replace("itrack", "mixed"))
vocal_stem, sr = sf.read(inst_path)
instrumental_stem, _ = sf.read(vocal_path)
mix_stem = (vocal_stem + instrumental_stem) / 1
sf.write(recons_mix, mix_stem, sr)
except Exception as e:
print(e)
# vocoder to upsample audios
vocal_decoder, inst_decoder = build_codec_model(config_path, vocal_decoder_path, inst_decoder_path)
vocoder_output_dir = os.path.join(output_dir, "vocoder")
vocoder_stems_dir = os.path.join(vocoder_output_dir, "stems")
vocoder_mix_dir = os.path.join(vocoder_output_dir, "mix")
os.makedirs(vocoder_mix_dir, exist_ok=True)
os.makedirs(vocoder_stems_dir, exist_ok=True)
for npy in stage2_result:
if "itrack" in npy:
# Process instrumental
instrumental_output = process_audio(npy, os.path.join(vocoder_stems_dir, "itrack.mp3"), rescale, device, inst_decoder, codec_model)
else:
# Process vocal
vocal_output = process_audio(npy, os.path.join(vocoder_stems_dir, "vtrack.mp3"), rescale, device, vocal_decoder, codec_model)
# mix tracks
try:
mix_output = instrumental_output + vocal_output
vocoder_mix = os.path.join(vocoder_mix_dir, os.path.basename(recons_mix))
save_audio(mix_output, vocoder_mix, 44100, rescale)
print(f"Created mix: {vocoder_mix}")
except RuntimeError as e:
print(e)
print(f"mix {vocoder_mix} failed! inst: {instrumental_output.shape}, vocal: {vocal_output.shape}")
# Post process
c_file=os.path.join(output_dir, f"yue_{file_prefix}_{os.path.basename(recons_mix)}")
replace_low_freq_with_energy_matched(
a_file=recons_mix, b_file=vocoder_mix, c_file=c_file, cutoff_freq=5500.0 # 16kHz # 48kHz
)
return mix_output,c_file
def main():
args = parser.parse_args()
if args.seed is not None:
seed_everything(args.seed)
device = torch.device(f"cuda:{args.cuda_idx}" if torch.cuda.is_available() else "cpu")
model_config = OmegaConf.load(args.basic_model_config)
assert model_config.generator.name == "SoundStream"
codec_model = SoundStream(**model_config.generator.config).to(device)
parameter_dict = torch.load(args.resume_path, map_location=device, weights_only=False)
codec_model.load_state_dict(parameter_dict["codec_model"])
codec_model.eval()
post_process(codec_model, device, args.output_dir, args.config_path, args.vocal_decoder_path, args.inst_decoder_path, args.rescale)
# if __name__ == "__main__":
# # enable inference mode globally
# torch.autograd.grad_mode._enter_inference_mode(True)
# torch.autograd.set_grad_enabled(False)
# main()
-511
View File
@@ -1,511 +0,0 @@
import os
import random
import re
from dataclasses import dataclass
import numpy as np
import torch
import torch.nn.functional as F
import torchaudio
from .codecmanipulator import CodecManipulator
from .common import BlockTokenRangeProcessor, seed_everything, get_cache_class
from einops import rearrange
from exllamav2 import ExLlamaV2, ExLlamaV2Config, ExLlamaV2Tokenizer
from exllamav2.generator import ExLlamaV2Sampler
from .mmtokenizer import _MMSentencePieceTokenizer
from .xcodec_mini_infer.models.soundstream_hubert_new import SoundStream
from omegaconf import OmegaConf
from torchaudio.transforms import Resample
from tqdm import tqdm
from transformers import AutoModelForCausalLM, LogitsProcessorList
from transformers.cache_utils import StaticCache
import folder_paths
@dataclass
class SampleSettings:
# Here is suggested decoding config
top_p = 0.93
temperature = 1
repetition_penalty = 1.1
guidance_scale_seg0 = 1.5 # None to disable cfg
guidance_scale = 1.2 # None to disable cfg
def __init__(self, use_guidance: bool = True, repetition_penalty: float = 1.1):
if not use_guidance:
self.guidance_scale_seg0 = None
self.guidance_scale = None
self.repetition_penalty = repetition_penalty
def load_audio_mono(filepath, sampling_rate=16000):
audio, sr = torchaudio.load(filepath)
# Convert to mono
audio = torch.mean(audio, dim=0, keepdim=True)
# Resample if needed
if sr != sampling_rate:
resampler = Resample(orig_freq=sr, new_freq=sampling_rate)
audio = resampler(audio)
return audio
def encode_audio(codec_model, audio_prompt, device, target_bw=0.5):
if len(audio_prompt.shape) < 3:
audio_prompt.unsqueeze_(0)
with torch.no_grad():
raw_codes = codec_model.encode(audio_prompt.to(device), target_bw=target_bw)
raw_codes = raw_codes.transpose(0, 1)
raw_codes = raw_codes.cpu().numpy().astype(np.int16)
return raw_codes
class Stage1Pipeline:
def __init__(self, device: torch.device, basic_model_config: str, resume_path: str):
self.device = device
self.codec_tool = CodecManipulator("xcodec", 0, 1)
self.basic_model_config = basic_model_config
self.resume_path = resume_path
self.codec_model = None
# Load tokenizer
self.mmtokenizer = _MMSentencePieceTokenizer(os.path.join(folder_paths.base_path,"custom_nodes/ComfyUI_YuE/inference/mm_tokenizer_v0.2_hf/tokenizer.model"))
self.start_of_segment = self.mmtokenizer.tokenize("[start_of_segment]")
self.end_of_segment = self.mmtokenizer.tokenize("[end_of_segment]")
def load_codec_model(self):
if self.codec_model is not None:
return
model_config = OmegaConf.load(self.basic_model_config)
assert model_config.generator.name == "SoundStream"
self.codec_model = SoundStream(**model_config.generator.config).to(self.device)
parameter_dict = torch.load(self.resume_path, map_location=self.device, weights_only=False)
self.codec_model.load_state_dict(parameter_dict["codec_model"])
self.codec_model.eval()
def get_prompt_texts(self, genres: str, lyrics: str):
def split_lyrics(lyrics):
pattern = r"\[(\w+)\](.*?)(?=\[|\Z)"
segments = re.findall(pattern, lyrics, re.DOTALL)
structured_lyrics = [f"[{seg[0]}]\n{seg[1].strip()}\n\n" for seg in segments]
return structured_lyrics
lyrics = split_lyrics(lyrics)
full_lyrics = "\n".join(lyrics)
prompt_texts = [f"Generate music from the given lyrics segment by segment.\n[Genre] {genres}\n{full_lyrics}"]
print(f"Full lyrics: {prompt_texts}")
prompt_texts += lyrics
return lyrics, prompt_texts
def get_audio_prompt_ids(
self,
use_dual_tracks_prompt: bool,
vocal_track_prompt_path: str,
instrumental_track_prompt_path: str,
use_audio_prompt: bool,
audio_prompt_path: str,
prompt_start_time: int,
prompt_end_time: int,
):
self.load_codec_model()
if use_dual_tracks_prompt:
vocals_ids = load_audio_mono(vocal_track_prompt_path)
instrumental_ids = load_audio_mono(instrumental_track_prompt_path)
vocals_ids = encode_audio(self.codec_model, vocals_ids, self.device, target_bw=0.5)
instrumental_ids = encode_audio(self.codec_model, instrumental_ids, self.device, target_bw=0.5)
vocals_ids = self.codec_tool.npy2ids(vocals_ids[0])
instrumental_ids = self.codec_tool.npy2ids(instrumental_ids[0])
ids_segment_interleaved = rearrange([np.array(vocals_ids), np.array(instrumental_ids)], "b n -> (n b)")
audio_prompt_codec = ids_segment_interleaved[int(prompt_start_time * 50 * 2) : int(prompt_end_time * 50 * 2)]
audio_prompt_codec = audio_prompt_codec.tolist()
elif use_audio_prompt:
audio_prompt = load_audio_mono(audio_prompt_path)
raw_codes = encode_audio(self.codec_model, audio_prompt, self.device, target_bw=0.5)
# Format audio prompt
code_ids = self.codec_tool.npy2ids(raw_codes[0])
audio_prompt_codec = code_ids[int(prompt_start_time * 50) : int(prompt_end_time * 50)] # 50 is tps of xcodec
audio_prompt_codec_ids = [self.mmtokenizer.soa] + self.codec_tool.sep_ids + audio_prompt_codec + [self.mmtokenizer.eoa]
sentence_ids = self.mmtokenizer.tokenize("[start_of_reference]") + audio_prompt_codec_ids + self.mmtokenizer.tokenize("[end_of_reference]")
return sentence_ids
def get_first_segment_prompt(
self,
segment_p: str,
prompt_text_0: str,
use_dual_tracks_prompt: bool,
vocal_track_prompt_path: str,
instrumental_track_prompt_path: str,
use_audio_prompt: bool,
audio_prompt_path: str,
prompt_start_time: int,
prompt_end_time: int,
):
section_text = segment_p.replace("[start_of_segment]", "").replace("[end_of_segment]", "")
head_id = self.mmtokenizer.tokenize(prompt_text_0)
if use_dual_tracks_prompt or use_audio_prompt:
head_id += self.get_audio_prompt_ids(
use_dual_tracks_prompt,
vocal_track_prompt_path,
instrumental_track_prompt_path,
use_audio_prompt,
audio_prompt_path,
prompt_start_time,
prompt_end_time,
)
return head_id + self.start_of_segment + self.mmtokenizer.tokenize(section_text) + [self.mmtokenizer.soa] + self.codec_tool.sep_ids
def get_segment_prompt(self, segment_p: str):
section_text = segment_p.replace("[start_of_segment]", "").replace("[end_of_segment]", "")
return self.end_of_segment + self.start_of_segment + self.mmtokenizer.tokenize(section_text) + [self.mmtokenizer.soa] + self.codec_tool.sep_ids
def save(self, raw_output: torch.Tensor, output_dir: str, use_audio_prompt: bool, use_dual_tracks_prompt: bool):
# save raw output and check sanity
ids = raw_output[0].cpu().numpy()
soa_idx = np.where(ids == self.mmtokenizer.soa)[0].tolist()
eoa_idx = np.where(ids == self.mmtokenizer.eoa)[0].tolist()
if len(soa_idx) != len(eoa_idx):
raise ValueError(f"invalid pairs of soa and eoa, Num of soa: {len(soa_idx)}, Num of eoa: {len(eoa_idx)}")
vocals = []
instrumentals = []
range_begin = 1 if use_audio_prompt or use_dual_tracks_prompt else 0
for i in range(range_begin, len(soa_idx)):
codec_ids = ids[soa_idx[i] + 1 : eoa_idx[i]]
if codec_ids[0] == 32016:
codec_ids = codec_ids[1:]
codec_ids = codec_ids[: 2 * (codec_ids.shape[0] // 2)]
vocals_ids = self.codec_tool.ids2npy(rearrange(codec_ids, "(n b) -> b n", b=2)[0])
vocals.append(vocals_ids)
instrumentals_ids = self.codec_tool.ids2npy(rearrange(codec_ids, "(n b) -> b n", b=2)[1])
instrumentals.append(instrumentals_ids)
vocals = np.concatenate(vocals, axis=1)
instrumentals = np.concatenate(instrumentals, axis=1)
stage1_output_dir = os.path.join(output_dir, "stage1")
os.makedirs(stage1_output_dir, exist_ok=True)
vocal_save_path = os.path.join(stage1_output_dir, "vtrack.npy")
inst_save_path = os.path.join(stage1_output_dir, "itrack.npy")
np.save(vocal_save_path, vocals)
np.save(inst_save_path, instrumentals)
def shorten_input(self, seq: torch.Tensor, max_context: int):
# Iteratively drop the oldest segment in the context until the sequence fits in context
pattern = torch.tensor(self.start_of_segment)
pattern_length = pattern.numel()
while seq.shape[-1] > max_context:
windows = seq[0].unfold(0, pattern_length, 1)
matches = (windows == pattern).all(dim=1)
match_indices = torch.nonzero(matches).flatten()
if match_indices.numel() < 3:
# Ensure that at least one other segment remains before the current segment for continuity
print("Unable to keep enough segments for smart context, falling back to simple truncation. " f"Now using the last {max_context} tokens.")
return seq[:, -max_context:]
first_segment_start = match_indices[0].item()
second_segment_start = match_indices[1].item()
seq = torch.cat((seq[:, :first_segment_start], seq[:, second_segment_start:]), dim=-1)
return seq
class Stage1Pipeline_HF(Stage1Pipeline):
def __init__(self, model_path: str, device: torch.device, cache_size: int, **kwargs):
super().__init__(device, **kwargs)
# Load HF model
self.model = AutoModelForCausalLM.from_pretrained(model_path, torch_dtype=torch.float16, attn_implementation="sdpa", device_map=self.device)
self.model.eval()
if torch.__version__ >= "2.0.0":
self.model = torch.compile(self.model)
self.cache_size = cache_size
def generate(
self,
use_dual_tracks_prompt: bool,
vocal_track_prompt_path: str,
instrumental_track_prompt_path: str,
use_audio_prompt: bool,
audio_prompt_path: str,
genres: str,
lyrics: str,
run_n_segments: int,
max_new_tokens: int,
prompt_start_time: int,
prompt_end_time: int,
sample_settings: SampleSettings,
) -> torch.Tensor:
lyrics, prompt_texts = self.get_prompt_texts(genres, lyrics)
run_n_segments = min(run_n_segments, len(lyrics))
for i in tqdm(range(run_n_segments)):
# Get prompt
if i == 0:
prompt_ids = self.get_first_segment_prompt(
prompt_texts[1],
prompt_texts[0],
use_dual_tracks_prompt,
vocal_track_prompt_path,
instrumental_track_prompt_path,
use_audio_prompt,
audio_prompt_path,
prompt_start_time,
prompt_end_time,
)
else:
prompt_ids = self.get_segment_prompt(prompt_texts[i + 1])
prompt_ids = torch.as_tensor(prompt_ids).unsqueeze(0).to(self.device)
input_ids = torch.cat([raw_output, prompt_ids], dim=1) if i > 0 else prompt_ids
# Use window slicing in case output sequence exceeds the context of model
max_context = self.cache_size - max_new_tokens - 1
if input_ids.shape[-1] > max_context:
print(f"Section {i}: output length {input_ids.shape[-1]} exceeding context length {max_context}, " f"dropping early segment(s) from prompt.")
input_ids = self.shorten_input(input_ids, max_context)
past_key_values = StaticCache(
self.model.config, max_batch_size=1, max_cache_len=input_ids.shape[-1] + max_new_tokens, device=self.model.device, dtype=self.model.dtype
)
processors = LogitsProcessorList([BlockTokenRangeProcessor(0, 32002), BlockTokenRangeProcessor(32016, 32016)])
output_seq = self.model.generate(
input_ids=input_ids,
max_new_tokens=max_new_tokens,
min_new_tokens=100,
do_sample=True,
top_p=sample_settings.top_p,
temperature=sample_settings.temperature,
repetition_penalty=sample_settings.repetition_penalty,
eos_token_id=self.mmtokenizer.eoa,
pad_token_id=self.mmtokenizer.eoa,
logits_processor=processors,
guidance_scale=sample_settings.guidance_scale_seg0 if i == 0 else sample_settings.guidance_scale,
past_key_values=past_key_values,
)
if output_seq[0][-1].item() != self.mmtokenizer.eoa:
tensor_eoa = torch.tensor([[self.mmtokenizer.eoa]], dtype=torch.long, device=output_seq.device)
output_seq = torch.cat((output_seq, tensor_eoa), dim=1)
if i > 0:
raw_output = torch.cat([raw_output, prompt_ids, output_seq[:, input_ids.shape[-1] :]], dim=1)
else:
raw_output = output_seq
return raw_output
class Stage1Pipeline_EXL2(Stage1Pipeline):
def __init__(self, model_path: str, device: torch.device, cache_size: int, cache_mode: str, **kwargs):
super().__init__(device, **kwargs)
assert device != "cpu", "ExLlamaV2 does not support CPU inference."
# Load EXL2 model
device_idx = self.device.index
gpu_split = [0] * torch.cuda.device_count()
gpu_split[device_idx] = 9999
exl2_config = ExLlamaV2Config(model_path)
exl2_config.no_sdpa = True # TODO: Figure out why SDPA slows to a crawl when given custom attn mask
self.model = ExLlamaV2(exl2_config)
self.model.load(gpu_split)
# Load tokenizer (only needed for vocab size in disallow_tokens)
self.tokenizer = ExLlamaV2Tokenizer(exl2_config)
# Define cache
self.cache_size = cache_size
self.cache_mode = get_cache_class(cache_mode)
# TODO: Output layer could be trimmed here to avoid masking out the first 32k tokens during generation
def generate(
self,
use_dual_tracks_prompt: bool,
vocal_track_prompt_path: str,
instrumental_track_prompt_path: str,
use_audio_prompt: bool,
audio_prompt_path: str,
genres: str,
lyrics: str,
run_n_segments: int,
max_new_tokens: int,
prompt_start_time: int,
prompt_end_time: int,
sample_settings: SampleSettings,
) -> torch.Tensor:
if sample_settings.guidance_scale_seg0 is None:
bsz = 1
cfg = False
position_offsets = None
input_mask = None
else:
bsz = 2
cfg = True
lyrics, prompt_texts = self.get_prompt_texts(genres, lyrics)
run_n_segments = min(run_n_segments, len(lyrics))
# Cache for the whole output sequence
cache = self.cache_mode(self.model, batch_size=bsz, max_seq_len=self.cache_size)
# Collect output here
seq = torch.empty((bsz, 0), dtype=torch.long)
# Sample settings
gen_settings = ExLlamaV2Sampler.Settings(
top_k=0, top_p=sample_settings.top_p, token_repetition_penalty=sample_settings.repetition_penalty, temperature=sample_settings.temperature
)
gen_settings.allow_tokens(self.tokenizer, [32002] + list(range(45334, 56722)))
# RNG for sampling, could seed here
rng = random.Random()
for i in tqdm(range(run_n_segments)):
# Get prompt for this segment
if i == 0:
prompt_ids = self.get_first_segment_prompt(
prompt_texts[1],
prompt_texts[0],
use_dual_tracks_prompt,
vocal_track_prompt_path,
instrumental_track_prompt_path,
use_audio_prompt,
audio_prompt_path,
prompt_start_time,
prompt_end_time,
)
else:
prompt_ids = self.get_segment_prompt(prompt_texts[i + 1])
prompt_ids = torch.tensor([prompt_ids] * bsz, dtype=torch.long)
# Accept prompt tokens
seq = torch.cat((seq, prompt_ids), dim=-1)
# Use window slicing in case output sequence exceeds the context of model
max_context = self.cache_size - max_new_tokens - 1
if seq.shape[-1] > max_context:
print(f"Section {i}: output length {seq.shape[-1]} exceeding context length {max_context}, " f"dropping early segment(s) from prompt.")
cache.current_seq_len = 0
full_ids = self.shorten_input(seq, max_context)
incremental_ids = full_ids
else:
full_ids = seq
incremental_ids = prompt_ids
# For the unconditional context, mask out all but the last token
if cfg:
mask_len = full_ids.shape[-1] - 1
full_mask = torch.zeros((2, cache.max_seq_len), dtype=torch.half, device=self.device)
full_mask[1, :mask_len] = -65504.0
position_offsets = torch.tensor([[0], [-mask_len]], dtype=torch.int)
input_mask = full_mask[:, : full_ids.shape[-1]]
# Forward prompt
logits = self.model.forward(incremental_ids[:, :], cache=cache, input_mask=input_mask, position_offsets=position_offsets, last_id_only=True)
# Generate until EOS or max_new_tokens
for new_tokens in tqdm(range(max_new_tokens)):
# Transformers-equiv. CFG
if cfg:
cfg_scale = sample_settings.guidance_scale_seg0 if i == 0 else sample_settings.guidance_scale
logits = logits.float()
logits = F.log_softmax(logits, dim=-1)
logits = cfg_scale * logits[0] + (1 - cfg_scale) * logits[1]
logits = logits.unsqueeze(0)
# Sample
logits = logits.float().cpu()
sample, _, _, _, _ = ExLlamaV2Sampler.sample(logits, gen_settings, full_ids[:1], rng.random(), self.tokenizer)
if cfg:
sample = torch.cat((sample, sample), dim=0)
# Accept token
full_ids = torch.cat((full_ids, sample), dim=-1)
seq = torch.cat((seq, sample), dim=-1)
# Get next logits (update cache even if sample is EOA and we don't need next logits)
if cfg:
input_mask = full_mask[:, : full_ids.shape[-1]]
logits = self.model.forward(sample, cache=cache, input_mask=input_mask, position_offsets=position_offsets)
# End on EOA
if sample[0].item() == self.mmtokenizer.eoa:
break
# Make sure sequence ends with EOA if we reached max_new_tokens
else:
sample = torch.tensor([[self.mmtokenizer.eoa]] * bsz, dtype=torch.long)
seq = torch.cat((seq, sample), dim=-1)
# Update cache with forced token
self.model.forward(sample, cache=cache)
raw_output = seq[:1, :]
return raw_output
def main():
args = parser.parse_args()
if args.use_audio_prompt and not args.audio_prompt_path:
raise FileNotFoundError("Please offer audio prompt filepath using '--audio_prompt_path', when you enable 'use_audio_prompt'!")
if args.use_dual_tracks_prompt and not args.vocal_track_prompt_path and not args.instrumental_track_prompt_path:
raise FileNotFoundError(
"Please offer dual tracks prompt filepath using '--vocal_track_prompt_path' and '--inst_decoder_path', when you enable '--use_dual_tracks_prompt'!"
)
if args.seed is not None:
seed_everything(args.seed)
device = torch.device(f"cuda:{args.cuda_idx}" if torch.cuda.is_available() else "cpu")
with open(args.genre_txt) as f:
genres = f.read().strip()
with open(args.lyrics_txt) as f:
lyrics = f.read().strip()
if args.stage1_use_exl2:
pipeline = Stage1Pipeline_EXL2(
model_path=args.stage1_model,
device=device,
basic_model_config=args.basic_model_config,
resume_path=args.resume_path,
cache_size=args.stage1_cache_size,
cache_mode=args.stage1_cache_mode,
)
else:
pipeline = Stage1Pipeline_HF(
model_path=args.stage1_model,
device=device,
basic_model_config=args.basic_model_config,
resume_path=args.resume_path,
cache_size=args.stage1_cache_size,
)
# Load tokenizer and models
raw_output = pipeline.generate(
use_dual_tracks_prompt=args.use_dual_tracks_prompt,
vocal_track_prompt_path=args.vocal_track_prompt_path,
instrumental_track_prompt_path=args.instrumental_track_prompt_path,
use_audio_prompt=args.use_audio_prompt,
audio_prompt_path=args.audio_prompt_path,
genres=genres,
lyrics=lyrics,
run_n_segments=args.run_n_segments,
max_new_tokens=args.max_new_tokens,
prompt_start_time=args.prompt_start_time,
prompt_end_time=args.prompt_end_time,
sample_settings=SampleSettings(use_guidance=not args.stage1_no_guidance, repetition_penalty=args.repetition_penalty),
)
# Save result
pipeline.save(raw_output, args.output_dir, args.use_audio_prompt, args.use_dual_tracks_prompt)
# if __name__ == "__main__":
# # enable inference mode globally
# torch.autograd.grad_mode._enter_inference_mode(True)
# torch.autograd.set_grad_enabled(False)
# main()
#Full lyrics: ["Generate music from the given lyrics segment by segment.\n[Genre] inspiring female uplifting pop airy vocal electronic bright vocal vocal.\n[verse]\nStaring at the sunset, colors paint the sky.\nThoughts of you keep swirling, can't deny.\nI know I let you down, I made mistakes.\nBut I'm here to mend the heart I didn't break.\n\n\n[chorus]\nEvery road you take, I'll be one step behind.\nEvery dream you chase, I'm reaching for the light.\nYou can't fight this feeling now.\nI won't back down.\nYou know you can't deny it now.\n I won't back down\n\n\n[verse]\nThey might say I'm foolish, chasing after you.\nBut they don't feel this love the way we do.\nMy heart beats only for you, can't you see?\nI won't let you slip away from me.\n\n\n[chorus]\nEvery road you take, I'll be one step behind.\nEvery dream you chase, I'm reaching for the light.\nYou can't fight this feeling now.\nI won't back down.\nYou know you can't deny it now.\n I won't back down\n\n\n[bridge]\nNo, I won't back down, won't turn around.\nUntil you're back where you belong.\nI'll cross the oceans wide, stand by your side.\nTogether we are strong.\n\n\n[outro]\nEvery road you take, I'll be one step behind.\nEvery dream you chase, love's the tie that binds.\nYou can't fight this feeling now.\nI won't back down.\n\n"]
-365
View File
@@ -1,365 +0,0 @@
import copy
import math
import os
from collections import Counter
import numpy as np
import torch
from .codecmanipulator import CodecManipulator
from .common import BlockTokenRangeProcessor, seed_everything, get_cache_class
from exllamav2 import ExLlamaV2, ExLlamaV2Config, ExLlamaV2Tokenizer
from .mmtokenizer import _MMSentencePieceTokenizer
from tqdm import tqdm
from transformers import AutoModelForCausalLM, LogitsProcessorList
from transformers.cache_utils import StaticCache
import gc
import folder_paths
def align(n, m):
return ((n + m - 1) // m) * m
def split_bsz(bsz, maxbsz):
n_sub_batches = math.ceil(bsz / maxbsz)
base_size = bsz // n_sub_batches
remainder = bsz % n_sub_batches
sub_batch_sizes = [base_size + 1] * remainder + [base_size] * (n_sub_batches - remainder)
indices = []
start = 0
for size in sub_batch_sizes:
end = start + size
indices.append((start, end))
start = end
return indices
class Stage2Pipeline:
def __init__(self, device: torch.device):
self.device = device
self.codec_tool = CodecManipulator("xcodec", 0, 1)
self.codec_tool_stage2 = CodecManipulator("xcodec", 0, 8)
# Load tokenizer
self.mmtokenizer = _MMSentencePieceTokenizer(os.path.join(folder_paths.base_path,"custom_nodes/ComfyUI_YuE/inference/mm_tokenizer_v0.2_hf/tokenizer.model"))
def get_codec_ids(self, prompt: np.array):
codec_ids = self.codec_tool.unflatten(prompt, n_quantizer=1)
codec_ids = self.codec_tool.offset_tok_ids(
codec_ids, global_offset=self.codec_tool.global_offset, codebook_size=self.codec_tool.codebook_size, num_codebooks=self.codec_tool.num_codebooks
).astype(np.int32)
return codec_ids
def fix_output(self, output):
# Fix invalid codes (a dirty solution, which may harm the quality of audio)
# We are trying to find better one
fixed_output = copy.deepcopy(output)
for i, line in enumerate(output):
for j, element in enumerate(line):
if element < 0 or element > 1023:
counter = Counter(line)
most_frequant = sorted(counter.items(), key=lambda x: x[1], reverse=True)[0][0]
fixed_output[i, j] = most_frequant
return fixed_output
def save(self, output_dir: str, outputs):
for output_name, output in outputs.items():
# save output
stage2_output_dir = os.path.join(output_dir, "stage2")
os.makedirs(stage2_output_dir, exist_ok=True)
output_filename = os.path.join(stage2_output_dir, output_name)
np.save(output_filename, output)
def get_stage1_prompt(self, output_dir: str, output_name: str):
stage1_output_dir = os.path.join(output_dir, "stage1")
prompt = np.load(os.path.join(stage1_output_dir, output_name)).astype(np.int32)
return prompt
def prepare_prompt_batch(self, prompt: np.array, batch_size: int):
codec_ids = self.get_codec_ids(prompt)
# Prepare prompt_ids based on batch size or single input
if batch_size > 1:
codec_list = []
for i in range(batch_size):
idx_begin = i * 300
idx_end = (i + 1) * 300
codec_list.append(codec_ids[:, idx_begin:idx_end])
codec_ids = np.concatenate(codec_list, axis=0)
prompt_ids = np.concatenate(
[np.tile([self.mmtokenizer.soa, self.mmtokenizer.stage_1], (batch_size, 1)), codec_ids, np.tile([self.mmtokenizer.stage_2], (batch_size, 1))],
axis=1,
)
else:
prompt_ids = np.concatenate(
[
np.array([self.mmtokenizer.soa, self.mmtokenizer.stage_1]),
codec_ids.flatten(), # Flatten the 2D array to 1D
np.array([self.mmtokenizer.stage_2]),
]
).astype(np.int32)
prompt_ids = prompt_ids[np.newaxis, ...]
codec_ids = torch.as_tensor(codec_ids, dtype=torch.long)
prompt_ids = torch.as_tensor(prompt_ids, dtype=torch.long)
return codec_ids, prompt_ids
class Stage2Pipeline_HF(Stage2Pipeline):
def __init__(self, model_path: str, device: torch.device, batch_size: int):
super().__init__(device)
self.batch_size = batch_size
self.model = AutoModelForCausalLM.from_pretrained(model_path, torch_dtype=torch.float16, attn_implementation="sdpa")
self.model.to(device)
self.model.eval()
if torch.__version__ >= "2.0.0":
self.model = torch.compile(self.model)
def generate_batch(self, prompt: np.array, batch_size: int):
codec_ids, prompt_ids = self.prepare_prompt_batch(prompt, batch_size)
len_prompt = prompt_ids.shape[-1]
# Teacher forcing generate loop
codec_ids = codec_ids.to(self.device)
prompt_ids = prompt_ids.to(self.device)
block_list = LogitsProcessorList([BlockTokenRangeProcessor(0, 46358), BlockTokenRangeProcessor(53526, self.mmtokenizer.vocab_size)])
past_key_values = StaticCache(
self.model.config,
max_batch_size=batch_size,
max_cache_len=prompt_ids.shape[1] + codec_ids.shape[1] * 8,
device=self.model.device,
dtype=self.model.dtype,
)
for frames_idx in range(codec_ids.shape[1]):
cb0 = codec_ids[:, frames_idx : frames_idx + 1]
prompt_ids = torch.cat([prompt_ids, cb0], dim=1)
input_ids = prompt_ids
stage2_output = self.model.generate(
input_ids=input_ids,
min_new_tokens=7,
max_new_tokens=7,
eos_token_id=self.mmtokenizer.eoa,
pad_token_id=self.mmtokenizer.eoa,
logits_processor=block_list,
past_key_values=past_key_values,
)
assert stage2_output.shape[1] - prompt_ids.shape[1] == 7, f"output new tokens={stage2_output.shape[1]-prompt_ids.shape[1]}"
prompt_ids = stage2_output
# Return output based on batch size
if batch_size > 1:
output = prompt_ids.cpu().numpy()[:, len_prompt:]
output_list = [output[i] for i in range(batch_size)]
output = np.concatenate(output_list, axis=0)
else:
output = prompt_ids[0].cpu().numpy()[len_prompt:]
return output
def generate(self, output_dir: str) -> dict[str, np.array]:
outputs = {}
for output_name in tqdm(["vtrack.npy", "itrack.npy"]):
# Load the prompt
prompt = self.get_stage1_prompt(output_dir, output_name)
# Only accept 6s segments
output_duration = prompt.shape[-1] // 50 // 6 * 6
num_batch = output_duration // 6
if num_batch <= self.batch_size:
# If num_batch is less than or equal to batch_size, we can infer the entire prompt at once
output = self.generate_batch(prompt[:, : output_duration * 50], batch_size=num_batch)
else:
# If num_batch is greater than batch_size, process in chunks of batch_size
segments = []
num_segments = (num_batch // self.batch_size) + (1 if num_batch % self.batch_size != 0 else 0)
for seg in range(num_segments):
start_idx = seg * self.batch_size * 300
# Ensure the end_idx does not exceed the available length
end_idx = min((seg + 1) * self.batch_size * 300, output_duration * 50) # Adjust the last segment
current_batch_size = self.batch_size if seg != num_segments - 1 or num_batch % self.batch_size == 0 else num_batch % self.batch_size
segment = self.generate_batch(prompt[:, start_idx:end_idx], batch_size=current_batch_size)
segments.append(segment)
# Concatenate all the segments
output = np.concatenate(segments, axis=0)
# Process the ending part of the prompt
if output_duration * 50 != prompt.shape[-1]:
ending = self.generate_batch(prompt[:, output_duration * 50 :], batch_size=1)
output = np.concatenate([output, ending], axis=0)
output = self.codec_tool_stage2.ids2npy(output)
output = self.fix_output(output)
outputs[output_name] = output
return outputs
class Stage2Pipeline_EXL2(Stage2Pipeline):
def __init__(self, model_path: str, device: torch.device, cache_size: int, cache_mode: str):
super().__init__(device)
self.cache_size = cache_size
assert device != "cpu", "ExLlamaV2 does not support CPU inference."
# Load EXL2 model
device_idx = self.device.index
gpu_split = [0] * torch.cuda.device_count()
gpu_split[device_idx] = 9999
exl2_config = ExLlamaV2Config(model_path)
self.model = ExLlamaV2(exl2_config)
self.model.load(gpu_split)
# Move embedding layer to GPU to avoid CPU sync during argmax gen loop
self.model.modules[0].device_idx = self.model.modules[1].device_idx
self.model.modules[0].reload()
# Load tokenizer (only needed for vocab size in disallow_tokens)
self.tokenizer = ExLlamaV2Tokenizer(exl2_config)
# Define cache
self.cache_mode = get_cache_class(cache_mode)
def generate(self, output_dir: str) -> dict[str, np.array]:
parts = ["vtrack.npy", "itrack.npy"]
full_batch = []
# Collect up to 300 token (6s) segments for all parts
for output_idx, output_name in tqdm(enumerate(parts)):
prompt = self.get_stage1_prompt(output_dir, output_name)
prompt = self.get_codec_ids(prompt)
prompt = torch.as_tensor(prompt, dtype=torch.long)
segs = torch.split(prompt, 300, dim=-1)
for seg_idx, seg in enumerate(segs):
seg_len = seg.shape[-1]
full_batch.append((seg_len, seg_idx, output_idx, seg))
# Prepare segments
prefix = torch.tensor([[self.mmtokenizer.soa, self.mmtokenizer.stage_1]], dtype=torch.long)
suffix = torch.tensor([[self.mmtokenizer.stage_2]], dtype=torch.long)
for i in range(len(full_batch)):
seg_len, seg_idx, output_idx, codec_ids = full_batch[i]
prompt_ids = torch.cat((prefix, codec_ids, suffix), dim=-1)
full_batch[i] = (seg_len, seg_idx, output_idx, codec_ids, prompt_ids)
# Group prompts by length
batches = {}
for seq in full_batch:
if not seq[0] in batches:
batches[seq[0]] = []
batches[seq[0]].append(seq)
# Split into on minibatches
split_batch = []
for idx, (seg_len, batch) in enumerate(batches.items()):
b_seg_order = [b[1] for b in batch]
b_part_order = [b[2] for b in batch]
b_codec_ids = torch.cat([b[3] for b in batch], dim=0)
b_prompt_ids = torch.cat([b[4] for b in batch], dim=0)
max_bsz = self.cache_size // align(b_prompt_ids.shape[1] + b_codec_ids.shape[1] * 8, 32)
assert max_bsz > 0
for a, b in split_bsz(b_prompt_ids.shape[0], max_bsz):
split_batch.append((b_seg_order[a:b], b_part_order[a:b], b_codec_ids[a:b], b_prompt_ids[a:b]))
# Inference
output_parts = []
for _ in parts:
output_parts.append([])
for seg_order, part_order, codec_ids, prompt_ids in tqdm(split_batch):
codec_ids = codec_ids.to(self.device)
prompt_ids = prompt_ids.to(self.device)
batch_size, len_prompt = prompt_ids.shape
cache = self.cache_mode(self.model, batch_size=batch_size, max_seq_len=align(prompt_ids.shape[1] + codec_ids.shape[1] * 8, 32))
output_ids = torch.empty((batch_size, 0), dtype=torch.long, device=self.device)
for frames_idx in tqdm(range(codec_ids.shape[1])):
cb0 = codec_ids[:, frames_idx : frames_idx + 1]
# Append the initial prompt to the first codec frame
if frames_idx == 0:
cb0 = torch.cat([prompt_ids, cb0], dim=-1)
# Forward prompt
output_ids = torch.cat((output_ids, cb0), dim=-1)
logits = self.model.forward(cb0, cache=cache, last_id_only=True)
for i in range(7):
# Slice logits instead of biasing start and end of distribution
first_logit = 46358
last_logit = 53526
logits = logits[:, :, first_logit:last_logit]
# Greedy sampling
sample = logits.argmax(dim=-1) + first_logit
output_ids = torch.cat((output_ids, sample), dim=-1)
# TODO: Here, original asserts that we didn't sample mmtokenizer.eoa (can we just mask it out?)
# Forward sample
logits = self.model.forward(sample, cache=cache)
# Trim prompt
output_ids = output_ids[:, len_prompt:]
# Split outputs
for i in range(batch_size):
output_parts[part_order[i]].append((seg_order[i], output_ids[i : i + 1, :]))
# Release cache tensors
del cache
torch.cuda.empty_cache()
gc.collect()
# Unshuffle and recombine output parts
output = {}
for i, p in enumerate(output_parts):
p = sorted(p, key=lambda x: x[0])
part_o = torch.cat([pp[1] for pp in p], dim=-1).flatten().cpu().numpy()
part_o = self.codec_tool_stage2.ids2npy(part_o)
part_o = self.fix_output(part_o)
output[parts[i]] = part_o
return output
def main():
args = parser.parse_args()
if args.seed is not None:
seed_everything(args.seed)
device = torch.device(f"cuda:{args.cuda_idx}" if torch.cuda.is_available() else "cpu")
if args.stage2_use_exl2:
pipeline = Stage2Pipeline_EXL2(model_path=args.stage2_model, device=device, cache_size=args.stage2_cache_size, cache_mode=args.stage2_cache_mode)
pass
else:
pipeline = Stage2Pipeline_HF(model_path=args.stage2_model, device=device, batch_size=args.stage2_batch_size)
outputs = pipeline.generate(output_dir=args.output_dir)
pipeline.save(output_dir=args.output_dir, outputs=outputs)
if __name__ == "__main__":
# enable inference mode globally
torch.autograd.grad_mode._enter_inference_mode(True)
torch.autograd.set_grad_enabled(False)
main()
Binary file not shown.
-367
View File
@@ -1,367 +0,0 @@
from abc import ABC
from abc import abstractmethod
class AbstractTokenizer(ABC):
"""Abstract class for tokenizer."""
def __init__(self, name):
self.name = name
super().__init__()
@property
@abstractmethod
def vocab_size(self):
pass
@property
@abstractmethod
def vocab(self):
"""Dictionary from vocab text token to id token."""
pass
@property
@abstractmethod
def inv_vocab(self):
"""Dictionary from vocab id token to text token."""
pass
@abstractmethod
def tokenize(self, text):
pass
def detokenize(self, token_ids):
raise NotImplementedError('detokenizer is not implemented for {} '
'tokenizer'.format(self.name))
@property
def cls(self):
raise NotImplementedError('CLS is not provided for {} '
'tokenizer'.format(self.name))
@property
def sep(self):
raise NotImplementedError('SEP is not provided for {} '
'tokenizer'.format(self.name))
@property
def pad(self):
raise NotImplementedError('PAD is not provided for {} '
'tokenizer'.format(self.name))
@property
def eod(self):
raise NotImplementedError('EOD is not provided for {} '
'tokenizer'.format(self.name))
@property
def mask(self):
raise NotImplementedError('MASK is not provided for {} '
'tokenizer'.format(self.name))
class _SentencePieceTokenizer(AbstractTokenizer):
"""SentencePieceTokenizer-Megatron wrapper"""
def __init__(self, model_file, vocab_extra_ids=0):
name = 'SentencePieceTokenizer'
super().__init__(name)
import sentencepiece
self.tokenizer = sentencepiece.SentencePieceProcessor(model_file=model_file)
self._initalize(vocab_extra_ids)
def _populate_vocab(self):
self._vocab = {}
self._inv_vocab = {}
for i in range(len(self.tokenizer)):
t = self.tokenizer.id_to_piece(i)
self._inv_vocab[i] = t
self._vocab[t] = i
def _initalize(self, vocab_extra_ids):
self._populate_vocab()
self._special_tokens = {}
self._inv_special_tokens = {}
self._t5_tokens = []
def _add_special_token(t):
if t not in self._vocab:
next_id = len(self._vocab)
self._vocab[t] = next_id
self._inv_vocab[next_id] = t
self._special_tokens[t] = self._vocab[t]
self._inv_special_tokens[self._vocab[t]] = t
_add_special_token('<CLS>')
self._cls_id = self._vocab['<CLS>']
_add_special_token('<SEP>')
self._sep_id = self._vocab['<SEP>']
_add_special_token('<EOD>')
self._eod_id = self._vocab['<EOD>']
_add_special_token('<MASK>')
self._mask_id = self._vocab['<MASK>']
pad_id = self.tokenizer.pad_id()
try:
pad_token = self.tokenizer.id_to_piece(pad_id)
except IndexError:
pad_token = '<PAD>'
_add_special_token(pad_token)
self._pad_id = self._vocab[pad_token]
bos_id = self.tokenizer.bos_id()
try:
bos_token = self.tokenizer.id_to_piece(bos_id)
except IndexError:
bos_token = '<BOS>'
_add_special_token(bos_token)
self._bos_id = self._vocab[bos_token]
eos_id = self.tokenizer.eos_id()
try:
eos_token = self.tokenizer.id_to_piece(eos_id)
except IndexError:
eos_token = '<EOS>'
_add_special_token(eos_token)
self._eos_id = self._vocab[eos_token]
for i in range(vocab_extra_ids):
t = "<extra_id_{}>".format(i)
_add_special_token(t)
self._t5_tokens += [t]
@property
def vocab_size(self):
return len(self._vocab)
@property
def vocab(self):
return self._vocab
@property
def inv_vocab(self):
return self._inv_vocab
@property
def decoder(self):
return self._inv_vocab
@property
def encoder(self):
return self._vocab
# From:
# https://github.com/NVIDIA/NeMo/blob/c8fa217e811d60d11d014827c7f3845ff6c99ae7/nemo/collections/common/tokenizers/sentencepiece_tokenizer.py#L89
def tokenize(self, text):
ids = []
idx = 0
while 1:
indices = {}
for token in self._special_tokens:
try:
indices[token] = text[idx:].index(token)
except ValueError:
continue
if len(indices) == 0:
break
next_token = min(indices, key=indices.get)
next_idx = idx + indices[next_token]
ids.extend(self.tokenizer.encode_as_ids(text[idx:next_idx]))
ids.append(self._special_tokens[next_token])
idx = next_idx + len(next_token)
ids.extend(self.tokenizer.encode_as_ids(text[idx:]))
return ids
# From:
# https://github.com/NVIDIA/NeMo/blob/c8fa217e811d60d11d014827c7f3845ff6c99ae7/nemo/collections/common/tokenizers/sentencepiece_tokenizer.py#L125
def detokenize(self, ids):
text = ""
last_i = 0
for i, id in enumerate(ids):
if id in self._inv_special_tokens:
text += self.tokenizer.decode_ids(ids[last_i:i]) + " "
text += self._inv_special_tokens[id] + " "
last_i = i + 1
text += self.tokenizer.decode_ids(ids[last_i:])
return text
@property
def cls(self):
return self._cls_id
@property
def sep(self):
return self._sep_id
@property
def pad(self):
return self._pad_id
@property
def bos_token_id(self):
return self._bos_id
@property
def bos(self):
return self._bos_id
@property
def eod(self):
return self._eod_id
@property
def eos_token_id(self):
return self._eos_id
@property
def eos(self):
return self._eos_id
@property
def mask(self):
return self._mask_id
@property
def additional_special_tokens_ids(self):
return [self.vocab[k] for k in self._t5_tokens]
class _MMSentencePieceTokenizer(_SentencePieceTokenizer):
"""SentencePieceTokenizer-Megatron wrapper"""
def __init__(self, model_file, vocab_extra_ids=0):
super().__init__(model_file, vocab_extra_ids)
def _initalize(self, vocab_extra_ids):
self._populate_vocab()
self._special_tokens = {}
self._inv_special_tokens = {}
self._t5_tokens = []
def _add_special_token(t):
if t not in self._vocab:
next_id = len(self._vocab)
self._vocab[t] = next_id
self._inv_vocab[next_id] = t
self._special_tokens[t] = self._vocab[t]
self._inv_special_tokens[self._vocab[t]] = t
_add_special_token('<CLS>')
self._cls_id = self._vocab['<CLS>']
_add_special_token('<SEP>')
self._sep_id = self._vocab['<SEP>']
_add_special_token('<EOD>')
self._eod_id = self._vocab['<EOD>']
_add_special_token('<MASK>')
self._mask_id = self._vocab['<MASK>']
_add_special_token('<SOA>')
self._soa_id = self._vocab['<SOA>']
_add_special_token('<EOA>')
self._eoa_id = self._vocab['<EOA>']
_add_special_token('<SOV>')
self._sov_id = self._vocab['<SOV>']
_add_special_token('<EOV>')
self._eov_id = self._vocab['<EOV>']
_add_special_token('<SOI>')
self._soi_id = self._vocab['<SOI>']
_add_special_token('<EOI>')
self._eoi_id = self._vocab['<EOI>']
_add_special_token('<s_local>')
self._s_local_id = self._vocab['<s_local>']
_add_special_token('<e_local>')
self._e_local_id = self._vocab['<e_local>']
_add_special_token('<s_global>')
self._s_global_id = self._vocab['<s_global>']
_add_special_token('<e_global>')
self._e_global_id = self._vocab['<e_global>']
_add_special_token('<stage_1>')
self._stage_1_id = self._vocab['<stage_1>']
_add_special_token('<stage_2>')
self._stage_2_id = self._vocab['<stage_2>']
pad_id = self.tokenizer.pad_id()
try:
pad_token = self.tokenizer.id_to_piece(pad_id)
except IndexError:
pad_token = '<PAD>'
_add_special_token(pad_token)
self._pad_id = self._vocab[pad_token]
bos_id = self.tokenizer.bos_id()
try:
bos_token = self.tokenizer.id_to_piece(bos_id)
except IndexError:
bos_token = '<BOS>'
_add_special_token(bos_token)
self._bos_id = self._vocab[bos_token]
eos_id = self.tokenizer.eos_id()
try:
eos_token = self.tokenizer.id_to_piece(eos_id)
except IndexError:
eos_token = '<EOS>'
_add_special_token(eos_token)
self._eos_id = self._vocab[eos_token]
for i in range(vocab_extra_ids):
t = "<extra_id_{}>".format(i)
_add_special_token(t)
self._t5_tokens += [t]
@property
def soa(self):
return self._soa_id
@property
def eoa(self):
return self._eoa_id
@property
def sov(self):
return self._sov_id
@property
def eov(self):
return self._eov_id
@property
def soi(self):
return self._soi_id
@property
def eoi(self):
return self._eoi_id
@property
def s_local(self):
return self._s_local_id
@property
def e_local(self):
return self._e_local_id
@property
def s_global(self):
return self._s_global_id
@property
def e_global(self):
return self._e_global_id
@property
def stage_1(self):
return self._stage_1_id
@property
def stage_2(self):
return self._stage_2_id
-3
View File
@@ -1,3 +0,0 @@
---
license: apache-2.0
---
@@ -1,428 +0,0 @@
MIT License
Copyright (c) ByteDance, Inc. and its affiliates.
Copyright (c) Chutong Meng
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.
Attribution-NonCommercial 4.0 International
=======================================================================
Creative Commons Corporation ("Creative Commons") is not a law firm and
does not provide legal services or legal advice. Distribution of
Creative Commons public licenses does not create a lawyer-client or
other relationship. Creative Commons makes its licenses and related
information available on an "as-is" basis. Creative Commons gives no
warranties regarding its licenses, any material licensed under their
terms and conditions, or any related information. Creative Commons
disclaims all liability for damages resulting from their use to the
fullest extent possible.
Using Creative Commons Public Licenses
Creative Commons public licenses provide a standard set of terms and
conditions that creators and other rights holders may use to share
original works of authorship and other material subject to copyright
and certain other rights specified in the public license below. The
following considerations are for informational purposes only, are not
exhaustive, and do not form part of our licenses.
Considerations for licensors: Our public licenses are
intended for use by those authorized to give the public
permission to use material in ways otherwise restricted by
copyright and certain other rights. Our licenses are
irrevocable. Licensors should read and understand the terms
and conditions of the license they choose before applying it.
Licensors should also secure all rights necessary before
applying our licenses so that the public can reuse the
material as expected. Licensors should clearly mark any
material not subject to the license. This includes other CC-
licensed material, or material used under an exception or
limitation to copyright. More considerations for licensors:
wiki.creativecommons.org/Considerations_for_licensors
Considerations for the public: By using one of our public
licenses, a licensor grants the public permission to use the
licensed material under specified terms and conditions. If
the licensor's permission is not necessary for any reason--for
example, because of any applicable exception or limitation to
copyright--then that use is not regulated by the license. Our
licenses grant only permissions under copyright and certain
other rights that a licensor has authority to grant. Use of
the licensed material may still be restricted for other
reasons, including because others have copyright or other
rights in the material. A licensor may make special requests,
such as asking that all changes be marked or described.
Although not required by our licenses, you are encouraged to
respect those requests where reasonable. More_considerations
for the public:
wiki.creativecommons.org/Considerations_for_licensees
=======================================================================
Creative Commons Attribution-NonCommercial 4.0 International Public
License
By exercising the Licensed Rights (defined below), You accept and agree
to be bound by the terms and conditions of this Creative Commons
Attribution-NonCommercial 4.0 International Public License ("Public
License"). To the extent this Public License may be interpreted as a
contract, You are granted the Licensed Rights in consideration of Your
acceptance of these terms and conditions, and the Licensor grants You
such rights in consideration of benefits the Licensor receives from
making the Licensed Material available under these terms and
conditions.
Section 1 -- Definitions.
a. Adapted Material means material subject to Copyright and Similar
Rights that is derived from or based upon the Licensed Material
and in which the Licensed Material is translated, altered,
arranged, transformed, or otherwise modified in a manner requiring
permission under the Copyright and Similar Rights held by the
Licensor. For purposes of this Public License, where the Licensed
Material is a musical work, performance, or sound recording,
Adapted Material is always produced where the Licensed Material is
synched in timed relation with a moving image.
b. Adapter's License means the license You apply to Your Copyright
and Similar Rights in Your contributions to Adapted Material in
accordance with the terms and conditions of this Public License.
c. Copyright and Similar Rights means copyright and/or similar rights
closely related to copyright including, without limitation,
performance, broadcast, sound recording, and Sui Generis Database
Rights, without regard to how the rights are labeled or
categorized. For purposes of this Public License, the rights
specified in Section 2(b)(1)-(2) are not Copyright and Similar
Rights.
d. Effective Technological Measures means those measures that, in the
absence of proper authority, may not be circumvented under laws
fulfilling obligations under Article 11 of the WIPO Copyright
Treaty adopted on December 20, 1996, and/or similar international
agreements.
e. Exceptions and Limitations means fair use, fair dealing, and/or
any other exception or limitation to Copyright and Similar Rights
that applies to Your use of the Licensed Material.
f. Licensed Material means the artistic or literary work, database,
or other material to which the Licensor applied this Public
License.
g. Licensed Rights means the rights granted to You subject to the
terms and conditions of this Public License, which are limited to
all Copyright and Similar Rights that apply to Your use of the
Licensed Material and that the Licensor has authority to license.
h. Licensor means the individual(s) or entity(ies) granting rights
under this Public License.
i. NonCommercial means not primarily intended for or directed towards
commercial advantage or monetary compensation. For purposes of
this Public License, the exchange of the Licensed Material for
other material subject to Copyright and Similar Rights by digital
file-sharing or similar means is NonCommercial provided there is
no payment of monetary compensation in connection with the
exchange.
j. Share means to provide material to the public by any means or
process that requires permission under the Licensed Rights, such
as reproduction, public display, public performance, distribution,
dissemination, communication, or importation, and to make material
available to the public including in ways that members of the
public may access the material from a place and at a time
individually chosen by them.
k. Sui Generis Database Rights means rights other than copyright
resulting from Directive 96/9/EC of the European Parliament and of
the Council of 11 March 1996 on the legal protection of databases,
as amended and/or succeeded, as well as other essentially
equivalent rights anywhere in the world.
l. You means the individual or entity exercising the Licensed Rights
under this Public License. Your has a corresponding meaning.
Section 2 -- Scope.
a. License grant.
1. Subject to the terms and conditions of this Public License,
the Licensor hereby grants You a worldwide, royalty-free,
non-sublicensable, non-exclusive, irrevocable license to
exercise the Licensed Rights in the Licensed Material to:
a. reproduce and Share the Licensed Material, in whole or
in part, for NonCommercial purposes only; and
b. produce, reproduce, and Share Adapted Material for
NonCommercial purposes only.
2. Exceptions and Limitations. For the avoidance of doubt, where
Exceptions and Limitations apply to Your use, this Public
License does not apply, and You do not need to comply with
its terms and conditions.
3. Term. The term of this Public License is specified in Section
6(a).
4. Media and formats; technical modifications allowed. The
Licensor authorizes You to exercise the Licensed Rights in
all media and formats whether now known or hereafter created,
and to make technical modifications necessary to do so. The
Licensor waives and/or agrees not to assert any right or
authority to forbid You from making technical modifications
necessary to exercise the Licensed Rights, including
technical modifications necessary to circumvent Effective
Technological Measures. For purposes of this Public License,
simply making modifications authorized by this Section 2(a)
(4) never produces Adapted Material.
5. Downstream recipients.
a. Offer from the Licensor -- Licensed Material. Every
recipient of the Licensed Material automatically
receives an offer from the Licensor to exercise the
Licensed Rights under the terms and conditions of this
Public License.
b. No downstream restrictions. You may not offer or impose
any additional or different terms or conditions on, or
apply any Effective Technological Measures to, the
Licensed Material if doing so restricts exercise of the
Licensed Rights by any recipient of the Licensed
Material.
6. No endorsement. Nothing in this Public License constitutes or
may be construed as permission to assert or imply that You
are, or that Your use of the Licensed Material is, connected
with, or sponsored, endorsed, or granted official status by,
the Licensor or others designated to receive attribution as
provided in Section 3(a)(1)(A)(i).
b. Other rights.
1. Moral rights, such as the right of integrity, are not
licensed under this Public License, nor are publicity,
privacy, and/or other similar personality rights; however, to
the extent possible, the Licensor waives and/or agrees not to
assert any such rights held by the Licensor to the limited
extent necessary to allow You to exercise the Licensed
Rights, but not otherwise.
2. Patent and trademark rights are not licensed under this
Public License.
3. To the extent possible, the Licensor waives any right to
collect royalties from You for the exercise of the Licensed
Rights, whether directly or through a collecting society
under any voluntary or waivable statutory or compulsory
licensing scheme. In all other cases the Licensor expressly
reserves any right to collect such royalties, including when
the Licensed Material is used other than for NonCommercial
purposes.
Section 3 -- License Conditions.
Your exercise of the Licensed Rights is expressly made subject to the
following conditions.
a. Attribution.
1. If You Share the Licensed Material (including in modified
form), You must:
a. retain the following if it is supplied by the Licensor
with the Licensed Material:
i. identification of the creator(s) of the Licensed
Material and any others designated to receive
attribution, in any reasonable manner requested by
the Licensor (including by pseudonym if
designated);
ii. a copyright notice;
iii. a notice that refers to this Public License;
iv. a notice that refers to the disclaimer of
warranties;
v. a URI or hyperlink to the Licensed Material to the
extent reasonably practicable;
b. indicate if You modified the Licensed Material and
retain an indication of any previous modifications; and
c. indicate the Licensed Material is licensed under this
Public License, and include the text of, or the URI or
hyperlink to, this Public License.
2. You may satisfy the conditions in Section 3(a)(1) in any
reasonable manner based on the medium, means, and context in
which You Share the Licensed Material. For example, it may be
reasonable to satisfy the conditions by providing a URI or
hyperlink to a resource that includes the required
information.
3. If requested by the Licensor, You must remove any of the
information required by Section 3(a)(1)(A) to the extent
reasonably practicable.
4. If You Share Adapted Material You produce, the Adapter's
License You apply must not prevent recipients of the Adapted
Material from complying with this Public License.
Section 4 -- Sui Generis Database Rights.
Where the Licensed Rights include Sui Generis Database Rights that
apply to Your use of the Licensed Material:
a. for the avoidance of doubt, Section 2(a)(1) grants You the right
to extract, reuse, reproduce, and Share all or a substantial
portion of the contents of the database for NonCommercial purposes
only;
b. if You include all or a substantial portion of the database
contents in a database in which You have Sui Generis Database
Rights, then the database in which You have Sui Generis Database
Rights (but not its individual contents) is Adapted Material; and
c. You must comply with the conditions in Section 3(a) if You Share
all or a substantial portion of the contents of the database.
For the avoidance of doubt, this Section 4 supplements and does not
replace Your obligations under this Public License where the Licensed
Rights include other Copyright and Similar Rights.
Section 5 -- Disclaimer of Warranties and Limitation of Liability.
a. UNLESS OTHERWISE SEPARATELY UNDERTAKEN BY THE LICENSOR, TO THE
EXTENT POSSIBLE, THE LICENSOR OFFERS THE LICENSED MATERIAL AS-IS
AND AS-AVAILABLE, AND MAKES NO REPRESENTATIONS OR WARRANTIES OF
ANY KIND CONCERNING THE LICENSED MATERIAL, WHETHER EXPRESS,
IMPLIED, STATUTORY, OR OTHER. THIS INCLUDES, WITHOUT LIMITATION,
WARRANTIES OF TITLE, MERCHANTABILITY, FITNESS FOR A PARTICULAR
PURPOSE, NON-INFRINGEMENT, ABSENCE OF LATENT OR OTHER DEFECTS,
ACCURACY, OR THE PRESENCE OR ABSENCE OF ERRORS, WHETHER OR NOT
KNOWN OR DISCOVERABLE. WHERE DISCLAIMERS OF WARRANTIES ARE NOT
ALLOWED IN FULL OR IN PART, THIS DISCLAIMER MAY NOT APPLY TO YOU.
b. TO THE EXTENT POSSIBLE, IN NO EVENT WILL THE LICENSOR BE LIABLE
TO YOU ON ANY LEGAL THEORY (INCLUDING, WITHOUT LIMITATION,
NEGLIGENCE) OR OTHERWISE FOR ANY DIRECT, SPECIAL, INDIRECT,
INCIDENTAL, CONSEQUENTIAL, PUNITIVE, EXEMPLARY, OR OTHER LOSSES,
COSTS, EXPENSES, OR DAMAGES ARISING OUT OF THIS PUBLIC LICENSE OR
USE OF THE LICENSED MATERIAL, EVEN IF THE LICENSOR HAS BEEN
ADVISED OF THE POSSIBILITY OF SUCH LOSSES, COSTS, EXPENSES, OR
DAMAGES. WHERE A LIMITATION OF LIABILITY IS NOT ALLOWED IN FULL OR
IN PART, THIS LIMITATION MAY NOT APPLY TO YOU.
c. The disclaimer of warranties and limitation of liability provided
above shall be interpreted in a manner that, to the extent
possible, most closely approximates an absolute disclaimer and
waiver of all liability.
Section 6 -- Term and Termination.
a. This Public License applies for the term of the Copyright and
Similar Rights licensed here. However, if You fail to comply with
this Public License, then Your rights under this Public License
terminate automatically.
b. Where Your right to use the Licensed Material has terminated under
Section 6(a), it reinstates:
1. automatically as of the date the violation is cured, provided
it is cured within 30 days of Your discovery of the
violation; or
2. upon express reinstatement by the Licensor.
For the avoidance of doubt, this Section 6(b) does not affect any
right the Licensor may have to seek remedies for Your violations
of this Public License.
c. For the avoidance of doubt, the Licensor may also offer the
Licensed Material under separate terms or conditions or stop
distributing the Licensed Material at any time; however, doing so
will not terminate this Public License.
d. Sections 1, 5, 6, 7, and 8 survive termination of this Public
License.
Section 7 -- Other Terms and Conditions.
a. The Licensor shall not be bound by any additional or different
terms or conditions communicated by You unless expressly agreed.
b. Any arrangements, understandings, or agreements regarding the
Licensed Material not stated herein are separate from and
independent of the terms and conditions of this Public License.
Section 8 -- Interpretation.
a. For the avoidance of doubt, this Public License does not, and
shall not be interpreted to, reduce, limit, restrict, or impose
conditions on any use of the Licensed Material that could lawfully
be made without permission under this Public License.
b. To the extent possible, if any provision of this Public License is
deemed unenforceable, it shall be automatically reformed to the
minimum extent necessary to make it enforceable. If the provision
cannot be reformed, it shall be severed from this Public License
without affecting the enforceability of the remaining terms and
conditions.
c. No term or condition of this Public License will be waived and no
failure to comply consented to unless expressly agreed to by the
Licensor.
d. Nothing in this Public License constitutes or may be interpreted
as a limitation upon, or waiver of, any privileges and immunities
that apply to the Licensor or You, including from the legal
processes of any jurisdiction or authority.
=======================================================================
Creative Commons is not a party to its public
licenses. Notwithstanding, Creative Commons may elect to apply one of
its public licenses to material it publishes and in those instances
will be considered the “Licensor.” The text of the Creative Commons
public licenses is dedicated to the public domain under the CC0 Public
Domain Dedication. Except for the limited purpose of indicating that
material is shared under a Creative Commons public license or as
otherwise permitted by the Creative Commons policies published at
creativecommons.org/policies, Creative Commons does not authorize the
use of the trademark "Creative Commons" or any other trademark or logo
of Creative Commons without its prior written consent including,
without limitation, in connection with any unauthorized modifications
to any of its public licenses or any other arrangements,
understandings, or agreements concerning use of licensed material. For
the avoidance of doubt, this paragraph does not form part of the
public licenses.
Creative Commons may be contacted at creativecommons.org.
@@ -1,273 +0,0 @@
# RepCodec: A Speech Representation Codec for Speech Tokenization
> [**RepCodec: A Speech Representation Codec for Speech Tokenization**](https://arxiv.org/abs/2309.00169)
## Introduction
**RepCodec** is a speech tokenization method for converting a speech waveform into a sequence of discrete semantic
tokens.
The main idea is to train a representation codec which learns a vector quantization codebook through reconstructing the
input speech representations from speech encoders like HuBERT or data2vec.
Extensive experiments show that RepCodec significantly outperforms the widely used k-means clustering approach in both
speech understanding and generation.
Also, RepCodec generalizes well across various speech encoders and languages.
<img src="images/RepCodec.png" alt="se" width="1000" />
## RepCodec Models
| Feature Type | Speech Data | RepCodec Model |
|-----------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------|----------------------------------------------------------------------------------------------------------|
| [HuBERT base](https://github.com/facebookresearch/fairseq/tree/main/examples/hubert#pre-trained-and-fine-tuned-asr-models) layer 9 | [Librispeech](http://www.openslr.org/12) train-clean-100 | [hubert_base_l9](https://drive.google.com/file/d/1XD0HKl607FFjri2-VJT7lHQeSpxsCCFO/view?usp=sharing) |
| [HuBERT large](https://github.com/facebookresearch/fairseq/tree/main/examples/hubert#pre-trained-and-fine-tuned-asr-models) layer 18 | [Librispeech](http://www.openslr.org/12) train-clean-100 | [hubert_large_l18](https://drive.google.com/file/d/1mTbm5GeJ7gp_5L3QLP-JGXdf8RnRw5n6/view?usp=sharing) |
| [data2vec base](https://github.com/facebookresearch/fairseq/blob/main/examples/data2vec/README.md#speech-2) layer 6 | [Librispeech](http://www.openslr.org/12) train-clean-100 | [data2vec_base_l6](https://drive.google.com/file/d/1d8sf3Ko_fYM9zlaiwxK_4xusLRKV5EMd/view?usp=sharing) |
| [data2vec large](https://github.com/facebookresearch/fairseq/blob/main/examples/data2vec/README.md#speech-2) layer 18 | [Librispeech](http://www.openslr.org/12) train-clean-100 | [data2vec_large_l18](https://drive.google.com/file/d/1nuRIHaejT-uVi4cluftbT8o_JZqar5SU/view?usp=sharing) |
| [Whisper medium](https://github.com/openai/whisper/tree/main#available-models-and-languages) layer 24 | [Librispeech](http://www.openslr.org/12) train-clean-100 | [whisper_medium_l24](https://drive.google.com/file/d/1V6YJSA2V4iywXrecJAN0oqsa3aHowexZ/view?usp=sharing) |
| [Whisper large-v2](https://github.com/openai/whisper/tree/main#available-models-and-languages) layer 32 | [Librispeech](http://www.openslr.org/12) train-clean-100 | [whisper_large_l32](https://drive.google.com/file/d/1k_X7ZMPg8iOeDrIJe70v6CHfFygzufXC/view?usp=sharing) |
## Speech Tokenization Using Pre-Trained Models
### Installation
Please first install RepCodec by
```
git clone https://github.com/mct10/RepCodec.git
cd RepCodec
pip install .
```
We used Python 3.9.18 and PyTorch 1.12.1 to test the usage, but the code should be compatible with other recent Python
and PyTorch versions.
### Representation Preparation
We adapt the `dump_hubert_feature.py` script
from [fairseq](https://github.com/facebookresearch/fairseq/tree/main/examples/hubert/simple_kmeans#hubert-feature)
to support dumping representations from **data2vec**, **HuBERT**, or **Whisper** encoders.
If you use our script (`examples/dump_feature.py`), please also install the following packages:
```
pip install npy_append_array soundfile
```
Additionally, if you want to dump representations from
- **data2vec** or **HuBERT**: please
follow [fairseq's instruction](https://github.com/facebookresearch/fairseq#requirements-and-installation) to install
the latest fairseq.
- **Whisper**: please follow [Whispers'instruction](https://github.com/openai/whisper/tree/main#setup) to install the
latest
Whisper.
Then, you can follow the given examples to dump representations:
```
# Example 1: dump from HuBERT base layer 9
# (for data2vec, simply change "model_type" to data2vec and "ckpt_path" to the path of data2vec model)
layer=9
python3 examples/dump_feature.py \
--model_type hubert \
--tsv_path /path/to/tsv/file \
--ckpt_path /path/to/HuBERT/model \
--layer ${layer} \
--feat_dir /dir/to/save/representations
# Example 2: dump from Whisper medium layer 24
layer=24
python3 examples/dump_feature.py \
--model_type whisper \
--tsv_path /path/to/tsv/file \
--whisper_root /directory/to/save/whisper/model \
--whisper_name medium \
--layer ${layer} \
--feat_dir /dir/to/save/representations
```
Explanations about the args:
- **model_type:** choose from `data2vec`, `hubert`, and `whisper`.
- **tsv_path:** path of the tsv file.
Should have the format of
```
/dir/to/dataset
path_of_utterance_1 number_of_frames
path_of_utterance_2 number_of_frames
```
You can follow [this script](https://github.com/facebookresearch/fairseq/blob/main/examples/wav2vec/wav2vec_manifest.py)
to generate the tsv file.
For example, by running
```
python wav2vec_manifest.py \
/dir/to/LibriSpeech/dev-clean \
--dest /dir/to/manifest \
--ext flac \
--valid-percent 0
```
you can obtain the `dev-clean.tsv` in `/dir/to/manifest` for LibriSpeech. (By default, the output file name
is `train.tsv`. Remember to rename the file.)
It should be similar to:
```
/dir/to/LibriSpeech/dev-clean
2277/149896/2277-149896-0026.flac 78720
2277/149896/2277-149896-0005.flac 89600
2277/149896/2277-149896-0033.flac 45520
```
- **ckpt_path**:
must provide for data2vec and HuBERT.
You need to download the model
from [data2vec website](https://github.com/facebookresearch/fairseq/blob/main/examples/data2vec/README.md#speech-2)
or [HuBERT website](https://github.com/facebookresearch/fairseq/tree/main/examples/hubert#pre-trained-and-fine-tuned-asr-models)
yourself.
`--ckpt_path` is the path of the data2vec/HuBERT model.
- **whisper_root** and **whisper_name**:
must provide **BOTH** `--whisper_root` and `--whisper_name` for Whisper.
If there is no corresponding model in `--whisper_root`, the script will download for you.
- **layer**:
which Transformer encoder layer of the model should the representations be extracted from.
It is **1-based**.
For example, if layer=9, then the outputs from the 9<sup>th</sup> Transformer encoder layer are dumped.
Range: [1, number of Transformer encoder layers]
- **feat_dir**: The output representations will be saved to `${feat_dir}/0_1.npy`
and `${feat_dir}/0_1.len`.
For other useful functionalities (e.g., sharding), please check the argument list in `examples/dump_feature.py`.
### Command Line Usage
We expect to have `${feat_dir}/0_1.npy` and `${feat_dir}/0_1.len` in the provided
directory `/dir/to/representaitons`.
Also, the tsv file should be the **same** as the one used in [Representation Preparation](#representation-preparation).
```
repcodec /dir/to/representaitons \
--model /path/to/repcodec/model \
--tsv_path /path/to/tsv/file \
[--model_config_path /path/to/train/config] \
[--use_gpu] \
[--out_dir /path/to/output]
```
If you trained the model yourself following [Training New RepCodec Models](#training-new-repcodec-models),
please provide the training config file using `--model_config_path`.
If you use the model we provide [here](#repcodec-models), then you do not have to provide that.
This command will tokenize the representations and the output discrete tokens will be saved to `${out_dir}/tokens`.
The tokens are in the same order as the provided tsv file.
An example of the output file:
```
/dir/to/LibriSpeech/dev-clean
2277/149896/2277-149896-0026.flac 696 696 198 198 198 498 ...
2277/149896/2277-149896-0005.flac 696 696 198 198 198 907 ...
2277/149896/2277-149896-0033.flac 696 696 198 198 198 696 ...
```
Under `examples/tokens`, we provide some token files as references. They are obtained from LibriSpeech dev-clean subset
using the 6 types of representations and corresponding [RepCodec Models](#repcodec-models).
Your results should be very similar to ours.
### Python Usage
```python
import torch
import yaml
from repcodec.RepCodec import RepCodec
# for feature types of HubERT base & data2vec base, please use repcodec_dim768.yaml;
# for feature types of HuBERT large & data2vec large & Whisper medium, please use repcodec_dim1024.yaml;
# for feature types of Whisper large-v2, please use repcodec_dim1280.yaml
config = "repcodec/configs/repcodec_dim768.yaml"
with open(config) as fp:
conf = yaml.load(fp, Loader=yaml.FullLoader)
model = RepCodec(**conf)
model.load_state_dict(torch.load("./hubert_base_l9.pkl", map_location="cpu")["model"]["repcodec"])
model.quantizer.initial()
model.eval()
# input shape: (batch size, hidden dim, sequence length)
random_features = torch.randn(size=(1, 768, 100))
with torch.no_grad():
x = model.encoder(random_features)
z = model.projector(x)
_, idx = model.quantizer.codebook.forward_index(z.transpose(2, 1))
tokens = idx.cpu().data.numpy().tolist()[0]
```
## Training New RepCodec Models
We use a config file to set up all the training configurations, e.g., data, model architecture,
optimizer, scheduler.
We provide an example [here](./train_configs/ex_dim768_mse.yaml).
Please first install required packages following [Installation](#installation)
and prepare the representations following [Representation Preparation](#representation-preparation).
The input data directory is expected to have the following structure
```
/dir/to/representations/
train_set_name/
0_1.npy
0_1.len
valid_set_name/
0_1.npy
0_1.len
test_set_name/
0_1.npy
0_1.len
```
The names of subsets should be the same as the fields in the config file.
Then, you can run training by
```
python train.py \
-c /path/to/config/file \
--tag $tag \
--exp_root exp
```
`tag` is the name of the output folder.
All outputs will be saved to `exp_root/tag/`.
## Acknowledge
Our implementation is based on [facebookresearch/AudioDec](https://github.com/facebookresearch/AudioDec).
We thank them for open-sourcing their code!
## Citation
If you find our work useful, please cite the following article.
```
@misc{huang2023repcodec,
title={RepCodec: A Speech Representation Codec for Speech Tokenization},
author={Zhichao Huang and Chutong Meng and Tom Ko},
year={2023},
eprint={2309.00169},
archivePrefix={arXiv},
primaryClass={eess.AS}
}
```
@@ -1,541 +0,0 @@
# Copyright (c) ByteDance, Inc. and its affiliates.
# Copyright (c) Chutong Meng
#
# This source code is licensed under the MIT license found in the
# LICENSE file in the root directory of this source tree.
# Based on fairseq (https://github.com/facebookresearch/fairseq)
# ref: https://github.com/facebookresearch/fairseq/blob/main/examples/data2vec/models/data2vec_audio.py
import logging
import math
from dataclasses import dataclass, field
from typing import Optional
from omegaconf import II
import torch
import torch.nn as nn
import torch.nn.functional as F
import torch.distributed as dist
from fairseq.modules import EMAModule, EMAModuleConfig
from fairseq.data.data_utils import compute_mask_indices
from fairseq.models import BaseFairseqModel, register_model
from fairseq.models.wav2vec import (
ConvFeatureExtractionModel,
Wav2Vec2Config,
TransformerEncoder,
)
from fairseq.modules import (
GradMultiply,
LayerNorm,
)
from fairseq.utils import index_put
logger = logging.getLogger(__name__)
@dataclass
class Data2VecAudioConfig(Wav2Vec2Config):
loss_beta: float = field(
default=0, metadata={"help": "beta for smooth l1 loss. 0 means use l2 loss"}
)
loss_scale: Optional[float] = field(
default=None,
metadata={
"help": "scale the reconstruction loss by this constant. if None then scales by 1/sqrt(dim)"
},
)
average_top_k_layers: int = field(
default=8, metadata={"help": "how many layers to average"}
)
layer_norm_target_layer: bool = False
instance_norm_target_layer: bool = False
instance_norm_targets: bool = False
layer_norm_targets: bool = False
batch_norm_target_layer: bool = False
group_norm_target_layer: bool = False
ema_decay: float = field(default=0.999, metadata={"help": "initial ema decay rate"})
ema_end_decay: float = field(
default=0.9999, metadata={"help": "final ema decay rate"}
)
# when to finish annealing ema decay rate
ema_anneal_end_step: int = II("optimization.max_update")
ema_transformer_only: bool = field(
default=True,
metadata={"help": "whether to momentum update only the transformer"},
)
ema_layers_only: bool = field(
default=True,
metadata={"help": "whether to momentum update only the transformer layers"},
)
max_update: int = II("optimization.max_update")
min_target_var: float = field(
default=0.1, metadata={"help": "stop training if target var falls below this"}
)
min_pred_var: float = field(
default=0.01,
metadata={"help": "stop training if prediction var falls below this"},
)
def get_annealed_rate(start, end, curr_step, total_steps):
r = end - start
pct_remaining = 1 - curr_step / total_steps
return end - r * pct_remaining
@register_model("data2vec_audio", dataclass=Data2VecAudioConfig)
class Data2VecAudioModel(BaseFairseqModel):
def __init__(self, cfg: Data2VecAudioConfig):
super().__init__()
self.cfg = cfg
feature_enc_layers = eval(cfg.conv_feature_layers)
self.extractor_embed = feature_enc_layers[-1][0]
self.ema = None
self.embed = cfg.encoder_embed_dim
self.average_top_k_layers = cfg.average_top_k_layers
self.loss_beta = cfg.loss_beta
self.loss_scale = cfg.loss_scale
self.feature_extractor = ConvFeatureExtractionModel(
conv_layers=feature_enc_layers,
dropout=0.0,
mode=cfg.extractor_mode,
conv_bias=cfg.conv_bias,
)
self.post_extract_proj = nn.Linear(self.extractor_embed, cfg.encoder_embed_dim)
self.mask_prob = cfg.mask_prob
self.mask_selection = cfg.mask_selection
self.mask_other = cfg.mask_other
self.mask_length = cfg.mask_length
self.no_mask_overlap = cfg.no_mask_overlap
self.mask_min_space = cfg.mask_min_space
self.mask_channel_prob = cfg.mask_channel_prob
self.mask_channel_before = cfg.mask_channel_before
self.mask_channel_selection = cfg.mask_channel_selection
self.mask_channel_other = cfg.mask_channel_other
self.mask_channel_length = cfg.mask_channel_length
self.no_mask_channel_overlap = cfg.no_mask_channel_overlap
self.mask_channel_min_space = cfg.mask_channel_min_space
self.dropout_input = nn.Dropout(cfg.dropout_input)
self.dropout_features = nn.Dropout(cfg.dropout_features)
self.feature_grad_mult = cfg.feature_grad_mult
self.mask_emb = nn.Parameter(
torch.FloatTensor(cfg.encoder_embed_dim).uniform_()
)
self.encoder = TransformerEncoder(cfg)
self.layer_norm = LayerNorm(self.extractor_embed)
self.final_proj = nn.Linear(self.embed, self.embed)
self.num_updates = 0
def make_ema_teacher(self):
ema_config = EMAModuleConfig(
ema_decay=self.cfg.ema_decay,
ema_fp32=True,
)
skip_keys = set()
if self.cfg.ema_layers_only:
self.cfg.ema_transformer_only = True
for k, _ in self.encoder.pos_conv.named_parameters():
skip_keys.add(f"pos_conv.{k}")
self.ema = EMAModule(
self.encoder if self.cfg.ema_transformer_only else self,
ema_config,
skip_keys=skip_keys,
)
def set_num_updates(self, num_updates):
super().set_num_updates(num_updates)
if self.ema is None and self.final_proj is not None:
logger.info(f"making ema teacher")
self.make_ema_teacher()
elif self.training and self.ema is not None:
if self.cfg.ema_decay != self.cfg.ema_end_decay:
if num_updates >= self.cfg.ema_anneal_end_step:
decay = self.cfg.ema_end_decay
else:
decay = get_annealed_rate(
self.cfg.ema_decay,
self.cfg.ema_end_decay,
num_updates,
self.cfg.ema_anneal_end_step,
)
self.ema.set_decay(decay)
if self.ema.get_decay() < 1:
self.ema.step(self.encoder if self.cfg.ema_transformer_only else self)
self.num_updates = num_updates
def state_dict(self, destination=None, prefix="", keep_vars=False):
state = super().state_dict(destination, prefix, keep_vars)
if self.ema is not None:
state[prefix + "_ema"] = self.ema.fp32_params
return state
def _load_from_state_dict(self, state_dict, prefix, *args, **kwargs):
if self.ema is not None:
k = prefix + "_ema"
assert k in state_dict
self.ema.restore(state_dict[k], True)
del state_dict[k]
return super()._load_from_state_dict(state_dict, prefix, *args, **kwargs)
@classmethod
def build_model(cls, cfg: Data2VecAudioConfig, task=None):
"""Build a new model instance."""
return cls(cfg)
def apply_mask(
self,
x,
padding_mask,
mask_indices=None,
mask_channel_indices=None,
):
B, T, C = x.shape
if self.mask_channel_prob > 0 and self.mask_channel_before:
mask_channel_indices = compute_mask_indices(
(B, C),
None,
self.mask_channel_prob,
self.mask_channel_length,
self.mask_channel_selection,
self.mask_channel_other,
no_overlap=self.no_mask_channel_overlap,
min_space=self.mask_channel_min_space,
)
mask_channel_indices = (
torch.from_numpy(mask_channel_indices)
.to(x.device)
.unsqueeze(1)
.expand(-1, T, -1)
)
x[mask_channel_indices] = 0
if self.mask_prob > 0:
if mask_indices is None:
mask_indices = compute_mask_indices(
(B, T),
padding_mask,
self.mask_prob,
self.mask_length,
self.mask_selection,
self.mask_other,
min_masks=1,
no_overlap=self.no_mask_overlap,
min_space=self.mask_min_space,
require_same_masks=self.cfg.require_same_masks,
mask_dropout=self.cfg.mask_dropout,
)
mask_indices = torch.from_numpy(mask_indices).to(x.device)
x = index_put(x, mask_indices, self.mask_emb)
else:
mask_indices = None
if self.mask_channel_prob > 0 and not self.mask_channel_before:
if mask_channel_indices is None:
mask_channel_indices = compute_mask_indices(
(B, C),
None,
self.mask_channel_prob,
self.mask_channel_length,
self.mask_channel_selection,
self.mask_channel_other,
no_overlap=self.no_mask_channel_overlap,
min_space=self.mask_channel_min_space,
)
mask_channel_indices = (
torch.from_numpy(mask_channel_indices)
.to(x.device)
.unsqueeze(1)
.expand(-1, T, -1)
)
x = index_put(x, mask_channel_indices, 0)
return x, mask_indices
def _get_feat_extract_output_lengths(self, input_lengths: torch.LongTensor):
"""
Computes the output length of the convolutional layers
"""
def _conv_out_length(input_length, kernel_size, stride):
return torch.floor((input_length - kernel_size) / stride + 1)
conv_cfg_list = eval(self.cfg.conv_feature_layers)
for i in range(len(conv_cfg_list)):
input_lengths = _conv_out_length(
input_lengths, conv_cfg_list[i][1], conv_cfg_list[i][2]
)
return input_lengths.to(torch.long)
def forward(
self,
source,
padding_mask=None,
mask=True,
features_only=False,
layer=None,
mask_indices=None,
mask_channel_indices=None,
padding_count=None,
):
features = source
if self.feature_grad_mult > 0:
features = self.feature_extractor(features)
if self.feature_grad_mult != 1.0:
features = GradMultiply.apply(features, self.feature_grad_mult)
else:
with torch.no_grad():
features = self.feature_extractor(features)
features = features.transpose(1, 2)
features = self.layer_norm(features)
orig_padding_mask = padding_mask
if padding_mask is not None and padding_mask.any():
input_lengths = (1 - padding_mask.long()).sum(-1)
# apply conv formula to get real output_lengths
output_lengths = self._get_feat_extract_output_lengths(input_lengths)
padding_mask = torch.zeros(
features.shape[:2], dtype=features.dtype, device=features.device
)
# these two operations makes sure that all values
# before the output lengths indices are attended to
padding_mask[
(
torch.arange(padding_mask.shape[0], device=padding_mask.device),
output_lengths - 1,
)
] = 1
padding_mask = (1 - padding_mask.flip([-1]).cumsum(-1).flip([-1])).bool()
else:
padding_mask = None
if self.post_extract_proj is not None:
features = self.post_extract_proj(features)
pre_encoder_features = None
if self.cfg.ema_transformer_only:
pre_encoder_features = features.clone()
features = self.dropout_input(features)
if mask:
x, mask_indices = self.apply_mask(
features,
padding_mask,
mask_indices=mask_indices,
mask_channel_indices=mask_channel_indices,
)
else:
x = features
mask_indices = None
x, layer_results = self.encoder(
x,
padding_mask=padding_mask,
layer=layer,
)
if features_only:
return {
"x": x,
"padding_mask": padding_mask,
"layer_results": layer_results,
}
result = {
"losses": {},
}
with torch.no_grad():
self.ema.model.eval()
if self.cfg.ema_transformer_only:
y, layer_results = self.ema.model.extract_features(
pre_encoder_features,
padding_mask=padding_mask,
min_layer=self.cfg.encoder_layers - self.average_top_k_layers,
)
y = {
"x": y,
"padding_mask": padding_mask,
"layer_results": layer_results,
}
else:
y = self.ema.model.extract_features(
source=source,
padding_mask=orig_padding_mask,
mask=False,
)
target_layer_results = [l[2] for l in y["layer_results"]]
permuted = False
if self.cfg.instance_norm_target_layer or self.cfg.batch_norm_target_layer:
target_layer_results = [
tl.permute(1, 2, 0) for tl in target_layer_results # TBC -> BCT
]
permuted = True
if self.cfg.batch_norm_target_layer:
target_layer_results = [
F.batch_norm(
tl.float(), running_mean=None, running_var=None, training=True
)
for tl in target_layer_results
]
if self.cfg.instance_norm_target_layer:
target_layer_results = [
F.instance_norm(tl.float()) for tl in target_layer_results
]
if permuted:
target_layer_results = [
tl.transpose(1, 2) for tl in target_layer_results # BCT -> BTC
]
if self.cfg.group_norm_target_layer:
target_layer_results = [
F.layer_norm(tl.float(), tl.shape[-2:])
for tl in target_layer_results
]
if self.cfg.layer_norm_target_layer:
target_layer_results = [
F.layer_norm(tl.float(), tl.shape[-1:])
for tl in target_layer_results
]
y = sum(target_layer_results) / len(target_layer_results)
if self.cfg.layer_norm_targets:
y = F.layer_norm(y.float(), y.shape[-1:])
if self.cfg.instance_norm_targets:
y = F.instance_norm(y.float().transpose(1, 2)).transpose(1, 2)
if not permuted:
y = y.transpose(0, 1)
y = y[mask_indices]
x = x[mask_indices]
x = self.final_proj(x)
sz = x.size(-1)
if self.loss_beta == 0:
loss = F.mse_loss(x.float(), y.float(), reduction="none").sum(dim=-1)
else:
loss = F.smooth_l1_loss(
x.float(), y.float(), reduction="none", beta=self.loss_beta
).sum(dim=-1)
if self.loss_scale is not None:
scale = self.loss_scale
else:
scale = 1 / math.sqrt(sz)
result["losses"]["regression"] = loss.sum() * scale
if "sample_size" not in result:
result["sample_size"] = loss.numel()
with torch.no_grad():
result["target_var"] = self.compute_var(y)
result["pred_var"] = self.compute_var(x.float())
if self.num_updates > 5000 and result["target_var"] < self.cfg.min_target_var:
logger.error(
f"target var is {result['target_var'].item()} < {self.cfg.min_target_var}, exiting"
)
raise Exception(
f"target var is {result['target_var'].item()} < {self.cfg.min_target_var}, exiting"
)
if self.num_updates > 5000 and result["pred_var"] < self.cfg.min_pred_var:
logger.error(
f"pred var is {result['pred_var'].item()} < {self.cfg.min_pred_var}, exiting"
)
raise Exception(
f"pred var is {result['pred_var'].item()} < {self.cfg.min_pred_var}, exiting"
)
if self.ema is not None:
result["ema_decay"] = self.ema.get_decay() * 1000
return result
@staticmethod
def compute_var(y):
y = y.view(-1, y.size(-1))
if dist.is_initialized():
zc = torch.tensor(y.size(0)).cuda()
zs = y.sum(dim=0)
zss = (y ** 2).sum(dim=0)
dist.all_reduce(zc)
dist.all_reduce(zs)
dist.all_reduce(zss)
var = zss / (zc - 1) - (zs ** 2) / (zc * (zc - 1))
return torch.sqrt(var + 1e-6).mean()
else:
return torch.sqrt(y.var(dim=0) + 1e-6).mean()
def extract_features(
self, source, padding_mask, mask=False, layer=None
):
res = self.forward(
source,
padding_mask,
mask=mask,
features_only=True,
layer=layer,
)
return res
def remove_pretraining_modules(self, last_layer=None):
self.final_proj = None
self.ema = None
if last_layer is not None:
self.encoder.layers = nn.ModuleList(
l for i, l in enumerate(self.encoder.layers) if i <= last_layer
)
@@ -1,87 +0,0 @@
# Copyright (c) ByteDance, Inc. and its affiliates.
# Copyright (c) Chutong Meng
#
# This source code is licensed under the MIT license found in the
# LICENSE file in the root directory of this source tree.
# Based on fairseq (https://github.com/facebookresearch/fairseq)
import logging
import torch
import torch.nn.functional as F
from fairseq import tasks
from fairseq.checkpoint_utils import load_checkpoint_to_cpu
from fairseq.data.audio.audio_utils import get_features_or_waveform
from omegaconf import OmegaConf
from data2vec_audio import Data2VecAudioModel
logger = logging.getLogger("dump_feature")
class Data2vecFeatureReader(object):
def __init__(self, ckpt_path: str, layer: int, device: str, max_chunk=1600000):
state = load_checkpoint_to_cpu(ckpt_path)
cfg = state["cfg"]
# load task
task = tasks.setup_task(cfg.task, from_checkpoint=True)
task.load_state_dict(state["task_state"])
# load model config
if "layer_type" not in cfg.model:
# fix a missing key
model_config = {k: v for k, v in cfg.model.items()}
model_config["layer_type"] = "transformer"
model_config = OmegaConf.create(model_config)
else:
model_config = cfg.model
# fix param name in the state
state["model"]["final_proj.weight"] = state["model"].pop("final_proj.0.weight")
state["model"]["final_proj.bias"] = state["model"].pop("final_proj.0.bias")
del state["model"]["_ema"]
# load model
model = Data2VecAudioModel.build_model(model_config)
model.load_state_dict(
state["model"], strict=True, model_cfg=model_config
)
self.device = device
logger.info(f"device = {self.device}")
self.model = model.eval().to(self.device)
self.task = task
self.layer = layer - 1 # make it 1-based
self.max_chunk = max_chunk
logger.info(f"TASK CONFIG:\n{self.task.cfg}")
logger.info(f" max_chunk = {self.max_chunk}")
def read_audio(self, path, ref_len=None):
wav = get_features_or_waveform(path, need_waveform=True, use_sample_rate=self.task.cfg.sample_rate)
if wav.ndim == 2:
wav = wav.mean(-1)
assert wav.ndim == 1, wav.ndim
if ref_len is not None and abs(ref_len - len(wav)) > 160:
logger.warning(f"ref {ref_len} != read {len(wav)} ({path})")
return wav
def get_feats(self, path, ref_len=None):
x = self.read_audio(path, ref_len=ref_len)
with torch.no_grad():
x = torch.from_numpy(x).float().to(self.device)
if self.task.cfg.normalize:
x = F.layer_norm(x, x.shape)
x = x.view(1, -1)
feat = []
for start in range(0, x.size(1), self.max_chunk):
x_chunk = x[:, start: start + self.max_chunk]
res = self.model.extract_features(
source=x_chunk,
padding_mask=None,
mask=False,
layer=self.layer,
)
feat_chunk = res["x"]
feat.append(feat_chunk)
return torch.cat(feat, 1).squeeze(0)
@@ -1,142 +0,0 @@
# Copyright (c) ByteDance, Inc. and its affiliates.
# Copyright (c) Chutong Meng
#
# This source code is licensed under the MIT license found in the
# LICENSE file in the root directory of this source tree.
# Based on fairseq (https://github.com/facebookresearch/fairseq)
import logging
import os
import sys
from feature_utils import get_path_iterator, dump_feature
logging.basicConfig(
format="%(asctime)s | %(levelname)s | %(name)s | %(message)s",
datefmt="%Y-%m-%d %H:%M:%S",
level=os.environ.get("LOGLEVEL", "INFO").upper(),
stream=sys.stdout,
)
logger = logging.getLogger("dump_feature")
def main(
model_type: str,
tsv_path: str,
ckpt_path: str,
whisper_root: str,
whisper_name: str,
layer: int,
nshard: int,
rank: int,
feat_dir: str,
max_chunk: int,
use_cpu: bool = False
):
device = "cpu" if use_cpu else "cuda"
# some checks
if model_type in ["hubert", "data2vec"]:
assert ckpt_path and os.path.exists(ckpt_path)
elif model_type in ["whisper"]:
assert whisper_name and whisper_root
else:
raise ValueError(f"Unsupported model type {model_type}")
reader = None
if model_type == "hubert":
from hubert_feature_reader import HubertFeatureReader
reader = HubertFeatureReader(ckpt_path, layer, device=device, max_chunk=max_chunk)
elif model_type == "data2vec":
from data2vec_feature_reader import Data2vecFeatureReader
reader = Data2vecFeatureReader(ckpt_path, layer, device=device, max_chunk=max_chunk)
elif model_type == "whisper":
from whisper_feature_reader import WhisperFeatureReader
reader = WhisperFeatureReader(whisper_root, whisper_name, layer, device=device)
assert reader is not None
generator, num = get_path_iterator(tsv_path, nshard, rank)
dump_feature(reader, generator, num, nshard, rank, feat_dir)
if __name__ == "__main__":
import argparse
parser = argparse.ArgumentParser()
parser.add_argument(
"--model_type",
required=True,
type=str,
choices=["data2vec", "hubert", "whisper"],
help="the type of the speech encoder."
)
parser.add_argument(
"--tsv_path",
required=True,
type=str,
help="the path to the tsv file."
)
parser.add_argument(
"--ckpt_path",
required=False,
type=str,
default=None,
help="path to the speech model. must provide for HuBERT and data2vec"
)
parser.add_argument(
"--whisper_root",
required=False,
type=str,
default=None,
help="root dir to download/store whisper model. must provide for whisper model."
)
parser.add_argument(
"--whisper_name",
required=False,
type=str,
default=None,
help="name of whisper model. e.g., large-v2. must provide for whisper model."
)
parser.add_argument(
"--layer",
required=True,
type=int,
help="which layer of the model. this is 1-based."
)
parser.add_argument(
"--feat_dir",
required=True,
type=str,
help="the output dir to save the representations."
)
parser.add_argument(
"--nshard",
required=False,
type=int,
default=1,
help="total number of shards."
)
parser.add_argument(
"--rank",
required=False,
type=int,
default=0,
help="shard id of this process."
)
parser.add_argument(
"--max_chunk",
type=int,
default=1600000,
help="max number of frames of each batch."
)
parser.add_argument(
"--use_cpu",
default=False,
action="store_true",
help="whether use cpu instead of gpu."
)
args = parser.parse_args()
logger.info(args)
main(**vars(args))
@@ -1,70 +0,0 @@
# Copyright (c) ByteDance, Inc. and its affiliates.
# Copyright (c) Chutong Meng
#
# This source code is licensed under the MIT license found in the
# LICENSE file in the root directory of this source tree.
# Based on fairseq (https://github.com/facebookresearch/fairseq)
# ref: https://github.com/facebookresearch/fairseq/blob/main/examples/hubert/simple_kmeans/feature_utils.py
import logging
import os
import sys
import tqdm
from npy_append_array import NpyAppendArray
logging.basicConfig(
format="%(asctime)s | %(levelname)s | %(name)s | %(message)s",
datefmt="%Y-%m-%d %H:%M:%S",
level=os.environ.get("LOGLEVEL", "INFO").upper(),
stream=sys.stdout,
)
logger = logging.getLogger("feature_utils")
def get_shard_range(tot, nshard, rank):
assert rank < nshard and rank >= 0, f"invaid rank/nshard {rank}/{nshard}"
start = round(tot / nshard * rank)
end = round(tot / nshard * (rank + 1))
assert start < end, f"start={start}, end={end}"
logger.info(
f"rank {rank} of {nshard}, process {end-start} "
f"({start}-{end}) out of {tot}"
)
return start, end
def get_path_iterator(tsv, nshard, rank):
with open(tsv, "r") as f:
root = f.readline().rstrip()
lines = [line.rstrip() for line in f]
start, end = get_shard_range(len(lines), nshard, rank)
lines = lines[start:end]
def iterate():
for line in lines:
subpath, nsample = line.split("\t")
yield f"{root}/{subpath}", int(nsample)
return iterate, len(lines)
def dump_feature(reader, generator, num, nshard, rank, feat_dir):
iterator = generator()
feat_path = f"{feat_dir}/{rank}_{nshard}.npy"
leng_path = f"{feat_dir}/{rank}_{nshard}.len"
os.makedirs(feat_dir, exist_ok=True)
if os.path.exists(feat_path):
os.remove(feat_path)
feat_f = NpyAppendArray(feat_path)
with open(leng_path, "w") as leng_f:
for path, nsample in tqdm.tqdm(iterator, total=num):
feat = reader.get_feats(path, nsample)
feat_f.append(feat.cpu().numpy())
leng_f.write(f"{len(feat)}\n")
logger.info("finished successfully")
@@ -1,64 +0,0 @@
# Copyright (c) ByteDance, Inc. and its affiliates.
# Copyright (c) Chutong Meng
#
# This source code is licensed under the MIT license found in the
# LICENSE file in the root directory of this source tree.
# Based on fairseq (https://github.com/facebookresearch/fairseq)
import logging
import fairseq
import torch
import torch.nn.functional as F
from fairseq.data.audio.audio_utils import get_features_or_waveform
logger = logging.getLogger("dump_feature")
class HubertFeatureReader(object):
def __init__(self, ckpt_path: str, layer: int, device: str, max_chunk=1600000):
(
model,
cfg,
task,
) = fairseq.checkpoint_utils.load_model_ensemble_and_task([ckpt_path])
self.device = device
logger.info(f"device = {self.device}")
self.model = model[0].eval().to(self.device)
self.task = task
self.layer = layer
self.max_chunk = max_chunk
logger.info(f"TASK CONFIG:\n{self.task.cfg}")
logger.info(f" max_chunk = {self.max_chunk}")
def read_audio(self, path, ref_len=None):
wav = get_features_or_waveform(path, need_waveform=True, use_sample_rate=self.task.cfg.sample_rate)
if wav.ndim == 2:
wav = wav.mean(-1)
assert wav.ndim == 1, wav.ndim
if ref_len is not None and abs(ref_len - len(wav)) > 160:
logger.warning(f"ref {ref_len} != read {len(wav)} ({path})")
return wav
def get_feats(self, path, ref_len=None):
x = self.read_audio(path, ref_len=ref_len)
with torch.no_grad():
x = torch.from_numpy(x).float().to(self.device)
if self.task.cfg.normalize:
x = F.layer_norm(x, x.shape)
x = x.view(1, -1)
feat = []
for start in range(0, x.size(1), self.max_chunk):
x_chunk = x[:, start: start + self.max_chunk]
feat_chunk, _ = self.model.extract_features(
source=x_chunk,
padding_mask=None,
mask=False,
output_layer=self.layer,
)
feat.append(feat_chunk)
return torch.cat(feat, 1).squeeze(0)
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long

Some files were not shown because too many files have changed in this diff Show More