From bc2926f2eee48bdb9b51f47930c2c32f866f5a77 Mon Sep 17 00:00:00 2001 From: NewCommer00 Date: Thu, 30 Apr 2026 20:58:01 +0800 Subject: [PATCH] feat(expressions, embedder, wavtool): add Breathiness & Voicing params, mHuBERT backend feat(expressions): add BrecLoader and VoicLoader for Breathiness and Voicing - Expression parameter extraction using HPSS-based breath/voice index analysis feat(wavtool): add extract_wav_breath_voice() for HPSS frequency-band RMS extraction; add extract_wav_embeddings() with mHuBERT/PCA support and caching feat(embedder): introduce mHuBERT as advanced speech feature extraction backend for improved cross-language phoneme alignment refactor(expressions): rename pitd cache dir from "pitd" to "f0"; remove add_cuda_to_path import in crepe branch; force RMVPE to device="cpu" docs(readme): restructure ML models section with Pitch Extraction and mHuBERT Feature Extractor subsections; add CC-BY-NC-SA-4.0 license warning docs(readme): add Get Model Weights section with HF mirror tip and separate RMVPE/mHuBERT download instructions docs(readme): add Breathiness and Voicing to supported params and features list; add BrecLoader/VoicLoader subgraphs to algorithm workflow diagram; fold v0.5.0/v0.6.0 version tips into inline bullet text; remove v0.9.1 tag examples: add demo scripts for mHuBERT embedder and Breathiness/Voicing expressions - examples/mhubert_embedder_demo.py: minimal usage of mHuBERT feature extraction - examples/breath_voicing_control.ipynb: interactive tutorial for expression parameter tuning with visualization and audio preview Acknowledgements: - Special thanks to @ma0shu for foundational work on expressive parameter extraction: https://github.com/ma0shu/expressive/commit/a15858d License Reminder: - Model weights, pretrained assets, and derived features are distributed under CC-BY-NC-SA-4.0. Commercial use requires separate authorization. Please review LICENSE and README.md for full terms. --- README.en.md | 116 +- README.md | 116 +- examples/Lullaby for a Soldier/README.md | 16 + .../expressive_config.brec.json | 59 + .../expressive_config.voic.json | 59 + examples/Lullaby for a Soldier/project.ustx | 1059 +++++++++++++++++ examples/Lullaby for a Soldier/reference.wav | 3 + examples/Lullaby for a Soldier/utau.wav | 3 + expressions/brec.py | 128 ++ expressions/dyn.py | 4 +- expressions/pitd.py | 81 +- expressions/tenc.py | 4 +- expressions/voic.py | 129 ++ expressive.py | 24 +- expressive_gui.py | 101 +- locales/app.pot | 337 ++++-- locales/en/LC_MESSAGES/app.po | 389 ++++-- locales/zh_CN/LC_MESSAGES/app.po | 377 ++++-- pyproject.toml | 5 + tests/test_embedder.py | 600 ++++++++++ tests/test_expressive.py | 356 +++++- tests/test_fs.py | 87 +- tests/test_wavtool.py | 382 +++++- utils/embedder.py | 486 ++++++++ utils/fs.py | 18 +- utils/wavtool.py | 140 ++- 26 files changed, 4611 insertions(+), 468 deletions(-) create mode 100644 examples/Lullaby for a Soldier/README.md create mode 100644 examples/Lullaby for a Soldier/expressive_config.brec.json create mode 100644 examples/Lullaby for a Soldier/expressive_config.voic.json create mode 100644 examples/Lullaby for a Soldier/project.ustx create mode 100644 examples/Lullaby for a Soldier/reference.wav create mode 100644 examples/Lullaby for a Soldier/utau.wav create mode 100644 expressions/brec.py create mode 100644 expressions/voic.py create mode 100644 tests/test_embedder.py create mode 100644 utils/embedder.py diff --git a/README.en.md b/README.en.md index cde3196..f9b9e57 100644 --- a/README.en.md +++ b/README.en.md @@ -22,9 +22,11 @@ The current version supports importing the following expression parameters: +* `Breathiness (curve)` * `Dynamics (curve)` * `Pitch Deviation (curve)` * `Tension (curve)` +* `Voicing (curve)`
@@ -51,13 +53,28 @@ The current version supports importing the following expression parameters: * OpenUtau Beta (or other versions with DiffSinger support) * Python 3.10 \* +> \* On Windows, TensorFlow 2.10 is the last version that supports GPU acceleration, and Python 3.10 is the highest Python version supported by its `.whl` files. + +### Machine Learning Models + +#### Pitch Extraction + By default, this application uses [rmvpe-onnx](https://github.com/newcomer00/rmvpe-onnx) as the pitch extraction backend, which runs on CPU only. [RMVPE](https://arxiv.org/abs/2306.15412v2) is currently the best-performing publicly available pitch extraction algorithm, and its inference speed is fast enough to satisfy the vast majority of use cases. The [swift-f0](https://github.com/lars76/swift-f0) and [CREPE](https://github.com/marl/crepe) pitch extraction backends are also available. The former runs on CPU only and is the fastest option, though its accuracy is modest. The latter is a classic algorithm in the field and runs more slowly. In a CUDA environment, the CREPE backend will automatically enable GPU acceleration. There is also a newly added experimental **hybrid** backend available. The hybrid backend combines the prediction results of rmvpe-onnx and swift-f0, primarily using the pitch extraction results from rmvpe-onnx. In voiced segments of the audio, if the confidence of rmvpe-onnx is low and the confidence of swift-f0 is high, the result from swift-f0 is used for correction, improving the overall accuracy of pitch extraction. -> \* On Windows, TensorFlow 2.10 is the last version that supports GPU acceleration, and Python 3.10 is the highest Python version supported by its `.whl` files. +#### mHuBERT Feature Extractor + +> [!IMPORTANT] +> The mHuBERT model weights are distributed under the [CC-BY-NC-SA-4.0](https://creativecommons.org/licenses/by-nc-sa/4.0/) license and are **for non-commercial use only**. This application only provides a way to download the mHuBERT model weights and does not distribute them. When you download and use the mHuBERT model weights, please make sure to comply with the terms of the license. + +This application introduces [mHuBERT](https://huggingface.co/utter-project/mHuBERT-147) as an advanced speech feature extraction backend, used to analyze phoneme similarity across different language pronunciations. + +If the song you are creating has broadly similar lyrics to the reference vocal, but differs in musical arrangement (e.g., tempo, local lyric segmentation, etc.), enabling the mHuBERT backend can improve the quality of expression parameter alignment. + +mHuBERT runs well in both CPU and CUDA environments. ## 📌 Use Case @@ -67,15 +84,9 @@ When using a DiffSinger virtual singer for covers, users often already have an O ### Inputs -> [!TIP] -> Starting from `v0.6.0`, this application supports OpenUtau voice tracks with **multiple parts** and **multiple tempos**. - -> [!TIP] -> Starting from `v0.5.0`, users can define a selection region independently within the full audio of both the **Utau vocal** and the **Reference vocal**. The selected audio segment will be used as the final input. - -* **Utau vocal**: Emotionless synthesized vocal output from OpenUtau (WAV format). It is recommended to keep the segmentation and tempo as close to the **Reference vocal** as possible, as large discrepancies may affect alignment quality. +* **Utau vocal**: Emotionless synthesized vocal output from OpenUtau (WAV format). It is recommended to keep the segmentation and tempo as close to the **Reference vocal** as possible, as large discrepancies may affect alignment quality. In such cases, consider enabling the [mHuBERT Feature Extractor](#mhubert-feature-extractor). * **Reference vocal**: Original human vocal recording (WAV format). You can use tools like [UVR](https://github.com/Anjok07/ultimatevocalremovergui) or [MSST](https://github.com/SUC-DriverOld/MSST-WebUI) to remove instrumentals, harmonies, and reverb. -* **Input project**: Original OpenUtau project file (USTX format). +* **Input project**: Original OpenUtau project file (USTX format). This application supports OpenUtau voice tracks with **multiple parts** and **multiple tempos**. * **Output path**: Where the new processed project file will be saved. * **Track number**: The track number in the OpenUtau project where the **Utau vocal** resides (1-based). Expression parameters will be imported into this track. @@ -84,7 +95,7 @@ When using a DiffSinger virtual singer for covers, users often already have an O A new USTX file with expression parameters added. The original project will not be modified. > [!TIP] -> Starting from `v0.9.1`, if you prefer not to generate a new project file, you can **set the output path to be the same as the input project path**. In this case, the expression parameters will be written directly into the original project file. +> If you prefer not to generate a new project file, you can **set the output path to be the same as the input project path**. In this case, the expression parameters will be written directly into the original project file. > > Under normal circumstances, the program will only update the specified expression parameters in the selected track. It will not affect other parameters or modify other tracks. > @@ -100,6 +111,8 @@ A new USTX file with expression parameters added. The original project will not * [x] `Pitch Deviation` generation * [x] `Dynamics` generation * [x] `Tension` generation +* [x] `Breathiness` generation +* [x] `Voicing` generation ## 🚀 Direct Install @@ -115,7 +128,7 @@ CPU-only, no CUDA runtime libraries included. Small installation size, but slowe Expressive CLI / GUI / Viewer installer for Windows x64 architecture with GPU support. -Includes CUDA runtime libraries. When used on a computer with an NVIDIA GPU (driver version >= 450), it significantly improves CREPE backend inference speed. +Includes CUDA runtime libraries. When used on a computer with an NVIDIA GPU (driver version >= 450), it significantly improves CREPE backend inference speed. The mHuBERT feature extractor also benefits from GPU acceleration. ## 👨‍💻 Install from Source @@ -150,7 +163,7 @@ You can also launch a standalone expression curve visualization tool via the `ex ## 📖 Usage > [!TIP] -> All commands described in this section (as well as the executable files installed via the installer) will automatically adapt to your system language. If you need a different language interface, you can set the [`LANGUAGE` or `LANG` environment variable](https://www.gnu.org/software/gettext/manual/html_node/The-LANGUAGE-variable.html) to override the default. +> All Expressive commands described in this section (as well as the executable files installed via the installer) will automatically adapt to your system language. If you need a different language interface, you can set the [`LANGUAGE` or `LANG` environment variable](https://www.gnu.org/software/gettext/manual/html_node/The-LANGUAGE-variable.html) to override the default. > > For example, in Windows PowerShell: > ```powershell @@ -163,15 +176,46 @@ You can also launch a standalone expression curve visualization tool via the `ex > LANGUAGE="en_US" expressive-gui > ``` -> [!IMPORTANT] -> For users who installed from source, when using the [rmvpe-onnx](https://github.com/newcomer00/rmvpe-onnx) backend, the application will automatically download the model file [rmvpe.onnx (Copyright (c) 2022 lj1995 — MIT License)](https://huggingface.co/lj1995/VoiceConversionWebUI/blob/main/rmvpe.onnx) from Hugging Face. +### Get Model Weights + +> [!TIP] +> If you cannot directly access Hugging Face, you can obtain model files via a mirror site. > -> If you wish to download the model file in advance, you can run the following command after installation: -> ```bash +> Using [https://hf-mirror.com/](https://hf-mirror.com/) as an example, in Windows PowerShell: +> ```powershell +> $env:HF_ENDPOINT = "https://hf-mirror.com" > rmvpe-onnx download > ``` > -> If you installed the application via the installer, the model file is already included in the installation package, and no additional download is required. +> In Linux shell: +> ```bash +> HF_ENDPOINT="https://hf-mirror.com" rmvpe-onnx download +> ``` +> +> - The `HF_ENDPOINT` environment variable overrides the default Hugging Face API endpoint, causing model files to be downloaded from the specified mirror site. +> - `rmvpe-onnx download` can be replaced with any command that needs to download model files from Hugging Face, including the commands to launch this application. + +#### RMVPE + +If you installed the application via the installer, the relevant model files are already included in the installation package, and no additional download is required. + +For users who installed from source, when using the [rmvpe-onnx](https://github.com/newcomer00/rmvpe-onnx) backend, the application will automatically download the model file [rmvpe.onnx (Copyright (c) 2022 lj1995 — MIT License)](https://huggingface.co/lj1995/VoiceConversionWebUI/blob/main/rmvpe.onnx) from Hugging Face. + +If you wish to download the model file in advance, you can run the following command after installation: +```bash +rmvpe-onnx download +``` + +#### mHuBERT + +The [mHuBERT model weights](https://huggingface.co/utter-project/mHuBERT-147) are subject to the [CC-BY-NC-SA-4.0](https://creativecommons.org/licenses/by-nc-sa/4.0/) license and are **not distributed with the application**. + +When you use the mHuBERT feature extractor, the application will automatically download the [ONNX-format model weight files](https://huggingface.co/NewComer00/mHuBERT-147-ONNX) from Hugging Face. The weight files will be saved to the default Hugging Face cache path for the current user (i.e., the `~/.cache/huggingface/hub` directory; the exact location will be printed by the program), and will not be stored in the application's directory. + +You can also manually download the model weights using Hugging Face's [hf](https://huggingface.co/docs/huggingface_hub/guides/cli) tool. The files will be saved to the current user's cache directory with no additional configuration needed: +```bash +hf download NewComer00/mHuBERT-147-ONNX onnx/model_bnb4.onnx onnx/model_fp16.onnx onnx/model_fp32.onnx +``` ### Command Line Interface (CLI) @@ -251,7 +295,7 @@ graph TB; refwav[/"Reference WAV"/] utauwav[/"OpenUtau WAV"/] refwav-->feat_pitd - ustx_in-.->|Export|utauwav + ustx_in-.->utauwav utauwav-->feat_pitd ustx_editor["USTX Editor"] @@ -260,7 +304,7 @@ graph TB; subgraph PitdLoader direction TB - feat_pitd["Features Extraction
Pitch & MFCC & RMS"] + feat_pitd["Features Extraction
Pitch & MFCC & RMS
( & mHuBERT Embeddings )"] time_pitd["Time Alignment
FastDTW"] feat_pitd-->time_pitd @@ -272,8 +316,8 @@ graph TB; pitch_algn-->get_pitd end - utsx_out[/"OpenUtau Project Output"/] - get_pitd-->utsx_out + ustx_out[/"OpenUtau Project Output"/] + get_pitd-->ustx_out subgraph DynLoader direction TB @@ -296,6 +340,28 @@ graph TB; get_tenc["Get Tension"] time_tenc-->get_tenc end + + subgraph BrecLoader + direction TB + feat_brec["Features Extraction
HPSS → { Breath Index & Voice Index } & RMS"] + + time_brec["Time Alignment
FastDTW"] + feat_brec-->time_brec + + get_brec["Get Breathiness"] + time_brec-->get_brec + end + + subgraph VoicLoader + direction TB + feat_voic["Features Extraction
HPSS → { Breath Index & Voice Index } & RMS"] + + time_voic["Time Alignment
FastDTW"] + feat_voic-->time_voic + + get_voic["Get Voicing"] + time_voic-->get_voic + end ``` ## ⚠️ Troubleshooting @@ -335,9 +401,6 @@ The extracted PITD expression curve is too flat, with almost no significant vari If the issue persists, try lowering both confidence thresholds. You can use the pitch confidence curve in [`expressive-viewer`](#viewer) as a reference when tuning. In general, the **Utau vocal** is relatively clean, so it is recommended to adjust the confidence threshold for the **reference vocal** first. -#### Future Plans -Incorporate semantic information into the PITD expression extraction algorithm. - --- ### PITD expression curve has sudden jumps or spikes @@ -360,7 +423,4 @@ The PITD expression curve changes too abruptly at certain positions, with large 2. Try denoising the reference audio using tools such as [UVR](https://github.com/Anjok07/ultimatevocalremovergui) or [MSST](https://github.com/SUC-DriverOld/MSST-WebUI). 3. First try using the best-performing **rmvpe-onnx** or **hybrid** backend (with default confidence thresholds). If the issue persists, try increasing both confidence thresholds. You can use the pitch confidence curve in [`expressive-viewer`](#viewer) to guide your adjustments. - In general, the **Utau vocal** is relatively clean, so it is recommended to adjust the confidence threshold for the **reference vocal** first. - -#### Future Plans -Incorporate semantic information into the PITD expression extraction algorithm. + In general, the **Utau vocal** is relatively clean, so it is recommended to adjust the confidence diff --git a/README.md b/README.md index 29f1178..0ae9c76 100644 --- a/README.md +++ b/README.md @@ -22,9 +22,11 @@ 当前版本支持以下表情参数的导入: +* `Breathiness (curve)` * `Dynamics (curve)` * `Pitch Deviation (curve)` * `Tension (curve)` +* `Voicing (curve)`
@@ -51,13 +53,28 @@ * OpenUtau Beta(或支持 DiffSinger 的其他版本) * Python 3.10 \* +> \* 在 Windows 平台下,TensorFlow 2.10 是最后一个支持 GPU 加速的版本,Python 3.10 是它的 `.whl` 文件支持的最高 Python 版本。 + +### 机器学习模型 + +#### 音高提取 + 本应用默认选择 [rmvpe-onnx](https://github.com/newcomer00/rmvpe-onnx) 作为音高提取后端,仅需 CPU 即可运行。[RMVPE](https://arxiv.org/abs/2306.15412v2) 是目前公开的效果最好的音高提取算法,且推理速度较快,可以满足绝大多数使用场景。 应用也提供了 [swift-f0](https://github.com/lars76/swift-f0) 与 [CREPE](https://github.com/marl/crepe) 音高提取后端。前者仅依赖 CPU,效果一般,但速度最快。后者是业内的经典算法,速度较慢。在 CUDA 环境下,CREPE 后端会自动启用 GPU 加速。 应用还新增了一个实验性的 **hybrid** 后端。该后端融合了 rmvpe-onnx 与 swift-f0 的预测结果,以 rmvpe-onnx 的音高提取结果为主,在音频有声段中,如果 rmvpe-onnx 的置信度较低且 swift-f0 的置信度较高,则采用 swift-f0 的结果进行修正,从而提升整体音高提取的准确性。 -> \* 在 Windows 平台下,TensorFlow 2.10 是最后一个支持 GPU 加速的版本,Python 3.10 是它的 `.whl` 文件支持的最高 Python 版本。 +#### mHuBERT 特征提取器 + +> [!IMPORTANT] +> mHuBERT 模型权重遵循 [CC-BY-NC-SA-4.0](https://creativecommons.org/licenses/by-nc-sa/4.0/) 许可证分发,**仅供非商业用途使用**。本应用只提供 mHuBERT 模型权重的下载方式,不分发模型权重。您下载并使用 mHuBERT 模型权重时,请务必遵守其许可证的规定。 + +本应用引入了 [mHuBERT](https://huggingface.co/utter-project/mHuBERT-147) 作为高级语音特征提取后端,用于分析不同语言发音的音素相似度。 + +如果您创作的歌曲与参考歌手人声的歌词大体相似,但乐曲编排(如曲速、局部歌词的切分方式等)不同,那么启用 mHuBERT 后端会提升表情参数的对齐效果。 + +mHuBERT 在 CPU 与 CUDA 环境下均运行良好。 ## 📌 使用场景 @@ -67,15 +84,9 @@ ### 输入 -> [!TIP] -> 从 `v0.6.0` 开始,本应用支持带有**多分段**与**多曲速**的 OpenUtau 人声音轨。 - -> [!TIP] -> 从 `v0.5.0` 开始,用户可以分别在**歌姬音声**与**参考人声**的完整音频中划定选区,选区内的音频段落将作为最终输入。 - -* **歌姬音声**:由 OpenUtau 输出的无表情虚拟歌声音频(WAV 格式)。建议分段与曲速尽量与**参考人声**相近,若相差过大可能影响对齐效果。 +* **歌姬音声**:由 OpenUtau 输出的无表情虚拟歌声音频(WAV 格式)。建议分段与曲速尽量与**参考人声**相近,若相差过大可能影响对齐效果,此时可考虑启用 [mHuBERT 特征提取器](#mhubert-特征提取器)。 * **参考人声**:原始人声录音(WAV 格式),可使用 [UVR](https://github.com/Anjok07/ultimatevocalremovergui) 、[MSST](https://github.com/SUC-DriverOld/MSST-WebUI) 等工具去除伴奏、和声与混响。 -* **输入工程**:原始 OpenUtau 工程文件(USTX 格式)。 +* **输入工程**:原始 OpenUtau 工程文件(USTX 格式)。本应用支持带有**多分段**与**多曲速**的 OpenUtau 人声音轨。 * **输出路径**:处理完成后新工程文件的保存位置。 * **音轨编号**:OpenUtau 工程中**歌姬音声**所在的音轨编号(从 1 开始)。表情参数会被导入到该音轨中。 @@ -84,7 +95,7 @@ 一个携带表情参数的新 USTX 文件。原始工程不会被修改。 > [!TIP] -> 从 `v0.9.1` 开始,如果您不希望额外生成新的工程文件,可以**将输出路径设置为与输入工程路径一致**。这样,表情参数会直接写入原始工程文件中。 +> 如果您不希望额外生成新的工程文件,可以**将输出路径设置为与输入工程路径一致**。这样,表情参数会直接写入原始工程文件中。 > > 正常情况下,程序只会更新您所选音轨中的指定表情参数,不会影响其它参数,也不会修改其他音轨。 > @@ -100,20 +111,24 @@ * [x] `Pitch Deviation` 参数生成 * [x] `Dynamics` 参数生成 * [x] `Tension` 参数生成 +* [x] `Breathiness` 参数生成 +* [x] `Voicing` 参数生成 ## 🚀 直接安装 您可以直接在 [Releases](https://github.com/NewComer00/expressive/releases) 页面下载预编译的可执行文件: ### `Expressive--Windows-x64-CPU.exe` + 适用于 x64 架构 Windows 的 Expressive CLI / GUI / Viewer 安装包。 仅可使用 CPU,无 CUDA 运行时库。安装体积小,但选择 CREPE 后端提取音高时速度较慢。 ### `Expressive--Windows-x64-GPU.exe` + 带 GPU 支持的适用于 x64 架构 Windows 的 Expressive CLI / GUI / Viewer 安装包。 -含 CUDA 运行时库。在配备 NVIDIA 显卡(驱动版本 >= 450)的电脑上使用时,会大幅提高 CREPE 后端的推理速度。 +含 CUDA 运行时库。在配备 NVIDIA 显卡(驱动版本 >= 450)的电脑上使用时,会大幅提高 CREPE 后端的推理速度。mHuBERT 特征提取器也可享受 GPU 加速。 ## 👨‍💻 源码安装 @@ -150,7 +165,7 @@ pip install -e ".[gpu,gui]" ## 📖 使用方式 > [!TIP] -> 本节介绍的所有命令(以及您通过安装包安装的可执行文件)均会自动适配您的系统语言。如果您需要不同语言的界面,可以设置[环境变量 `LANGUAGE` 或 `LANG`](https://www.gnu.org/software/gettext/manual/html_node/The-LANGUAGE-variable.html) 来覆盖默认语言。 +> 本节介绍的所有 Expressive 命令(以及您通过安装包安装的可执行文件)均会自动适配您的系统语言。如果您需要不同语言的界面,可以设置[环境变量 `LANGUAGE` 或 `LANG`](https://www.gnu.org/software/gettext/manual/html_node/The-LANGUAGE-variable.html) 来覆盖默认语言。 > > 例如,在 Windows PowerShell 中: > ```powershell @@ -163,15 +178,46 @@ pip install -e ".[gpu,gui]" > LANGUAGE="en_US" expressive-gui > ``` -> [!IMPORTANT] -> 从源码安装的用户在运行 [rmvpe-onnx](https://github.com/newcomer00/rmvpe-onnx) 后端时,应用会自动从 Hugging Face 下载模型文件 [rmvpe.onnx(Copyright (c) 2022 lj1995 — MIT License)](https://huggingface.co/lj1995/VoiceConversionWebUI/blob/main/rmvpe.onnx)。 -> -> 如果您希望提前下载模型文件,可在安装完成后运行以下命令: -> ```bash +### 获取模型权重 + +> [!TIP] +> 如果您无法直接访问 Hugging Face,可通过镜像站点来获取模型文件。 +> +> 以 [https://hf-mirror.com/](https://hf-mirror.com/) 镜像站点为例,在 Windows PowerShell 中: +> ```powershell +> $env:HF_ENDPOINT = "https://hf-mirror.com" > rmvpe-onnx download > ``` +> +> 在 Linux Shell 中: +> ```bash +> HF_ENDPOINT="https://hf-mirror.com" rmvpe-onnx download +> ``` > -> 若您是通过安装包获取的本应用,安装包中已包含该模型文件,无需额外下载。 +> - `HF_ENDPOINT` 环境变量会覆盖 Hugging Face 的默认 API 端点,使得模型文件从指定的镜像站点下载。 +> - `rmvpe-onnx download` 可以替换为任何需要从 Hugging Face 下载模型文件的命令,包括启动本应用的命令。 + +#### RMVPE + +若您是通过安装包获取的本应用,安装包中已包含相关模型文件,无需额外下载。 + +从源码安装的用户在运行 [rmvpe-onnx](https://github.com/newcomer00/rmvpe-onnx) 后端时,应用会自动从 Hugging Face 下载模型文件 [rmvpe.onnx(Copyright (c) 2022 lj1995 — MIT License)](https://huggingface.co/lj1995/VoiceConversionWebUI/blob/main/rmvpe.onnx)。 + +如果您希望提前下载模型文件,可在安装完成后运行以下命令: +```bash +rmvpe-onnx download +``` + +#### mHuBERT + +[mHuBERT 模型权重](https://huggingface.co/utter-project/mHuBERT-147)受到 [CC-BY-NC-SA-4.0](https://creativecommons.org/licenses/by-nc-sa/4.0/) 许可约束,**不伴随应用分发**。 + +您在使用 mHuBERT 特征提取器时,应用会自动从 Hugging Face 下载 [ONNX 格式的模型权重文件](https://huggingface.co/NewComer00/mHuBERT-147-ONNX)。权重文件会被保存在 Hugging Face 默认的当前用户缓存路径中(即 `~/.cache/huggingface/hub` 目录,程序会输出具体位置),不会存储在本应用的目录里。 + +您也可以使用 Hugging Face 的 [hf](https://huggingface.co/docs/huggingface_hub/guides/cli) 工具手动下载模型权重,文件将被保存在当前用户的缓存目录,无需额外设置: +```bash +hf download NewComer00/mHuBERT-147-ONNX onnx/model_bnb4.onnx onnx/model_fp16.onnx onnx/model_fp32.onnx +``` ### 命令行界面(CLI) @@ -256,7 +302,7 @@ graph TB; refwav[/"Reference WAV"/] utauwav[/"OpenUtau WAV"/] refwav-->feat_pitd - ustx_in-.->|Export|utauwav + ustx_in-.->utauwav utauwav-->feat_pitd ustx_editor["USTX Editor"] @@ -265,7 +311,7 @@ graph TB; subgraph PitdLoader direction TB - feat_pitd["Features Extraction
Pitch & MFCC & RMS"] + feat_pitd["Features Extraction
Pitch & MFCC & RMS
( & mHuBERT Embeddings )"] time_pitd["Time Alignment
FastDTW"] feat_pitd-->time_pitd @@ -277,7 +323,7 @@ graph TB; pitch_algn-->get_pitd end - utsx_out[/"OpenUtau Project Output"/] + ustx_out[/"OpenUtau Project Output"/] get_pitd-->utsx_out subgraph DynLoader @@ -301,6 +347,28 @@ graph TB; get_tenc["Get Tension"] time_tenc-->get_tenc end + + subgraph BrecLoader + direction TB + feat_brec["Features Extraction
HPSS → { Breath Index & Voice Index } & RMS"] + + time_brec["Time Alignment
FastDTW"] + feat_brec-->time_brec + + get_brec["Get Breathiness"] + time_brec-->get_brec + end + + subgraph VoicLoader + direction TB + feat_voic["Features Extraction
HPSS → { Breath Index & Voice Index } & RMS"] + + time_voic["Time Alignment
FastDTW"] + feat_voic-->time_voic + + get_voic["Get Voicing"] + time_voic-->get_voic + end ``` ## ⚠️ 常见问题 @@ -334,9 +402,6 @@ NiceGUI 框架已经开始着手改进文件拖拽支持,应该在未来的版 1. 请下载安装 `v0.9.0` 及之后的版本。 2. 请先尝试使用效果最好的 rmvpe-onnx 或 hybrid 后端(默认置信度阈值)。若问题仍在,尝试降低两个置信度阈值。您可以参考 [`expressive-viewer`](#可视化工具viewer) 的音高置信度曲线来辅助调整。一般来说,**歌姬音声**比较纯净,可以先调整**参考人声**的置信度阈值。 -#### 未来计划 -为 PITD 表情提取算法引入语义信息。 - --- ### PITD 表情曲线在某些位置变化过快,出现跳跃或毛刺 @@ -353,6 +418,3 @@ PITD 表情曲线在某些位置变化过快,出现非常大的跳跃或毛刺 1. 请下载安装 `v0.9.0` 及之后的版本。 2. 可使用 [UVR](https://github.com/Anjok07/ultimatevocalremovergui) 、[MSST](https://github.com/SUC-DriverOld/MSST-WebUI) 等工具对参考音频去噪声(denoise)。 3. 请先尝试使用效果最好的 rmvpe-onnx 或 hybrid 后端(默认置信度阈值)。若问题仍在,尝试增加两个置信度阈值。您可以参考 [`expressive-viewer`](#可视化工具viewer) 的音高置信度曲线来辅助调整。一般来说,**歌姬音声**比较纯净,可以先调整**参考人声**的置信度阈值。 - -#### 未来计划 -为 PITD 表情提取算法引入语义信息。 diff --git a/examples/Lullaby for a Soldier/README.md b/examples/Lullaby for a Soldier/README.md new file mode 100644 index 0000000..8ad0a02 --- /dev/null +++ b/examples/Lullaby for a Soldier/README.md @@ -0,0 +1,16 @@ +# Example: Lullaby for a Soldier + +## Audio +- **Source:** [Lullaby for a Soldier (Arms of the Angels) - Maggie Siff](https://www.youtube.com/watch?v=KmucVmr60HQ) +- **Artist:** Maggie Siff (featuring The Forest Rangers) +- **Composer/Lyricist:** Dillon O'Brian +- **Label:** Columbia Records +- **Album:** *Songs of Anarchy, Vol. 3 (Music from "Sons of Anarchy")* (2013) + +## Voicebank +- **Source:** [泠鸢yousa DiffSinger V1.5](https://github.com/yousa-ling-official-production/yousa-ling-diffsinger-v1) +- **Voice Provider:** 泠鸢yousa + +## OpenUtau +- **Phonemizer:** DiffSinger English+ +- **Tested on version:** 0.1.565.0 diff --git a/examples/Lullaby for a Soldier/expressive_config.brec.json b/examples/Lullaby for a Soldier/expressive_config.brec.json new file mode 100644 index 0000000..355c663 --- /dev/null +++ b/examples/Lullaby for a Soldier/expressive_config.brec.json @@ -0,0 +1,59 @@ +{ + "utau_wav": "examples/Lullaby for a Soldier/utau.wav", + "ref_wav": "examples/Lullaby for a Soldier/reference.wav", + "ustx_input": "examples/Lullaby for a Soldier/project.ustx", + "ustx_output": "examples/Lullaby for a Soldier/output.ustx", + "track_number": 1, + "ref_start": "0:00.50", + "ref_end": "0:38.84", + "utau_start": "0:00.42", + "utau_end": "0:38.39", + "expressions": { + "brec": { + "selected": true, + "align_radius": 1, + "smoothness": 4, + "scaler": 2.5, + "bias": 20, + "spline_smoothing": true + }, + "dyn": { + "selected": true, + "trim_silence": true, + "align_radius": 1, + "smoothness": 2, + "scaler": 0.5, + "spline_smoothing": true + }, + "pitd": { + "selected": true, + "backend": "hybrid", + "confidence_utau": null, + "confidence_ref": null, + "align_radius": 1, + "semitone_shift": 0, + "smoothness": 2, + "scaler": 1.0, + "spline_smoothing": true, + "mhubert_embedder": "bnb4" + }, + "tenc": { + "selected": true, + "trim_silence": true, + "align_radius": 1, + "smoothness": 6, + "scaler": 1.0, + "bias": -30, + "spline_smoothing": true + }, + "voic": { + "selected": false, + "align_radius": 1, + "smoothness": 4, + "dynamic_range": 1.5, + "bias": -30, + "spline_smoothing": true, + "scaler": 1.0 + } + } +} \ No newline at end of file diff --git a/examples/Lullaby for a Soldier/expressive_config.voic.json b/examples/Lullaby for a Soldier/expressive_config.voic.json new file mode 100644 index 0000000..b74ffdc --- /dev/null +++ b/examples/Lullaby for a Soldier/expressive_config.voic.json @@ -0,0 +1,59 @@ +{ + "utau_wav": "examples/Lullaby for a Soldier/utau.wav", + "ref_wav": "examples/Lullaby for a Soldier/reference.wav", + "ustx_input": "examples/Lullaby for a Soldier/project.ustx", + "ustx_output": "examples/Lullaby for a Soldier/output.ustx", + "track_number": 1, + "ref_start": "0:00.50", + "ref_end": "0:38.84", + "utau_start": "0:00.42", + "utau_end": "0:38.39", + "expressions": { + "brec": { + "selected": false, + "align_radius": 1, + "smoothness": 4, + "scaler": 1.0, + "bias": 0, + "spline_smoothing": true + }, + "dyn": { + "selected": true, + "trim_silence": true, + "align_radius": 1, + "smoothness": 2, + "scaler": 0.5, + "spline_smoothing": true + }, + "pitd": { + "selected": true, + "backend": "hybrid", + "confidence_utau": null, + "confidence_ref": null, + "align_radius": 1, + "semitone_shift": 0, + "smoothness": 2, + "scaler": 1.0, + "spline_smoothing": true, + "mhubert_embedder": "bnb4" + }, + "tenc": { + "selected": true, + "trim_silence": true, + "align_radius": 1, + "smoothness": 6, + "scaler": 1.0, + "bias": -30, + "spline_smoothing": true + }, + "voic": { + "selected": true, + "align_radius": 1, + "smoothness": 4, + "dynamic_range": 1.5, + "bias": -30, + "spline_smoothing": true, + "scaler": 1.0 + } + } +} \ No newline at end of file diff --git a/examples/Lullaby for a Soldier/project.ustx b/examples/Lullaby for a Soldier/project.ustx new file mode 100644 index 0000000..d72e700 --- /dev/null +++ b/examples/Lullaby for a Soldier/project.ustx @@ -0,0 +1,1059 @@ +name: New Project +comment: "" +output_dir: Vocal +cache_dir: UCache +ustx_version: "0.9" +bpm: 120 +beat_per_bar: 4 +beat_unit: 4 +expressions: + dyn: + name: dynamics (curve) + abbr: dyn + type: Curve + min: -240 + max: 120 + default_value: 0 + is_flag: false + flag: "" + skip_output_if_default: false + pitd: + name: pitch deviation (curve) + abbr: pitd + type: Curve + min: -1200 + max: 1200 + default_value: 0 + is_flag: false + flag: "" + skip_output_if_default: false + clr: + name: voice color + abbr: clr + type: Options + min: 0 + max: -1 + default_value: 0 + is_flag: false + options: [] + skip_output_if_default: false + eng: + name: resampler engine + abbr: eng + type: Options + min: 0 + max: 1 + default_value: 0 + is_flag: false + options: + - "" + - worldline + skip_output_if_default: false + vel: + name: velocity + abbr: vel + type: Numerical + min: 0 + max: 200 + default_value: 100 + is_flag: false + flag: "" + skip_output_if_default: false + vol: + name: volume + abbr: vol + type: Numerical + min: 0 + max: 200 + default_value: 100 + is_flag: false + flag: "" + skip_output_if_default: false + atk: + name: attack + abbr: atk + type: Numerical + min: 0 + max: 200 + default_value: 100 + is_flag: false + flag: "" + skip_output_if_default: false + dec: + name: decay + abbr: dec + type: Numerical + min: 0 + max: 100 + default_value: 0 + is_flag: false + flag: "" + skip_output_if_default: false + gen: + name: gender + abbr: gen + type: Numerical + min: -100 + max: 100 + default_value: 0 + is_flag: true + flag: g + skip_output_if_default: false + genc: + name: gender (curve) + abbr: genc + type: Curve + min: -100 + max: 100 + default_value: 0 + is_flag: false + flag: "" + skip_output_if_default: false + bre: + name: breath + abbr: bre + type: Numerical + min: 0 + max: 100 + default_value: 0 + is_flag: true + flag: B + skip_output_if_default: false + brec: + name: breathiness (curve) + abbr: brec + type: Curve + min: -100 + max: 100 + default_value: 0 + is_flag: false + flag: "" + skip_output_if_default: false + lpf: + name: lowpass + abbr: lpf + type: Numerical + min: 0 + max: 100 + default_value: 0 + is_flag: true + flag: H + skip_output_if_default: false + norm: + name: normalize + abbr: norm + type: Numerical + min: 0 + max: 100 + default_value: 86 + is_flag: true + flag: P + skip_output_if_default: false + mod: + name: modulation + abbr: mod + type: Numerical + min: 0 + max: 100 + default_value: 0 + is_flag: false + flag: "" + skip_output_if_default: false + mod+: + name: modulation plus + abbr: mod+ + type: Numerical + min: 0 + max: 100 + default_value: 0 + is_flag: false + flag: "" + skip_output_if_default: false + alt: + name: alternate + abbr: alt + type: Numerical + min: 0 + max: 16 + default_value: 0 + is_flag: false + flag: "" + skip_output_if_default: false + dir: + name: direct + abbr: dir + type: Options + min: 0 + max: 1 + default_value: 0 + is_flag: false + options: + - off + - on + skip_output_if_default: false + shft: + name: tone shift + abbr: shft + type: Numerical + min: -36 + max: 36 + default_value: 0 + is_flag: false + flag: "" + skip_output_if_default: false + shfc: + name: tone shift (curve) + abbr: shfc + type: Curve + min: -1200 + max: 1200 + default_value: 0 + is_flag: false + flag: "" + skip_output_if_default: false + tenc: + name: tension (curve) + abbr: tenc + type: Curve + min: -100 + max: 100 + default_value: 0 + is_flag: false + flag: "" + skip_output_if_default: false + voic: + name: voicing (curve) + abbr: voic + type: Curve + min: 0 + max: 100 + default_value: 100 + is_flag: false + flag: "" + skip_output_if_default: false +exp_selectors: +- dyn +- pitd +- tenc +- brec +- voic +- clr +- atk +- dec +- gen +- bre +exp_primary: 4 +exp_secondary: 3 +key: 0 +time_signatures: +- bar_position: 0 + beat_per_bar: 4 + beat_unit: 4 +tempos: +- position: 0 + bpm: 120 +tracks: +- singer: yousaV1.5 + phonemizer: OpenUtau.Core.DiffSinger.DiffSingerARPAPlusEnglishPhonemizer + renderer_settings: + renderer: DIFFSINGER + track_name: main + track_color: Blue + mute: false + solo: false + volume: 0 + pan: 0 + track_expressions: [] + voice_color_names: + - 01:Bright + - 02:Cute + - 03:Normal + - 04:Whisper + - 05:Joyful + - 06:Classic +- phonemizer: OpenUtau.Core.DefaultPhonemizer + renderer_settings: {} + track_name: reference + track_color: Blue + mute: true + solo: false + volume: 0 + pan: 0 + track_expressions: [] + voice_color_names: + - "" +voice_parts: +- duration: 40320 + name: main + comment: "" + track_no: 0 + position: 0 + notes: + - position: 720 + duration: 360 + tone: 62 + lyric: may + pitch: + data: + - {x: -40, y: 0, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: + - index: 0 + offset: -43 + - position: 1080 + duration: 360 + tone: 60 + lyric: you + pitch: + data: + - {x: -40, y: 20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: [] + - position: 1440 + duration: 600 + tone: 58 + lyric: dream + pitch: + data: + - {x: -40, y: 20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + phoneme_overrides: + - index: 2 + offset: -135 + - position: 2040 + duration: 480 + tone: 58 + lyric: bring + pitch: + data: + - {x: -40, y: 0, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + - {index: 3, abbr: clr, value: 2} + phoneme_overrides: + - index: 0 + offset: -39 + - index: 2 + offset: 29 + - position: 2520 + duration: 600 + tone: 60 + lyric: you + pitch: + data: + - {x: -40, y: -20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: [] + - position: 3120 + duration: 240 + tone: 60 + lyric: peace + pitch: + data: + - {x: -40, y: 0, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + phoneme_overrides: + - index: 2 + offset: 136 + - position: 3360 + duration: 600 + tone: 62 + lyric: + + pitch: + data: + - {x: -40, y: -20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: [] + phoneme_overrides: [] + - position: 4440 + duration: 240 + tone: 60 + lyric: in + pitch: + data: + - {x: -40, y: 0, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: [] + - position: 4680 + duration: 360 + tone: 58 + lyric: the + pitch: + data: + - {x: -40, y: 20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: + - index: 0 + offset: -15 + - position: 5040 + duration: 1320 + tone: 55 + lyric: dark + pitch: + data: + - {x: -40, y: 30, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + - {index: 3, abbr: clr, value: 2} + phoneme_overrides: + - index: 3 + offset: 458 + - index: 2 + offset: 284 + - position: 6840 + duration: 600 + tone: 53 + lyric: ness + pitch: + data: + - {x: -40, y: 0, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + phoneme_overrides: + - index: 2 + offset: 145 + - position: 9960 + duration: 360 + tone: 62 + lyric: may + pitch: + data: + - {x: -40, y: 0, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: + - index: 0 + offset: -32 + - position: 10320 + duration: 240 + tone: 60 + lyric: you + pitch: + data: + - {x: -40, y: 20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: [] + - position: 10560 + duration: 480 + tone: 58 + lyric: always + pitch: + data: + - {x: -40, y: 20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + - {index: 3, abbr: clr, value: 2} + - {index: 4, abbr: clr, value: 2} + phoneme_overrides: + - index: 1 + offset: -261 + - index: 2 + offset: -97 + - index: 3 + offset: -335 + - index: 4 + offset: -116 + - position: 11040 + duration: 600 + tone: 58 + lyric: + + pitch: + data: + - {x: -40, y: 0, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: [] + phoneme_overrides: [] + - position: 11640 + duration: 600 + tone: 60 + lyric: rise + pitch: + data: + - {x: -40, y: -20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + phoneme_overrides: + - index: 0 + offset: -41 + - position: 12240 + duration: 360 + tone: 60 + lyric: over + pitch: + data: + - {x: -40, y: 0, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + phoneme_overrides: + - index: 2 + offset: 387 + - index: 1 + offset: 350 + - position: 12600 + duration: 360 + tone: 62 + lyric: + + pitch: + data: + - {x: -40, y: -20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: [] + phoneme_overrides: [] + - position: 12960 + duration: 960 + tone: 60 + lyric: + + pitch: + data: + - {x: -40, y: 20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: [] + phoneme_overrides: [] + - position: 13920 + duration: 360 + tone: 58 + lyric: the + pitch: + data: + - {x: -40, y: 20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: [] + - position: 14280 + duration: 1200 + tone: 55 + lyric: rain + pitch: + data: + - {x: -40, y: 30, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + phoneme_overrides: [] + - position: 17880 + duration: 360 + tone: 62 + lyric: may + pitch: + data: + - {x: -40, y: 0, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: + - index: 0 + offset: -39 + - position: 18240 + duration: 360 + tone: 63 + lyric: the + pitch: + data: + - {x: -40, y: -10, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: [] + - position: 18600 + duration: 1200 + tone: 65 + lyric: light + pitch: + data: + - {x: -40, y: -20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + phoneme_overrides: + - index: 2 + offset: 518 + - position: 20040 + duration: 240 + tone: 63 + lyric: from + pitch: + data: + - {x: -40, y: 0, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + - {index: 3, abbr: clr, value: 2} + phoneme_overrides: + - index: 3 + offset: -55 + - position: 20280 + duration: 240 + tone: 62 + lyric: above + pitch: + data: + - {x: -40, y: 10, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + - {index: 3, abbr: clr, value: 2} + phoneme_overrides: + - index: 3 + offset: 206 + - position: 20520 + duration: 1440 + tone: 63 + lyric: + + pitch: + data: + - {x: -40, y: -10, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: [] + phoneme_overrides: [] + - position: 22920 + duration: 360 + tone: 62 + lyric: always + pitch: + data: + - {x: -40, y: 0, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + - {index: 3, abbr: clr, value: 2} + - {index: 4, abbr: clr, value: 2} + phoneme_overrides: + - index: 4 + offset: -232 + - index: 1 + offset: -214 + - index: 2 + offset: -104 + - index: 3 + offset: -341 + - index: 0 + offset: -47 + - position: 23280 + duration: 600 + tone: 58 + lyric: + + pitch: + data: + - {x: -40, y: 40, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: [] + phoneme_overrides: [] + - position: 23880 + duration: 600 + tone: 58 + lyric: lead + pitch: + data: + - {x: -40, y: 0, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + phoneme_overrides: + - index: 0 + offset: -25 + - index: 1 + phoneme: en/iy + - position: 24480 + duration: 480 + tone: 60 + lyric: you + pitch: + data: + - {x: -40, y: -20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: [] + - position: 25200 + duration: 600 + tone: 62 + lyric: to + pitch: + data: + - {x: -40, y: 0, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: [] + - position: 25800 + duration: 840 + tone: 55 + lyric: love + pitch: + data: + - {x: -40, y: 70, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + phoneme_overrides: + - index: 0 + offset: -29 + - position: 28320 + duration: 360 + tone: 58 + lyric: may + pitch: + data: + - {x: -40, y: 0, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: + - index: 0 + offset: -47 + - position: 28680 + duration: 240 + tone: 55 + lyric: you + pitch: + data: + - {x: -40, y: 30, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: [] + - position: 28920 + duration: 840 + tone: 53 + lyric: stay + pitch: + data: + - {x: -40, y: 20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + phoneme_overrides: + - index: 0 + offset: 103 + - index: 2 + offset: 187 + - index: 1 + phoneme: en/d + offset: 74 + - position: 29760 + duration: 600 + tone: 58 + lyric: in + pitch: + data: + - {x: -40, y: -50, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: [] + - position: 30360 + duration: 720 + tone: 62 + lyric: the + pitch: + data: + - {x: -40, y: -40, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: + - index: 1 + phoneme: en/iy + - position: 31080 + duration: 1680 + tone: 65 + lyric: arms + pitch: + data: + - {x: -40, y: -30, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + - {index: 3, abbr: clr, value: 2} + phoneme_overrides: + - index: 1 + offset: -407 + - index: 3 + offset: -117 + - position: 32760 + duration: 360 + tone: 63 + lyric: of + pitch: + data: + - {x: -40, y: 20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: [] + - position: 33120 + duration: 360 + tone: 62 + lyric: the + pitch: + data: + - {x: -40, y: 10, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + phoneme_overrides: + - index: 1 + phoneme: en/iy + - position: 33480 + duration: 1200 + tone: 60 + lyric: android + pitch: + data: + - {x: -40, y: 20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: + - {index: 0, abbr: clr, value: 2} + - {index: 1, abbr: clr, value: 2} + - {index: 2, abbr: clr, value: 2} + - {index: 3, abbr: clr, value: 2} + - {index: 4, abbr: clr, value: 4} + phoneme_overrides: + - index: 3 + offset: 974 + - index: 2 + offset: 921 + - index: 4 + offset: 194 + - index: 1 + offset: -463 + - position: 34680 + duration: 2160 + tone: 58 + lyric: + + pitch: + data: + - {x: -40, y: 20, shape: io} + - {x: 40, y: 0, shape: io} + snap_first: true + vibrato: {length: 0, period: 175, depth: 25, in: 10, out: 10, shift: 0, drift: 0, vol_link: 0} + tuning: 0 + phoneme_expressions: [] + phoneme_overrides: [] + curves: + - xs: + - -235 + - 10410 + ys: + - 0 + - 0 + abbr: brec + - xs: [] + ys: [] + abbr: voic + - xs: [] + ys: [] + abbr: tenc +wave_parts: +- name: reference.wav + comment: "" + track_no: 1 + position: 0 + relative_path: reference.wav + file_duration_ms: 39107.9591 + skip: 0 + trim: 0 + fadein: 0 + fadeout: 0 diff --git a/examples/Lullaby for a Soldier/reference.wav b/examples/Lullaby for a Soldier/reference.wav new file mode 100644 index 0000000..561bb55 --- /dev/null +++ b/examples/Lullaby for a Soldier/reference.wav @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5f53d634e3b61b3f1782dcb41fc8bd96a0a810cbeb649705389a601721a98926 +size 6898688 diff --git a/examples/Lullaby for a Soldier/utau.wav b/examples/Lullaby for a Soldier/utau.wav new file mode 100644 index 0000000..83ccedb --- /dev/null +++ b/examples/Lullaby for a Soldier/utau.wav @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5d0a4582c2b00fa65e548097af25bd4ed9ef9071263461b1c1801d5c47eda39f +size 3392620 diff --git a/expressions/brec.py b/expressions/brec.py new file mode 100644 index 0000000..b0d729f --- /dev/null +++ b/expressions/brec.py @@ -0,0 +1,128 @@ +from types import SimpleNamespace + +import numpy as np +from scipy.stats import zscore +from skimage.filters import threshold_otsu + +from .base import ( + Args, + Plot, + ExpressionLoader, + register_expression, +) +from utils.wavtool import ( + extract_wav_rms, + extract_wav_breath_voice, +) +from utils.seqtool import ( + unify_sequence_time, + align_sequence_tick, + seq_dynamics_trends, + seq_spline_smoothing, + gaussian_filter1d_with_nan, +) +from utils.i18n import _, _l +from utils.log import StreamToLogger + + +@register_expression +class BrecLoader(ExpressionLoader): + expression_name = "brec" + expression_info = _l("Breathiness (curve)") + args = SimpleNamespace( + align_radius = Args(name="align_radius" , type=int , default=1 , help=_l("**Radius** for the FastDTW alignment algorithm; larger values allow more flexible alignment but increase computation time")), # noqa: E501 + smoothness = Args(name="smoothness" , type=int , default=4 , help=_l("Controls the **smoothness** of the expression curve using Gaussian filtering. Higher values produce smoother curves but may lose fine detail")), # noqa: E501 + scaler = Args(name="scaler" , type=float, default=1.0 , help=_l("**Scaling factor** applied to the expression curve. Values >1 amplify the expression, =1 keeps original intensity, <1 reduces it")), # noqa: E501 + bias = Args(name="bias" , type=int , default=10 , help=_l("**Bias** offset added to the expression curve. Positive values shift the curve upward; negative values shift it downward")), # noqa: E501 + spline_smoothing = Args(name="spline_smoothing", type=bool , default=True, help=_l("Perform **spline smoothing** on the final expression curve for extra smoothness")), # noqa: E501 + ) + plots = SimpleNamespace( + expression = Plot(tag=expression_info , title=expression_info , x_label=_l("Tick") , y_label=expression_name , legends=[expression_name] ), # noqa: E501 + raw_breath_index = Plot(tag=_l("raw_breath_index") , title=_l("Raw Breath Index") , x_label=_l("Time (s)"), y_label=_l("Breath Index") , legends=[_l("Reference"), _l("UTAU")]), # noqa: E501 + aligned_breath_index = Plot(tag=_l("aligned_breath_index") , title=_l("Aligned Breath Index") , x_label=_l("Tick") , y_label=_l("Breath Index") , legends=[_l("Reference"), _l("UTAU")]), # noqa: E501 + ) + + def get_expression( + self, + align_radius = args.align_radius .default, + smoothness = args.smoothness .default, + scaler = args.scaler .default, + bias = args.bias .default, + spline_smoothing = args.spline_smoothing.default, + ): + self.logger.info(_("Extracting expression...")) + + with StreamToLogger(self.logger, tee=False): + utau_time, utau_bi, utau_vi, utau_features = \ + get_wav_features(wav_path=self.utau_path) + ref_time, ref_bi, ref_vi, ref_features = \ + get_wav_features(wav_path=self.ref_path) + + ( + brec_tick, + (time_aligned_ref_bi, time_aligned_ref_vi, *_unused), + (time_unified_utau_bi, time_unified_utau_vi, *_unused), + ) = align_sequence_tick( + query_time=ref_time, + queries=(ref_bi, ref_vi, *ref_features), + reference_time=utau_time, + references=(utau_bi, utau_vi, *utau_features), + align_radius=align_radius, + ) + + brec_val = get_expression_breathiness(time_aligned_ref_bi, time_aligned_ref_vi, smoothness, scaler, bias) + + if spline_smoothing: + brec_val = seq_spline_smoothing(brec_tick, brec_val, nan_policy='preserve_head_tail') + + self.collect_plot(self.plots.expression, (brec_tick, brec_val)) + self.collect_plot(self.plots.raw_breath_index, (ref_time, ref_bi), (utau_time, utau_bi)) + self.collect_plot(self.plots.aligned_breath_index, (brec_tick, time_aligned_ref_bi), (brec_tick, time_unified_utau_bi)) + + self.expression_tick, self.expression_val = brec_tick, brec_val + self.logger.info(_("Expression extraction complete.")) + return self.expression_tick, self.expression_val + + +def get_wav_features(wav_path): + feature_times = [] + feature_vals = [] + + time, bi, vi = extract_wav_breath_voice(wav_path) + feature_times += [time, time] + feature_vals += [bi, vi] + + bi_trends = seq_dynamics_trends(bi) + feature_times += [time] * len(bi_trends) + feature_vals += list(bi_trends) + + rms_time, rms = extract_wav_rms(wav_path, mask_silence=True) + feature_times += [rms_time] + feature_vals += [rms] + + rms_trends = seq_dynamics_trends(rms) + feature_times += [rms_time] * len(rms_trends) + feature_vals += list(rms_trends) + + wav_time, (wav_bi, wav_vi, *wav_features) = unify_sequence_time( + seq_times=feature_times, seq_vals=feature_vals + ) + return wav_time, wav_bi, wav_vi, wav_features + + +def get_expression_breathiness(breath_index, voice_index, smoothness=4, scaler=1.0, bias=10): + base_scaler = 10.0 + breath_index = breath_index.copy() + + valid = np.isfinite(voice_index) + if valid.any(): + thresh = threshold_otsu(voice_index[valid]) + breath_index[~valid | (voice_index < thresh)] = np.nan + else: + breath_index[:] = np.nan + + smoothed_brec = gaussian_filter1d_with_nan( + base_scaler * zscore(breath_index, nan_policy='omit'), + sigma=smoothness, + ) + return scaler * smoothed_brec + bias diff --git a/expressions/dyn.py b/expressions/dyn.py index 8eb6ad9..cf7423e 100644 --- a/expressions/dyn.py +++ b/expressions/dyn.py @@ -69,7 +69,7 @@ def get_expression( time_aligned_ref_rms[np.isnan(time_unified_utau_rms)] = np.nan # Generate expression curve - dyn_val = get_experssion_dynamics(time_aligned_ref_rms, smoothness, scaler) + dyn_val = get_expression_dynamics(time_aligned_ref_rms, smoothness, scaler) if spline_smoothing: # Final spline smoothing of the expression curve @@ -108,7 +108,7 @@ def get_wav_features(wav_path, mask_silence=True): return wav_time, wav_rms, wav_features -def get_experssion_dynamics(time_aligned_rms, smoothness=2, scaler=1.0): +def get_expression_dynamics(time_aligned_rms, smoothness=2, scaler=1.0): base_scaler = 10.0 smoothed_dyn = gaussian_filter1d_with_nan( base_scaler * zscore(time_aligned_rms, nan_policy='omit'), diff --git a/expressions/pitd.py b/expressions/pitd.py index 81ef51a..0840712 100644 --- a/expressions/pitd.py +++ b/expressions/pitd.py @@ -19,7 +19,7 @@ seq_dynamics_trends, ) from utils.log import StreamToLogger -from utils.wavtool import extract_wav_mfcc, extract_wav_frequency, extract_wav_rms +from utils.wavtool import extract_wav_embedding, extract_wav_mfcc, extract_wav_frequency, extract_wav_rms @register_expression @@ -28,21 +28,41 @@ class PitdLoader(ExpressionLoader): expression_info = _l("Pitch Deviation (curve)") backend_choices = { "rmvpe-onnx": _l("finest accuracy, fast, CPU only (ONNX Runtime)"), - "swift-f0": _l("fair accuracy, fastest, CPU only (ONNX Runtime)"), - "crepe": _l("good accuracy, slow, CPU & NVIDIA GPU (TensorFlow)"), - "hybrid": _l("based on rmvpe-onnx, improved by swift-f0, CPU only (ONNX Runtime)"), + "swift-f0": _l("fair accuracy, fastest, CPU only (ONNX Runtime)"), + "crepe": _l("good accuracy, slow, CPU & NVIDIA GPU (TensorFlow)"), + "hybrid": _l("based on rmvpe-onnx, improved by swift-f0, CPU only (ONNX Runtime)"), } + mhubert_embedder_choices = { + "fp32": _l("(378 MB) maximum accuracy, full size"), + "fp16": _l("(189 MB) high accuracy, medium size"), + "bnb4": _l("(89.7 MB) fair accuracy, small size"), + "disabled": _l("disabled, skip this feature"), + } + mhubert_embedder_warning = _l(""" +mHuBERT model weights are licensed under CC-BY-NC-SA-4.0. +Do NOT use the mHuBERT model weights for commercial purposes. +This software does not grant you rights to the model weights. +You are solely responsible for compliance with the model license: +https://creativecommons.org/licenses/by-nc-sa/4.0/ + +Original model author: utter-project + +Original model page: https://huggingface.co/utter-project/mHuBERT-147 + +ONNX conversion: https://huggingface.co/NewComer00/mHuBERT-147-ONNX +""") confidence_utau_recommended = {"rmvpe-onnx": 0.03, "swift-f0": 0.95, "crepe": 0.80, "hybrid": 0.03} confidence_ref_recommended = {"rmvpe-onnx": 0.03, "swift-f0": 0.93, "crepe": 0.60, "hybrid": 0.03} args = SimpleNamespace( - backend = Args(name="backend" , type=str , default="rmvpe-onnx", choices=list(backend_choices.keys()), help=_lf("**F0 detection backend** for extracting pitch from WAV files. Available options:\n\n%s\n\n", lambda: "\n".join([f"- `{k}`: {v}" for k, v in PitdLoader.backend_choices.items()]))), # noqa: E501 - confidence_utau = Args(name="confidence_utau" , type=float, default=None, help=_lf("Minimum **confidence level** for keeping detected pitch values in the **UTAU** WAV. Lower values retain more frames but may include errors. Omit to use the recommended value for the selected backend:\n\n%s\n\n", lambda: "\n".join([f"- `{k}`: {v}" for k, v in PitdLoader.confidence_utau_recommended.items()]))), # noqa: E501 - confidence_ref = Args(name="confidence_ref" , type=float, default=None, help=_lf("Minimum **confidence level** for keeping detected pitch values in the **reference** WAV. Lower values retain more frames but may include errors. Omit to use the recommended value for the selected backend:\n\n%s\n\n", lambda: "\n".join([f"- `{k}`: {v}" for k, v in PitdLoader.confidence_ref_recommended.items()]))), # noqa: E501 - align_radius = Args(name="align_radius" , type=int , default=1 , help=_l("**Radius** for the FastDTW alignment algorithm; larger values allow more flexible alignment but increase computation time")), # noqa: E501 - semitone_shift = Args(name="semitone_shift" , type=int , default=None, help=_l("**Semitone shift** between the UTAU and reference WAV. If the UTAU WAV is an octave higher than the reference WAV, set to 12; if lower, set to -12. Omit to enable automatic shift estimation")), # noqa: E501 - smoothness = Args(name="smoothness" , type=int , default=2 , help=_l("Controls the **smoothness** of the expression curve using Gaussian filtering. Higher values produce smoother curves but may lose fine detail")), # noqa: E501 - scaler = Args(name="scaler" , type=float, default=1.0 , help=_l("**Scaling factor** applied to the expression curve. Values >1 amplify the expression, =1 keeps original intensity, <1 reduces it")), # noqa: E501 - spline_smoothing = Args(name="spline_smoothing", type=bool , default=True, help=_l("Perform **spline smoothing** on the final expression curve for extra smoothness")), # noqa: E501 + backend = Args(name="backend" , type=str , default="rmvpe-onnx", choices=list(backend_choices.keys()), help=_lf("**F0 detection backend** for extracting pitch from WAV files. Available options:\n\n%s\n\n", lambda: "\n".join([f"- `{k}`: {v}" for k, v in PitdLoader.backend_choices.items()]))), # noqa: E501 + confidence_utau = Args(name="confidence_utau" , type=float, default=None, help=_lf("Minimum **confidence level** for keeping detected pitch values in the **UTAU** WAV. Lower values retain more frames but may include errors. Omit to use the recommended value for the selected backend:\n\n%s\n\n", lambda: "\n".join([f"- `{k}`: {v}" for k, v in PitdLoader.confidence_utau_recommended.items()]))), # noqa: E501 + confidence_ref = Args(name="confidence_ref" , type=float, default=None, help=_lf("Minimum **confidence level** for keeping detected pitch values in the **reference** WAV. Lower values retain more frames but may include errors. Omit to use the recommended value for the selected backend:\n\n%s\n\n", lambda: "\n".join([f"- `{k}`: {v}" for k, v in PitdLoader.confidence_ref_recommended.items()]))), # noqa: E501 + align_radius = Args(name="align_radius" , type=int , default=1 , help=_l("**Radius** for the FastDTW alignment algorithm; larger values allow more flexible alignment but increase computation time")), # noqa: E501 + semitone_shift = Args(name="semitone_shift" , type=int , default=None, help=_l("**Semitone shift** between the UTAU and reference WAV. If the UTAU WAV is an octave higher than the reference WAV, set to 12; if lower, set to -12. Omit to enable automatic shift estimation")), # noqa: E501 + smoothness = Args(name="smoothness" , type=int , default=2 , help=_l("Controls the **smoothness** of the expression curve using Gaussian filtering. Higher values produce smoother curves but may lose fine detail")), # noqa: E501 + scaler = Args(name="scaler" , type=float, default=1.0 , help=_l("**Scaling factor** applied to the expression curve. Values >1 amplify the expression, =1 keeps original intensity, <1 reduces it")), # noqa: E501 + spline_smoothing = Args(name="spline_smoothing" , type=bool , default=True, help=_l("Perform **spline smoothing** on the final expression curve for extra smoothness")), # noqa: E501 + mhubert_embedder = Args(name="mhubert_embedder" , type=str , default="disabled", choices=list(mhubert_embedder_choices.keys()), help=_lf("**mHuBERT embedder variant** for feature extraction. If not `disabled`, mHuBERT embedder will be downloaded automatically and used for advanced feature extraction. Use this when the difference between UTAU and reference audio so significant that conventional features are not enough. mHuBERT embedder runs well both on CPU and NVIDIA GPU. Available options:\n\n%s\n\n", lambda: "\n".join([f"- `{k}`: {v}" for k, v in PitdLoader.mhubert_embedder_choices.items()]))), # noqa: E501 ) plots = SimpleNamespace( expression = Plot(tag=expression_info , title=expression_info , x_label=_l("Tick") , y_label=expression_name , legends=[expression_name] ), # noqa: E501 @@ -61,9 +81,14 @@ def get_expression( smoothness = args.smoothness .default, scaler = args.scaler .default, spline_smoothing = args.spline_smoothing.default, + mhubert_embedder = args.mhubert_embedder.default, ): self.logger.info(_("Extracting expression...")) + # Show license warning for mhubert_embedder + if mhubert_embedder != "disabled": + self.logger.warning(self.__class__.mhubert_embedder_warning) + # Resolve per-backend confidence defaults if confidence_utau is None: confidence_utau = self.__class__.confidence_utau_recommended[backend] @@ -71,12 +96,14 @@ def get_expression( confidence_ref = self.__class__.confidence_ref_recommended[backend] # Extract pitch features from WAV files - with StreamToLogger(self.logger, tee=True): + with StreamToLogger(self.logger, tee=False): utau_time, utau_pitch, utau_confidence, utau_features = get_wav_features( - wav_path=self.utau_path, confidence_threshold=confidence_utau, backend=backend + wav_path=self.utau_path, confidence_threshold=confidence_utau, + backend=backend, mhubert_embedder=mhubert_embedder, ) ref_time, ref_pitch, ref_confidence, ref_features = get_wav_features( - wav_path=self.ref_path, confidence_threshold=confidence_ref, backend=backend + wav_path=self.ref_path, confidence_threshold=confidence_ref, + backend=backend, mhubert_embedder=mhubert_embedder, ) # Align all sequences to a common MIDI tick time base. @@ -92,7 +119,7 @@ def get_expression( ) # Align pitch sequences along the pitch axis - with StreamToLogger(self.logger, tee=True): + with StreamToLogger(self.logger, tee=False): time_pitch_aligned_ref_pitch, _unused = align_sequence_pitch( time_aligned_ref_pitch, unified_utau_pitch, @@ -124,7 +151,13 @@ def get_expression( return self.expression_tick, self.expression_val -def get_wav_features(wav_path, backend="rmvpe-onnx", confidence_threshold=0.8, confidence_filter_size=9): +def get_wav_features( + wav_path, + backend="rmvpe-onnx", + confidence_threshold=0.8, + confidence_filter_size=9, + mhubert_embedder="disabled", +): """Extract features from a WAV file. Args: @@ -132,6 +165,7 @@ def get_wav_features(wav_path, backend="rmvpe-onnx", confidence_threshold=0.8, c backend (str, optional): F0 detection backend ("crepe" or "swift-f0" or "rmvpe-onnx"). Defaults to "rmvpe-onnx". confidence_threshold (float, optional): Confidence threshold for pitch detection. Defaults to 0.8. confidence_filter_size (int, optional): Size of the median filter for confidence. Defaults to 9. + mhubert_embedder (str, optional): Model type of mHuBERT embedder. Default to "disabled". Returns: tuple: (wav_time, wav_pitch, wav_confidence, wav_features) @@ -170,6 +204,11 @@ def get_wav_features(wav_path, backend="rmvpe-onnx", confidence_threshold=0.8, c feature_times += [rms_time] * len(rms_dynamics_trends) feature_vals += list(rms_dynamics_trends) + if mhubert_embedder != "disabled": + embedding_time, embedding = extract_mhubert_embedding(wav_path, variant=mhubert_embedder) + feature_times += [embedding_time] * len(embedding) + feature_vals += list(embedding) + wav_time, (wav_pitch, wav_confidence, *wav_features) = unify_sequence_time( seq_times=feature_times, seq_vals=feature_vals ) @@ -193,7 +232,7 @@ def align_sequence_pitch(query, reference, semitone_shift=None): base_pitch_vocal = np.nanmedian(reference) semitone_shift = int( np.round(hz_to_midi(base_pitch_vocal)) - - np.round(hz_to_midi(base_pitch_wav)).astype(int) + - np.round(hz_to_midi(base_pitch_wav)) ) print(_("Estimated Semitone-shift: {}").format(semitone_shift)) @@ -224,3 +263,9 @@ def get_pitch_delta(query, reference, smoothness=2, scaler=1.0): delta = gaussian_filter1d_with_nan(delta, sigma=smoothness) return scaler * delta + + +def extract_mhubert_embedding(wav_path, variant="bnb4"): + embedding_time, embedding = extract_wav_embedding( + wav_path, embedder="mhubert", device="cuda", variant=variant) + return embedding_time, embedding diff --git a/expressions/tenc.py b/expressions/tenc.py index d9a0ac9..1f63e29 100644 --- a/expressions/tenc.py +++ b/expressions/tenc.py @@ -71,7 +71,7 @@ def get_expression( time_aligned_ref_rms[np.isnan(time_unified_utau_rms)] = np.nan # Generate expression curve - tenc_val = get_experssion_tension(time_aligned_ref_rms, smoothness, scaler, bias) + tenc_val = get_expression_tension(time_aligned_ref_rms, smoothness, scaler, bias) if spline_smoothing: # Final spline smoothing of the expression curve @@ -107,7 +107,7 @@ def get_wav_features(wav_path, mask_silence=True): return wav_time, wav_rms, wav_features -def get_experssion_tension(time_aligned_rms, smoothness=2, scaler=1.0, bias=0): +def get_expression_tension(time_aligned_rms, smoothness=2, scaler=1.0, bias=0): base_scaler = 10.0 smoothed_tenc = gaussian_filter1d_with_nan( base_scaler * zscore(time_aligned_rms, nan_policy='omit'), diff --git a/expressions/voic.py b/expressions/voic.py new file mode 100644 index 0000000..e85faa6 --- /dev/null +++ b/expressions/voic.py @@ -0,0 +1,129 @@ +from types import SimpleNamespace + +import numpy as np +from scipy.stats import zscore +from skimage.filters import threshold_otsu + +from .base import ( + Args, + Plot, + ExpressionLoader, + register_expression, +) +from utils.wavtool import ( + extract_wav_rms, + extract_wav_breath_voice, +) +from utils.seqtool import ( + unify_sequence_time, + align_sequence_tick, + seq_dynamics_trends, + seq_spline_smoothing, + gaussian_filter1d_with_nan, +) +from utils.i18n import _, _l +from utils.log import StreamToLogger + + +@register_expression +class VoicLoader(ExpressionLoader): + expression_name = "voic" + expression_info = _l("Voicing (curve)") + args = SimpleNamespace( + align_radius = Args(name="align_radius" , type=int , default=1 , help=_l("**Radius** for the FastDTW alignment algorithm; larger values allow more flexible alignment but increase computation time")), # noqa: E501 + smoothness = Args(name="smoothness" , type=int , default=4 , help=_l("Controls the **smoothness** of the expression curve using Gaussian filtering. Higher values produce smoother curves but may lose fine detail")), # noqa: E501 + dynamic_range = Args(name="dynamic_range" , type=float, default=1.0 , help=_l("**Dynamic range** of the expression curve. Values >1 amplify the variation, =1 keeps original range, <1 compresses it")), # noqa: E501 + bias = Args(name="bias" , type=int , default=-10 , help=_l("**Bias** offset added to the expression curve. Positive values shift the curve upward; negative values shift it downward")), # noqa: E501 + spline_smoothing = Args(name="spline_smoothing", type=bool , default=True, help=_l("Perform **spline smoothing** on the final expression curve for extra smoothness")), # noqa: E501 + ) + plots = SimpleNamespace( + expression = Plot(tag=expression_info , title=expression_info , x_label=_l("Tick") , y_label=expression_name , legends=[expression_name] ), # noqa: E501 + raw_voice_index = Plot(tag=_l("raw_voice_index") , title=_l("Raw Voice Index") , x_label=_l("Time (s)"), y_label=_l("Voice Index") , legends=[_l("Reference"), _l("UTAU")]), # noqa: E501 + aligned_voice_index = Plot(tag=_l("aligned_voice_index") , title=_l("Aligned Voice Index") , x_label=_l("Tick") , y_label=_l("Voice Index") , legends=[_l("Reference"), _l("UTAU")]), # noqa: E501 + ) + + def get_expression( + self, + align_radius = args.align_radius .default, + smoothness = args.smoothness .default, + dynamic_range = args.dynamic_range .default, + bias = args.bias .default, + spline_smoothing = args.spline_smoothing.default, + ): + self.logger.info(_("Extracting expression...")) + + with StreamToLogger(self.logger, tee=False): + utau_time, utau_bi, utau_vi, utau_features = \ + get_wav_features(wav_path=self.utau_path) + ref_time, ref_bi, ref_vi, ref_features = \ + get_wav_features(wav_path=self.ref_path) + + ( + voic_tick, + (time_aligned_ref_bi, time_aligned_ref_vi, *_unused), + (time_unified_utau_bi, time_unified_utau_vi, *_unused), + ) = align_sequence_tick( + query_time=ref_time, + queries=(ref_bi, ref_vi, *ref_features), + reference_time=utau_time, + references=(utau_bi, utau_vi, *utau_features), + align_radius=align_radius, + ) + + voic_val = get_expression_voicing(time_aligned_ref_vi, smoothness, dynamic_range, bias) + + if spline_smoothing: + voic_val = seq_spline_smoothing(voic_tick, voic_val, nan_policy='preserve_head_tail') + + self.collect_plot(self.plots.expression, (voic_tick, voic_val)) + self.collect_plot(self.plots.raw_voice_index, (ref_time, ref_vi), (utau_time, utau_vi)) + self.collect_plot(self.plots.aligned_voice_index, (voic_tick, time_aligned_ref_vi), (voic_tick, time_unified_utau_vi)) + + self.expression_tick, self.expression_val = voic_tick, voic_val + self.logger.info(_("Expression extraction complete.")) + return self.expression_tick, self.expression_val + + +def get_wav_features(wav_path): + feature_times = [] + feature_vals = [] + + time, bi, vi = extract_wav_breath_voice(wav_path) + feature_times += [time, time] + feature_vals += [bi, vi] + + bi_trends = seq_dynamics_trends(bi) + feature_times += [time] * len(bi_trends) + feature_vals += list(bi_trends) + + rms_time, rms = extract_wav_rms(wav_path, mask_silence=True) + feature_times += [rms_time] + feature_vals += [rms] + + rms_trends = seq_dynamics_trends(rms) + feature_times += [rms_time] * len(rms_trends) + feature_vals += list(rms_trends) + + wav_time, (wav_bi, wav_vi, *wav_features) = unify_sequence_time( + seq_times=feature_times, seq_vals=feature_vals + ) + return wav_time, wav_bi, wav_vi, wav_features + + +def get_expression_voicing(voice_index, smoothness=4, dynamic_range=1.0, bias=-10): + base_scaler = 10.0 + base_bias = 100.0 + voice_index = voice_index.copy() + + valid = np.isfinite(voice_index) + if not valid.any(): + return np.zeros_like(voice_index) + + thresh = threshold_otsu(voice_index[valid]) + voice_index[~valid | (voice_index < thresh)] = np.nan + + smoothed_voic = gaussian_filter1d_with_nan( + base_bias + dynamic_range * base_scaler * zscore(voice_index, nan_policy='omit'), + sigma=smoothness, + ) + return smoothed_voic + bias diff --git a/expressive.py b/expressive.py index 1baa320..0cdddb8 100644 --- a/expressive.py +++ b/expressive.py @@ -7,6 +7,7 @@ from shutil import copy, SameFileError from os.path import splitext, basename +from utils.gpu import add_cuda_to_path from utils.i18n import init_gettext, _ from utils.fs import APP_LOG_DIR from __version__ import VERSION @@ -77,6 +78,9 @@ def process_expressions( ] ``` """ + # Ensure CUDA libraries are available before any ML frameworks are imported + add_cuda_to_path(skip_missing=True) + try: copy(ustx_input, ustx_output) except SameFileError: @@ -107,27 +111,34 @@ def setup_loggers(): log_dir.mkdir(exist_ok=True, parents=True) log_path = log_dir / f"{datetime.now().strftime('%Y%m%d_%H%M%S')}.log" - formatter = logging.Formatter('%(asctime)s - %(name)s - %(levelname)s - %(message)s') + formatter_file = logging.Formatter('%(asctime)s %(levelname)s [%(name)s]: %(message)s') + formatter_app = logging.Formatter("%(asctime)s %(levelname)s [%(name)s]: %(message)s", datefmt="%H:%M:%S") + formatter_exp = logging.Formatter("%(asctime)s %(levelname)s [%(expression)s]: %(message)s", datefmt="%H:%M:%S") # Set up file handler file_handler = logging.FileHandler(log_path, encoding="utf-8-sig") file_handler.setLevel(logging.DEBUG) - file_handler.setFormatter(formatter) + file_handler.setFormatter(formatter_file) # Set up stdout handler - stream_handler = logging.StreamHandler() - stream_handler.setLevel(logging.DEBUG) + app_handler = logging.StreamHandler() + app_handler.setLevel(logging.DEBUG) + app_handler.setFormatter(formatter_app) + exp_handler = logging.StreamHandler() + exp_handler.setLevel(logging.DEBUG) + exp_handler.setFormatter(formatter_exp) # Main application logger logger_app = logging.getLogger(splitext(basename(__file__))[0]) logger_app.setLevel(logging.DEBUG) logger_app.addHandler(file_handler) - logger_app.addHandler(stream_handler) + logger_app.addHandler(app_handler) # Expression loader logger logger_exp = logging.getLogger(getExpressionLoader(None).__name__) logger_exp.setLevel(logging.DEBUG) logger_exp.addHandler(file_handler) + logger_exp.addHandler(exp_handler) try: yield logger_app, logger_exp, log_path @@ -194,8 +205,7 @@ def main(): expressions, ) except Exception as e: - logger_app.error(_("Error occurred during processing: {e}").format(e=e)) - raise + logger_app.exception(_("Error occurred during processing: {e}").format(e=e)) else: logger_app.info(_("Processing completed successfully!")) diff --git a/expressive_gui.py b/expressive_gui.py index e635413..592aecf 100644 --- a/expressive_gui.py +++ b/expressive_gui.py @@ -537,6 +537,40 @@ async def update_placeholders(): state["expressions"][exp_name], "selected" ) + # Brec parameters + brec_args = getExpressionLoader("brec").args + brec_info = getExpressionLoader("brec").expression_info + with ui.card().classes("w-full").bind_visibility_from( + state["expressions"]["brec"], "selected" + ): + ui.label(brec_info).classes("text-lg font-bold") + + with ui.grid(columns=2).classes("w-full"): + ui.switch(_("Spline Smoothing")).bind_value( + state["expressions"]["brec"], "spline_smoothing", + ).tooltip_md(brec_args.spline_smoothing.help) + + with ui.grid(columns=3).classes("w-full"): + ui.number(label=_("Align Radius"), min=1, format="%d").bind_value( + state["expressions"]["brec"], "align_radius", + forward=lambda v: brec_args.align_radius.type(v) if v is not None else None, + ).tooltip_md(brec_args.align_radius.help) + + ui.number(label=_("Smoothness"), min=0, format="%d").bind_value( + state["expressions"]["brec"], "smoothness", + forward=lambda v: brec_args.smoothness.type(v) if v is not None else None, + ).tooltip_md(brec_args.smoothness.help) + + ui.number(label=_("Scaler"), min=0.0, step=0.1, format="%.1f").bind_value( + state["expressions"]["brec"], "scaler", + forward=lambda v: brec_args.scaler.type(v) if v is not None else None, + ).tooltip_md(brec_args.scaler.help) + + ui.number(label=_("Bias"), format="%d").bind_value( + state["expressions"]["brec"], "bias", + forward=lambda v: brec_args.bias.type(v) if v is not None else None, + ).tooltip_md(brec_args.bias.help) + # Dyn parameters dyn_args = getExpressionLoader("dyn").args dyn_info = getExpressionLoader("dyn").expression_info @@ -578,6 +612,20 @@ async def update_placeholders(): ui.label(pitd_info).classes("text-lg font-bold") with ui.grid(columns=2).classes("w-full"): + with ui.row().classes("w-full items-center gap-2"): + ui.select( + label=_("mHuBERT Embedder"), + options=pitd_args.mhubert_embedder.choices, + ).classes("flex-1").bind_value( + state["expressions"]["pitd"], "mhubert_embedder" + ).tooltip_md(pitd_args.mhubert_embedder.help) + ui.icon("warning", color="warning").tooltip_md( + getExpressionLoader("pitd").mhubert_embedder_warning + ).bind_visibility_from( + state["expressions"]["pitd"], "mhubert_embedder", + backward=lambda v: v != "disabled", + ) + ui.switch(_("Spline Smoothing")).bind_value( state["expressions"]["pitd"], "spline_smoothing", ).tooltip_md(pitd_args.spline_smoothing.help) @@ -672,6 +720,40 @@ def on_backend_change(e): forward=lambda v: tenc_args.bias.type(v) if v is not None else None, ).tooltip_md(tenc_args.bias.help) + # Voic parameters + voic_args = getExpressionLoader("voic").args + voic_info = getExpressionLoader("voic").expression_info + with ui.card().classes("w-full").bind_visibility_from( + state["expressions"]["voic"], "selected" + ): + ui.label(voic_info).classes("text-lg font-bold") + + with ui.grid(columns=2).classes("w-full"): + ui.switch(_("Spline Smoothing")).bind_value( + state["expressions"]["voic"], "spline_smoothing", + ).tooltip_md(voic_args.spline_smoothing.help) + + with ui.grid(columns=3).classes("w-full"): + ui.number(label=_("Align Radius"), min=1, format="%d").bind_value( + state["expressions"]["voic"], "align_radius", + forward=lambda v: voic_args.align_radius.type(v) if v is not None else None, + ).tooltip_md(voic_args.align_radius.help) + + ui.number(label=_("Smoothness"), min=0, format="%d").bind_value( + state["expressions"]["voic"], "smoothness", + forward=lambda v: voic_args.smoothness.type(v) if v is not None else None, + ).tooltip_md(voic_args.smoothness.help) + + ui.number(label=_("Dynamic Range"), min=0.0, step=0.1, format="%.1f").bind_value( + state["expressions"]["voic"], "dynamic_range", + forward=lambda v: voic_args.dynamic_range.type(v) if v is not None else None, + ).tooltip_md(voic_args.dynamic_range.help) + + ui.number(label=_("Bias"), format="%d").bind_value( + state["expressions"]["voic"], "bias", + forward=lambda v: voic_args.bias.type(v) if v is not None else None, + ).tooltip_md(voic_args.bias.help) + # Add the config buttons above the Process button with ui.row().classes("w-full justify-between"): ui.button( @@ -694,26 +776,9 @@ def on_backend_change(e): process_button = ui.button( _("Process"), on_click=process_files, icon="play_arrow" ).classes("flex-grow") - with ui.element().classes("relative w-full h-40"): + with ui.element().classes("relative w-full h-60"): log_element = ui.log().classes("w-full h-full select-text cursor-text") - # Log auto scrolling on update - # Embed the script in the head of the HTML document - # since the id of the log element is static during the app's lifetime - ui.add_head_html(f''' - - ''') - # Clipboard button ui.button("📋").props("flat dense").classes( "absolute top-0 right-5 m-1 text-sm px-1 py-0.5 opacity-70 hover:opacity-100" diff --git a/locales/app.pot b/locales/app.pot index 253509e..4aaead1 100644 --- a/locales/app.pot +++ b/locales/app.pot @@ -8,7 +8,7 @@ msgid "" msgstr "" "Project-Id-Version: expressive VERSION\n" "Report-Msgid-Bugs-To: EMAIL@ADDRESS\n" -"POT-Creation-Date: 2026-04-15 21:12+0800\n" +"POT-Creation-Date: 2026-04-30 20:29+0800\n" "PO-Revision-Date: YEAR-MO-DA HO:MI+ZONE\n" "Last-Translator: FULL NAME \n" "Language-Team: LANGUAGE \n" @@ -17,40 +17,40 @@ msgstr "" "Content-Transfer-Encoding: 8bit\n" "Generated-By: Babel 2.18.0\n" -#: expressive.py:86 +#: expressive.py:93 #, python-brace-format msgid "Expression '{exp_type}' is not supported." msgstr "" -#: expressive.py:132 +#: expressive.py:146 #, python-brace-format msgid "Logs saved to '{log_path}'" msgstr "" -#: expressive.py:143 +#: expressive.py:157 msgid "Migrate expressions from real singers to DiffSingers (CLI)" msgstr "" -#: expressive.py:152 expressive_gui.py:511 +#: expressive.py:166 expressive_gui.py:511 msgid "Path to save the processed `.ustx` file" msgstr "" -#: expressive.py:160 +#: expressive.py:174 msgid "" "**Expression(s)** to apply. Repeat the flag for multiple expressions " "(e.g., `-e dyn -e pitd`)" msgstr "" -#: expressive.py:184 +#: expressive.py:198 msgid "Starting Expressive CLI..." msgstr "" -#: expressive.py:194 +#: expressive.py:208 #, python-brace-format msgid "Error occurred during processing: {e}" msgstr "" -#: expressive.py:197 expressive_gui.py:235 expressive_gui.py:236 +#: expressive.py:210 expressive_gui.py:235 expressive_gui.py:236 msgid "Processing completed successfully!" msgstr "" @@ -148,63 +148,75 @@ msgstr "" msgid "Expression Selection" msgstr "" -#: expressive_gui.py:549 expressive_gui.py:647 -msgid "Trim Silence" -msgstr "" - -#: expressive_gui.py:552 expressive_gui.py:581 expressive_gui.py:650 +#: expressive_gui.py:549 expressive_gui.py:586 expressive_gui.py:629 +#: expressive_gui.py:698 expressive_gui.py:732 msgid "Spline Smoothing" msgstr "" -#: expressive_gui.py:557 expressive_gui.py:617 expressive_gui.py:655 +#: expressive_gui.py:554 expressive_gui.py:591 expressive_gui.py:665 +#: expressive_gui.py:703 expressive_gui.py:737 msgid "Align Radius" msgstr "" -#: expressive_gui.py:562 expressive_gui.py:628 expressive_gui.py:660 +#: expressive_gui.py:559 expressive_gui.py:596 expressive_gui.py:676 +#: expressive_gui.py:708 expressive_gui.py:742 msgid "Smoothness" msgstr "" -#: expressive_gui.py:567 expressive_gui.py:633 expressive_gui.py:665 +#: expressive_gui.py:564 expressive_gui.py:601 expressive_gui.py:681 +#: expressive_gui.py:713 msgid "Scaler" msgstr "" -#: expressive_gui.py:601 +#: expressive_gui.py:569 expressive_gui.py:718 expressive_gui.py:752 +msgid "Bias" +msgstr "" + +#: expressive_gui.py:583 expressive_gui.py:695 +msgid "Trim Silence" +msgstr "" + +#: expressive_gui.py:617 +msgid "mHuBERT Embedder" +msgstr "" + +#: expressive_gui.py:649 msgid "UTAU Confidence" msgstr "" -#: expressive_gui.py:607 +#: expressive_gui.py:655 msgid "Reference Confidence" msgstr "" -#: expressive_gui.py:613 +#: expressive_gui.py:661 msgid "Backend" msgstr "" -#: expressive_gui.py:622 +#: expressive_gui.py:670 msgid "Semitone Shift" msgstr "" -#: expressive_gui.py:623 +#: expressive_gui.py:671 msgid "Auto Estimation" msgstr "" -#: expressive_gui.py:670 -msgid "Bias" +#: expressive_gui.py:747 +msgid "Dynamic Range" msgstr "" -#: expressive_gui.py:678 +#: expressive_gui.py:760 msgid "Import Config" msgstr "" -#: expressive_gui.py:684 +#: expressive_gui.py:766 msgid "Export Config" msgstr "" -#: expressive_gui.py:695 +#: expressive_gui.py:777 msgid "Process" msgstr "" -#: expressive_gui.py:737 +#: expressive_gui.py:802 msgid "Migrate expressions from real singers to DiffSingers (GUI)" msgstr "" @@ -283,95 +295,135 @@ msgstr "" msgid "Expression result is empty. Skipping USTX update." msgstr "" -#: expressions/dyn.py:26 -msgid "Dynamics (curve)" -msgstr "" - -#: expressions/dyn.py:28 expressions/tenc.py:28 -msgid "" -"**Trim silence** from the leading and trailing edges of the audio before " -"extracting expression\n" -"\n" -"**NOTICE**: This may slightly cut into the beginning and ending of voiced" -" segments. If the effect is too severe, consider disabling this option\n" -"\n" +#: expressions/brec.py:31 +msgid "Breathiness (curve)" msgstr "" -#: expressions/dyn.py:29 expressions/pitd.py:41 expressions/tenc.py:29 +#: expressions/brec.py:33 expressions/dyn.py:29 expressions/pitd.py:60 +#: expressions/tenc.py:29 expressions/voic.py:33 msgid "" "**Radius** for the FastDTW alignment algorithm; larger values allow more " "flexible alignment but increase computation time" msgstr "" -#: expressions/dyn.py:30 expressions/pitd.py:43 expressions/tenc.py:30 +#: expressions/brec.py:34 expressions/dyn.py:30 expressions/pitd.py:62 +#: expressions/tenc.py:30 expressions/voic.py:34 msgid "" "Controls the **smoothness** of the expression curve using Gaussian " "filtering. Higher values produce smoother curves but may lose fine detail" msgstr "" -#: expressions/dyn.py:31 expressions/pitd.py:44 expressions/tenc.py:31 +#: expressions/brec.py:35 expressions/dyn.py:31 expressions/pitd.py:63 +#: expressions/tenc.py:31 msgid "" "**Scaling factor** applied to the expression curve. Values >1 amplify the" " expression, =1 keeps original intensity, <1 reduces it" msgstr "" -#: expressions/dyn.py:32 expressions/pitd.py:45 expressions/tenc.py:33 +#: expressions/brec.py:36 expressions/tenc.py:32 expressions/voic.py:36 +msgid "" +"**Bias** offset added to the expression curve. Positive values shift the " +"curve upward; negative values shift it downward" +msgstr "" + +#: expressions/brec.py:37 expressions/dyn.py:32 expressions/pitd.py:64 +#: expressions/tenc.py:33 expressions/voic.py:37 msgid "" "Perform **spline smoothing** on the final expression curve for extra " "smoothness" msgstr "" -#: expressions/dyn.py:35 expressions/dyn.py:37 expressions/pitd.py:48 -#: expressions/pitd.py:51 expressions/tenc.py:36 expressions/tenc.py:38 +#: expressions/brec.py:40 expressions/brec.py:42 expressions/dyn.py:35 +#: expressions/dyn.py:37 expressions/pitd.py:68 expressions/pitd.py:71 +#: expressions/tenc.py:36 expressions/tenc.py:38 expressions/voic.py:40 +#: expressions/voic.py:42 msgid "Tick" msgstr "" -#: expressions/dyn.py:36 expressions/tenc.py:37 -msgid "raw_rms" +#: expressions/brec.py:41 +msgid "raw_breath_index" msgstr "" -#: expressions/dyn.py:36 expressions/tenc.py:37 -msgid "Raw RMS" +#: expressions/brec.py:41 +msgid "Raw Breath Index" msgstr "" -#: expressions/dyn.py:36 expressions/pitd.py:49 expressions/pitd.py:50 -#: expressions/tenc.py:37 +#: expressions/brec.py:41 expressions/dyn.py:36 expressions/pitd.py:69 +#: expressions/pitd.py:70 expressions/tenc.py:37 expressions/voic.py:41 msgid "Time (s)" msgstr "" -#: expressions/dyn.py:36 expressions/dyn.py:37 expressions/tenc.py:37 -#: expressions/tenc.py:38 -msgid "RMS" +#: expressions/brec.py:41 expressions/brec.py:42 +msgid "Breath Index" msgstr "" -#: expressions/dyn.py:36 expressions/dyn.py:37 expressions/pitd.py:49 -#: expressions/pitd.py:50 expressions/pitd.py:51 expressions/tenc.py:37 -#: expressions/tenc.py:38 +#: expressions/brec.py:41 expressions/brec.py:42 expressions/dyn.py:36 +#: expressions/dyn.py:37 expressions/pitd.py:69 expressions/pitd.py:70 +#: expressions/pitd.py:71 expressions/tenc.py:37 expressions/tenc.py:38 +#: expressions/voic.py:41 expressions/voic.py:42 msgid "Reference" msgstr "" -#: expressions/dyn.py:36 expressions/dyn.py:37 expressions/pitd.py:49 -#: expressions/pitd.py:50 expressions/pitd.py:51 expressions/tenc.py:37 -#: expressions/tenc.py:38 +#: expressions/brec.py:41 expressions/brec.py:42 expressions/dyn.py:36 +#: expressions/dyn.py:37 expressions/pitd.py:69 expressions/pitd.py:70 +#: expressions/pitd.py:71 expressions/tenc.py:37 expressions/tenc.py:38 +#: expressions/voic.py:41 expressions/voic.py:42 msgid "UTAU" msgstr "" -#: expressions/dyn.py:37 expressions/tenc.py:38 -msgid "aligned_rms" +#: expressions/brec.py:42 +msgid "aligned_breath_index" msgstr "" -#: expressions/dyn.py:37 expressions/tenc.py:38 -msgid "Aligned RMS" +#: expressions/brec.py:42 +msgid "Aligned Breath Index" msgstr "" -#: expressions/dyn.py:48 expressions/pitd.py:65 expressions/tenc.py:50 +#: expressions/brec.py:53 expressions/dyn.py:48 expressions/pitd.py:86 +#: expressions/tenc.py:50 expressions/voic.py:53 msgid "Extracting expression..." msgstr "" -#: expressions/dyn.py:86 expressions/pitd.py:124 expressions/tenc.py:88 +#: expressions/brec.py:83 expressions/dyn.py:86 expressions/pitd.py:150 +#: expressions/tenc.py:88 expressions/voic.py:83 msgid "Expression extraction complete." msgstr "" +#: expressions/dyn.py:26 +msgid "Dynamics (curve)" +msgstr "" + +#: expressions/dyn.py:28 expressions/tenc.py:28 +msgid "" +"**Trim silence** from the leading and trailing edges of the audio before " +"extracting expression\n" +"\n" +"**NOTICE**: This may slightly cut into the beginning and ending of voiced" +" segments. If the effect is too severe, consider disabling this option\n" +"\n" +msgstr "" + +#: expressions/dyn.py:36 expressions/tenc.py:37 +msgid "raw_rms" +msgstr "" + +#: expressions/dyn.py:36 expressions/tenc.py:37 +msgid "Raw RMS" +msgstr "" + +#: expressions/dyn.py:36 expressions/dyn.py:37 expressions/tenc.py:37 +#: expressions/tenc.py:38 +msgid "RMS" +msgstr "" + +#: expressions/dyn.py:37 expressions/tenc.py:38 +msgid "aligned_rms" +msgstr "" + +#: expressions/dyn.py:37 expressions/tenc.py:38 +msgid "Aligned RMS" +msgstr "" + #: expressions/pitd.py:28 msgid "Pitch Deviation (curve)" msgstr "" @@ -392,7 +444,40 @@ msgstr "" msgid "based on rmvpe-onnx, improved by swift-f0, CPU only (ONNX Runtime)" msgstr "" +#: expressions/pitd.py:36 +msgid "(378 MB) maximum accuracy, full size" +msgstr "" + +#: expressions/pitd.py:37 +msgid "(189 MB) high accuracy, medium size" +msgstr "" + #: expressions/pitd.py:38 +msgid "(89.7 MB) fair accuracy, small size" +msgstr "" + +#: expressions/pitd.py:39 +msgid "disabled, skip this feature" +msgstr "" + +#: expressions/pitd.py:41 +msgid "" +"\n" +"mHuBERT model weights are licensed under CC-BY-NC-SA-4.0.\n" +"Do NOT use the mHuBERT model weights for commercial purposes.\n" +"This software does not grant you rights to the model weights.\n" +"You are solely responsible for compliance with the model license:\n" +"https://creativecommons.org/licenses/by-nc-sa/4.0/\n" +"\n" +"Original model author: utter-project\n" +"\n" +"Original model page: https://huggingface.co/utter-project/mHuBERT-147\n" +"\n" +"ONNX conversion: https://huggingface.co/NewComer00/mHuBERT-147-ONNX" +"\n" +msgstr "" + +#: expressions/pitd.py:57 #, python-format msgid "" "**F0 detection backend** for extracting pitch from WAV files. Available " @@ -402,7 +487,7 @@ msgid "" "\n" msgstr "" -#: expressions/pitd.py:39 +#: expressions/pitd.py:58 #, python-format msgid "" "Minimum **confidence level** for keeping detected pitch values in the " @@ -413,7 +498,7 @@ msgid "" "\n" msgstr "" -#: expressions/pitd.py:40 +#: expressions/pitd.py:59 #, python-format msgid "" "Minimum **confidence level** for keeping detected pitch values in the " @@ -424,46 +509,60 @@ msgid "" "\n" msgstr "" -#: expressions/pitd.py:42 +#: expressions/pitd.py:61 msgid "" "**Semitone shift** between the UTAU and reference WAV. If the UTAU WAV is" " an octave higher than the reference WAV, set to 12; if lower, set to " "-12. Omit to enable automatic shift estimation" msgstr "" -#: expressions/pitd.py:49 +#: expressions/pitd.py:65 +#, python-format +msgid "" +"**mHuBERT embedder variant** for feature extraction. If not `disabled`, " +"mHuBERT embedder will be downloaded automatically and used for advanced " +"feature extraction. Use this when the difference between UTAU and " +"reference audio so significant that conventional features are not enough." +" mHuBERT embedder runs well both on CPU and NVIDIA GPU. Available " +"options:\n" +"\n" +"%s\n" +"\n" +msgstr "" + +#: expressions/pitd.py:69 msgid "confidence" msgstr "" -#: expressions/pitd.py:49 +#: expressions/pitd.py:69 msgid "Pitch Extraction Confidence" msgstr "" -#: expressions/pitd.py:49 +#: expressions/pitd.py:69 msgid "Confidence" msgstr "" -#: expressions/pitd.py:50 +#: expressions/pitd.py:70 msgid "raw_pitch" msgstr "" -#: expressions/pitd.py:50 +#: expressions/pitd.py:70 msgid "Raw Pitch" msgstr "" -#: expressions/pitd.py:50 expressions/pitd.py:51 +#: expressions/pitd.py:70 expressions/pitd.py:71 msgid "Pitch (Hz)" msgstr "" -#: expressions/pitd.py:51 +#: expressions/pitd.py:71 msgid "aligned_pitch" msgstr "" -#: expressions/pitd.py:51 +#: expressions/pitd.py:71 msgid "Aligned Pitch" msgstr "" -#: expressions/pitd.py:199 +#: expressions/pitd.py:237 #, python-brace-format msgid "Estimated Semitone-shift: {}" msgstr "" @@ -472,44 +571,102 @@ msgstr "" msgid "Tension (curve)" msgstr "" -#: expressions/tenc.py:32 +#: expressions/voic.py:31 +msgid "Voicing (curve)" +msgstr "" + +#: expressions/voic.py:35 msgid "" -"**Bias** offset added to the expression curve. Positive values shift the " -"curve upward; negative values shift it downward" +"**Dynamic range** of the expression curve. Values >1 amplify the " +"variation, =1 keeps original range, <1 compresses it" +msgstr "" + +#: expressions/voic.py:41 +msgid "raw_voice_index" +msgstr "" + +#: expressions/voic.py:41 +msgid "Raw Voice Index" +msgstr "" + +#: expressions/voic.py:41 expressions/voic.py:42 +msgid "Voice Index" +msgstr "" + +#: expressions/voic.py:42 +msgid "aligned_voice_index" msgstr "" -#: utils/ui.py:350 +#: expressions/voic.py:42 +msgid "Aligned Voice Index" +msgstr "" + +#: utils/embedder.py:244 +#, python-brace-format +msgid "Loaded {name} model from '{path}'." +msgstr "" + +#: utils/embedder.py:246 +msgid "ONNX Runtime is using CUDA for inference acceleration." +msgstr "" + +#: utils/embedder.py:284 +#, python-brace-format +msgid "Fetching {name} ({variant}) from Internet or local cache..." +msgstr "" + +#: utils/ui.py:349 msgid "Failed to load audio" msgstr "" -#: utils/ui.py:377 +#: utils/ui.py:376 msgid "Play/Pause" msgstr "" -#: utils/ui.py:378 +#: utils/ui.py:377 msgid "Loop region" msgstr "" -#: utils/ui.py:379 +#: utils/ui.py:378 msgid "Zoom" msgstr "" -#: utils/wavtool.py:88 +#: utils/wavtool.py:67 +#, python-brace-format +msgid "Loading embedding data from cache file: '{}'" +msgstr "" + +#: utils/wavtool.py:83 +#, python-brace-format +msgid "Embedding data saved to cache file: '{}'" +msgstr "" + +#: utils/wavtool.py:117 +#, python-brace-format +msgid "Loading breath/voice data from cache file: '{}'" +msgstr "" + +#: utils/wavtool.py:140 +#, python-brace-format +msgid "Breath/voice data saved to cache file: '{}'" +msgstr "" + +#: utils/wavtool.py:217 #, python-brace-format msgid "Loading F0 data from cache file: '{}'" msgstr "" -#: utils/wavtool.py:133 +#: utils/wavtool.py:259 #, python-brace-format msgid "F0 data saved to cache file: '{}'" msgstr "" -#: utils/wavtool.py:380 +#: utils/wavtool.py:506 #, python-brace-format msgid "start {:.3f}s clamped to {:.3f}s (total duration: {:.3f}s)" msgstr "" -#: utils/wavtool.py:386 +#: utils/wavtool.py:512 #, python-brace-format msgid "end {:.3f}s clamped to {:.3f}s (total duration: {:.3f}s)" msgstr "" diff --git a/locales/en/LC_MESSAGES/app.po b/locales/en/LC_MESSAGES/app.po index c268610..0224598 100644 --- a/locales/en/LC_MESSAGES/app.po +++ b/locales/en/LC_MESSAGES/app.po @@ -7,7 +7,7 @@ msgid "" msgstr "" "Project-Id-Version: expressive\n" "Report-Msgid-Bugs-To: https://github.com/NewComer00/expressive/issues\n" -"POT-Creation-Date: 2026-04-15 21:12+0800\n" +"POT-Creation-Date: 2026-04-30 20:29+0800\n" "PO-Revision-Date: 2026-03-01 20:21+0800\n" "Last-Translator: NewComer00\n" "Language: en\n" @@ -18,25 +18,25 @@ msgstr "" "Content-Transfer-Encoding: 8bit\n" "Generated-By: Babel 2.18.0\n" -#: expressive.py:86 +#: expressive.py:93 #, python-brace-format msgid "Expression '{exp_type}' is not supported." msgstr "Expression '{exp_type}' is not supported." -#: expressive.py:132 +#: expressive.py:146 #, python-brace-format msgid "Logs saved to '{log_path}'" msgstr "Logs saved to '{log_path}'" -#: expressive.py:143 +#: expressive.py:157 msgid "Migrate expressions from real singers to DiffSingers (CLI)" msgstr "Migrate expressions from real singers to DiffSingers (CLI)" -#: expressive.py:152 expressive_gui.py:511 +#: expressive.py:166 expressive_gui.py:511 msgid "Path to save the processed `.ustx` file" msgstr "Path to save the processed `.ustx` file" -#: expressive.py:160 +#: expressive.py:174 msgid "" "**Expression(s)** to apply. Repeat the flag for multiple expressions " "(e.g., `-e dyn -e pitd`)" @@ -44,16 +44,16 @@ msgstr "" "**Expression(s)** to apply. Repeat the flag for multiple expressions " "(e.g., `-e dyn -e pitd`)" -#: expressive.py:184 +#: expressive.py:198 msgid "Starting Expressive CLI..." msgstr "Starting Expressive CLI..." -#: expressive.py:194 +#: expressive.py:208 #, python-brace-format msgid "Error occurred during processing: {e}" msgstr "Error occurred during processing: {e}" -#: expressive.py:197 expressive_gui.py:235 expressive_gui.py:236 +#: expressive.py:210 expressive_gui.py:235 expressive_gui.py:236 msgid "Processing completed successfully!" msgstr "Processing completed successfully!" @@ -151,63 +151,75 @@ msgstr "Track Number" msgid "Expression Selection" msgstr "Expression Selection" -#: expressive_gui.py:549 expressive_gui.py:647 -msgid "Trim Silence" -msgstr "Trim Silence" - -#: expressive_gui.py:552 expressive_gui.py:581 expressive_gui.py:650 +#: expressive_gui.py:549 expressive_gui.py:586 expressive_gui.py:629 +#: expressive_gui.py:698 expressive_gui.py:732 msgid "Spline Smoothing" msgstr "Spline Smoothing" -#: expressive_gui.py:557 expressive_gui.py:617 expressive_gui.py:655 +#: expressive_gui.py:554 expressive_gui.py:591 expressive_gui.py:665 +#: expressive_gui.py:703 expressive_gui.py:737 msgid "Align Radius" msgstr "Align Radius" -#: expressive_gui.py:562 expressive_gui.py:628 expressive_gui.py:660 +#: expressive_gui.py:559 expressive_gui.py:596 expressive_gui.py:676 +#: expressive_gui.py:708 expressive_gui.py:742 msgid "Smoothness" msgstr "Smoothness" -#: expressive_gui.py:567 expressive_gui.py:633 expressive_gui.py:665 +#: expressive_gui.py:564 expressive_gui.py:601 expressive_gui.py:681 +#: expressive_gui.py:713 msgid "Scaler" msgstr "Scaler" -#: expressive_gui.py:601 +#: expressive_gui.py:569 expressive_gui.py:718 expressive_gui.py:752 +msgid "Bias" +msgstr "Bias" + +#: expressive_gui.py:583 expressive_gui.py:695 +msgid "Trim Silence" +msgstr "Trim Silence" + +#: expressive_gui.py:617 +msgid "mHuBERT Embedder" +msgstr "mHuBERT Embedder" + +#: expressive_gui.py:649 msgid "UTAU Confidence" msgstr "UTAU Confidence" -#: expressive_gui.py:607 +#: expressive_gui.py:655 msgid "Reference Confidence" msgstr "Reference Confidence" -#: expressive_gui.py:613 +#: expressive_gui.py:661 msgid "Backend" msgstr "Backend" -#: expressive_gui.py:622 +#: expressive_gui.py:670 msgid "Semitone Shift" msgstr "Semitone Shift" -#: expressive_gui.py:623 +#: expressive_gui.py:671 msgid "Auto Estimation" msgstr "Auto Estimation" -#: expressive_gui.py:670 -msgid "Bias" -msgstr "Bias" +#: expressive_gui.py:747 +msgid "Dynamic Range" +msgstr "Dynamic Range" -#: expressive_gui.py:678 +#: expressive_gui.py:760 msgid "Import Config" msgstr "Import Config" -#: expressive_gui.py:684 +#: expressive_gui.py:766 msgid "Export Config" msgstr "Export Config" -#: expressive_gui.py:695 +#: expressive_gui.py:777 msgid "Process" msgstr "Process" -#: expressive_gui.py:737 +#: expressive_gui.py:802 msgid "Migrate expressions from real singers to DiffSingers (GUI)" msgstr "Migrate expressions from real singers to DiffSingers (GUI)" @@ -294,27 +306,12 @@ msgstr "Expression written to USTX file: '{}'" msgid "Expression result is empty. Skipping USTX update." msgstr "Expression result is empty. Skipping USTX update." -#: expressions/dyn.py:26 -msgid "Dynamics (curve)" -msgstr "Dynamics (curve)" +#: expressions/brec.py:31 +msgid "Breathiness (curve)" +msgstr "Breathiness (curve)" -#: expressions/dyn.py:28 expressions/tenc.py:28 -msgid "" -"**Trim silence** from the leading and trailing edges of the audio before " -"extracting expression\n" -"\n" -"**NOTICE**: This may slightly cut into the beginning and ending of voiced" -" segments. If the effect is too severe, consider disabling this option\n" -"\n" -msgstr "" -"**Trim silence** from the leading and trailing edges of the audio before " -"extracting expression\n" -"\n" -"**NOTICE**: This may slightly cut into the beginning and ending of voiced" -" segments. If the effect is too severe, consider disabling this option\n" -"\n" - -#: expressions/dyn.py:29 expressions/pitd.py:41 expressions/tenc.py:29 +#: expressions/brec.py:33 expressions/dyn.py:29 expressions/pitd.py:60 +#: expressions/tenc.py:29 expressions/voic.py:33 msgid "" "**Radius** for the FastDTW alignment algorithm; larger values allow more " "flexible alignment but increase computation time" @@ -322,7 +319,8 @@ msgstr "" "**Radius** for the FastDTW alignment algorithm; larger values allow more " "flexible alignment but increase computation time" -#: expressions/dyn.py:30 expressions/pitd.py:43 expressions/tenc.py:30 +#: expressions/brec.py:34 expressions/dyn.py:30 expressions/pitd.py:62 +#: expressions/tenc.py:30 expressions/voic.py:34 msgid "" "Controls the **smoothness** of the expression curve using Gaussian " "filtering. Higher values produce smoother curves but may lose fine detail" @@ -330,7 +328,8 @@ msgstr "" "Controls the **smoothness** of the expression curve using Gaussian " "filtering. Higher values produce smoother curves but may lose fine detail" -#: expressions/dyn.py:31 expressions/pitd.py:44 expressions/tenc.py:31 +#: expressions/brec.py:35 expressions/dyn.py:31 expressions/pitd.py:63 +#: expressions/tenc.py:31 msgid "" "**Scaling factor** applied to the expression curve. Values >1 amplify the" " expression, =1 keeps original intensity, <1 reduces it" @@ -338,7 +337,16 @@ msgstr "" "**Scaling factor** applied to the expression curve. Values >1 amplify the" " expression, =1 keeps original intensity, <1 reduces it" -#: expressions/dyn.py:32 expressions/pitd.py:45 expressions/tenc.py:33 +#: expressions/brec.py:36 expressions/tenc.py:32 expressions/voic.py:36 +msgid "" +"**Bias** offset added to the expression curve. Positive values shift the " +"curve upward; negative values shift it downward" +msgstr "" +"**Bias** offset added to the expression curve. Positive values shift the " +"curve upward; negative values shift it downward" + +#: expressions/brec.py:37 expressions/dyn.py:32 expressions/pitd.py:64 +#: expressions/tenc.py:33 expressions/voic.py:37 msgid "" "Perform **spline smoothing** on the final expression curve for extra " "smoothness" @@ -346,11 +354,82 @@ msgstr "" "Perform **spline smoothing** on the final expression curve for extra " "smoothness" -#: expressions/dyn.py:35 expressions/dyn.py:37 expressions/pitd.py:48 -#: expressions/pitd.py:51 expressions/tenc.py:36 expressions/tenc.py:38 +#: expressions/brec.py:40 expressions/brec.py:42 expressions/dyn.py:35 +#: expressions/dyn.py:37 expressions/pitd.py:68 expressions/pitd.py:71 +#: expressions/tenc.py:36 expressions/tenc.py:38 expressions/voic.py:40 +#: expressions/voic.py:42 msgid "Tick" msgstr "Tick" +#: expressions/brec.py:41 +msgid "raw_breath_index" +msgstr "raw_breath_index" + +#: expressions/brec.py:41 +msgid "Raw Breath Index" +msgstr "Raw Breath Index" + +#: expressions/brec.py:41 expressions/dyn.py:36 expressions/pitd.py:69 +#: expressions/pitd.py:70 expressions/tenc.py:37 expressions/voic.py:41 +msgid "Time (s)" +msgstr "Time (s)" + +#: expressions/brec.py:41 expressions/brec.py:42 +msgid "Breath Index" +msgstr "Breath Index" + +#: expressions/brec.py:41 expressions/brec.py:42 expressions/dyn.py:36 +#: expressions/dyn.py:37 expressions/pitd.py:69 expressions/pitd.py:70 +#: expressions/pitd.py:71 expressions/tenc.py:37 expressions/tenc.py:38 +#: expressions/voic.py:41 expressions/voic.py:42 +msgid "Reference" +msgstr "Reference" + +#: expressions/brec.py:41 expressions/brec.py:42 expressions/dyn.py:36 +#: expressions/dyn.py:37 expressions/pitd.py:69 expressions/pitd.py:70 +#: expressions/pitd.py:71 expressions/tenc.py:37 expressions/tenc.py:38 +#: expressions/voic.py:41 expressions/voic.py:42 +msgid "UTAU" +msgstr "UTAU" + +#: expressions/brec.py:42 +msgid "aligned_breath_index" +msgstr "aligned_breath_index" + +#: expressions/brec.py:42 +msgid "Aligned Breath Index" +msgstr "Aligned Breath Index" + +#: expressions/brec.py:53 expressions/dyn.py:48 expressions/pitd.py:86 +#: expressions/tenc.py:50 expressions/voic.py:53 +msgid "Extracting expression..." +msgstr "Extracting expression..." + +#: expressions/brec.py:83 expressions/dyn.py:86 expressions/pitd.py:150 +#: expressions/tenc.py:88 expressions/voic.py:83 +msgid "Expression extraction complete." +msgstr "Expression extraction complete." + +#: expressions/dyn.py:26 +msgid "Dynamics (curve)" +msgstr "Dynamics (curve)" + +#: expressions/dyn.py:28 expressions/tenc.py:28 +msgid "" +"**Trim silence** from the leading and trailing edges of the audio before " +"extracting expression\n" +"\n" +"**NOTICE**: This may slightly cut into the beginning and ending of voiced" +" segments. If the effect is too severe, consider disabling this option\n" +"\n" +msgstr "" +"**Trim silence** from the leading and trailing edges of the audio before " +"extracting expression\n" +"\n" +"**NOTICE**: This may slightly cut into the beginning and ending of voiced" +" segments. If the effect is too severe, consider disabling this option\n" +"\n" + #: expressions/dyn.py:36 expressions/tenc.py:37 msgid "raw_rms" msgstr "raw_rms" @@ -359,28 +438,11 @@ msgstr "raw_rms" msgid "Raw RMS" msgstr "Raw RMS" -#: expressions/dyn.py:36 expressions/pitd.py:49 expressions/pitd.py:50 -#: expressions/tenc.py:37 -msgid "Time (s)" -msgstr "Time (s)" - #: expressions/dyn.py:36 expressions/dyn.py:37 expressions/tenc.py:37 #: expressions/tenc.py:38 msgid "RMS" msgstr "RMS" -#: expressions/dyn.py:36 expressions/dyn.py:37 expressions/pitd.py:49 -#: expressions/pitd.py:50 expressions/pitd.py:51 expressions/tenc.py:37 -#: expressions/tenc.py:38 -msgid "Reference" -msgstr "Reference" - -#: expressions/dyn.py:36 expressions/dyn.py:37 expressions/pitd.py:49 -#: expressions/pitd.py:50 expressions/pitd.py:51 expressions/tenc.py:37 -#: expressions/tenc.py:38 -msgid "UTAU" -msgstr "UTAU" - #: expressions/dyn.py:37 expressions/tenc.py:38 msgid "aligned_rms" msgstr "aligned_rms" @@ -389,14 +451,6 @@ msgstr "aligned_rms" msgid "Aligned RMS" msgstr "Aligned RMS" -#: expressions/dyn.py:48 expressions/pitd.py:65 expressions/tenc.py:50 -msgid "Extracting expression..." -msgstr "Extracting expression..." - -#: expressions/dyn.py:86 expressions/pitd.py:124 expressions/tenc.py:88 -msgid "Expression extraction complete." -msgstr "Expression extraction complete." - #: expressions/pitd.py:28 msgid "Pitch Deviation (curve)" msgstr "Pitch Deviation (curve)" @@ -417,7 +471,53 @@ msgstr "good accuracy, slow, CPU & NVIDIA GPU (TensorFlow)" msgid "based on rmvpe-onnx, improved by swift-f0, CPU only (ONNX Runtime)" msgstr "based on rmvpe-onnx, improved by swift-f0, CPU only (ONNX Runtime)" +#: expressions/pitd.py:36 +msgid "(378 MB) maximum accuracy, full size" +msgstr "(378 MB) maximum accuracy, full size" + +#: expressions/pitd.py:37 +msgid "(189 MB) high accuracy, medium size" +msgstr "(189 MB) high accuracy, medium size" + #: expressions/pitd.py:38 +msgid "(89.7 MB) fair accuracy, small size" +msgstr "(89.7 MB) fair accuracy, small size" + +#: expressions/pitd.py:39 +msgid "disabled, skip this feature" +msgstr "disabled, skip this feature" + +#: expressions/pitd.py:41 +msgid "" +"\n" +"mHuBERT model weights are licensed under CC-BY-NC-SA-4.0.\n" +"Do NOT use the mHuBERT model weights for commercial purposes.\n" +"This software does not grant you rights to the model weights.\n" +"You are solely responsible for compliance with the model license:\n" +"https://creativecommons.org/licenses/by-nc-sa/4.0/\n" +"\n" +"Original model author: utter-project\n" +"\n" +"Original model page: https://huggingface.co/utter-project/mHuBERT-147\n" +"\n" +"ONNX conversion: https://huggingface.co/NewComer00/mHuBERT-147-ONNX" +"\n" +msgstr "" +"\n" +"mHuBERT model weights are licensed under CC-BY-NC-SA-4.0.\n" +"Do NOT use the mHuBERT model weights for commercial purposes.\n" +"This software does not grant you rights to the model weights.\n" +"You are solely responsible for compliance with the model license:\n" +"https://creativecommons.org/licenses/by-nc-sa/4.0/\n" +"\n" +"Original model author: utter-project\n" +"\n" +"Original model page: https://huggingface.co/utter-project/mHuBERT-147\n" +"\n" +"ONNX conversion: https://huggingface.co/NewComer00/mHuBERT-147-ONNX" +"\n" + +#: expressions/pitd.py:57 #, python-format msgid "" "**F0 detection backend** for extracting pitch from WAV files. Available " @@ -432,7 +532,7 @@ msgstr "" "%s\n" "\n" -#: expressions/pitd.py:39 +#: expressions/pitd.py:58 #, python-format msgid "" "Minimum **confidence level** for keeping detected pitch values in the " @@ -449,7 +549,7 @@ msgstr "" "%s\n" "\n" -#: expressions/pitd.py:40 +#: expressions/pitd.py:59 #, python-format msgid "" "Minimum **confidence level** for keeping detected pitch values in the " @@ -466,7 +566,7 @@ msgstr "" "%s\n" "\n" -#: expressions/pitd.py:42 +#: expressions/pitd.py:61 msgid "" "**Semitone shift** between the UTAU and reference WAV. If the UTAU WAV is" " an octave higher than the reference WAV, set to 12; if lower, set to " @@ -476,39 +576,62 @@ msgstr "" " an octave higher than the reference WAV, set to 12; if lower, set to " "-12. Omit to enable automatic shift estimation" -#: expressions/pitd.py:49 +#: expressions/pitd.py:65 +#, python-format +msgid "" +"**mHuBERT embedder variant** for feature extraction. If not `disabled`, " +"mHuBERT embedder will be downloaded automatically and used for advanced " +"feature extraction. Use this when the difference between UTAU and " +"reference audio so significant that conventional features are not enough." +" mHuBERT embedder runs well both on CPU and NVIDIA GPU. Available " +"options:\n" +"\n" +"%s\n" +"\n" +msgstr "" +"**mHuBERT embedder variant** for feature extraction. If not `disabled`, " +"mHuBERT embedder will be downloaded automatically and used for advanced " +"feature extraction. Use this when the difference between UTAU and " +"reference audio so significant that conventional features are not enough." +" mHuBERT embedder runs well both on CPU and NVIDIA GPU. Available " +"options:\n" +"\n" +"%s\n" +"\n" + +#: expressions/pitd.py:69 msgid "confidence" msgstr "confidence" -#: expressions/pitd.py:49 +#: expressions/pitd.py:69 msgid "Pitch Extraction Confidence" msgstr "Pitch Extraction Confidence" -#: expressions/pitd.py:49 +#: expressions/pitd.py:69 msgid "Confidence" msgstr "Confidence" -#: expressions/pitd.py:50 +#: expressions/pitd.py:70 msgid "raw_pitch" msgstr "raw_pitch" -#: expressions/pitd.py:50 +#: expressions/pitd.py:70 msgid "Raw Pitch" msgstr "Raw Pitch" -#: expressions/pitd.py:50 expressions/pitd.py:51 +#: expressions/pitd.py:70 expressions/pitd.py:71 msgid "Pitch (Hz)" msgstr "Pitch (Hz)" -#: expressions/pitd.py:51 +#: expressions/pitd.py:71 msgid "aligned_pitch" msgstr "aligned_pitch" -#: expressions/pitd.py:51 +#: expressions/pitd.py:71 msgid "Aligned Pitch" msgstr "Aligned Pitch" -#: expressions/pitd.py:199 +#: expressions/pitd.py:237 #, python-brace-format msgid "Estimated Semitone-shift: {}" msgstr "Estimated Semitone-shift: {}" @@ -517,46 +640,104 @@ msgstr "Estimated Semitone-shift: {}" msgid "Tension (curve)" msgstr "Tension (curve)" -#: expressions/tenc.py:32 +#: expressions/voic.py:31 +msgid "Voicing (curve)" +msgstr "Voicing (curve)" + +#: expressions/voic.py:35 msgid "" -"**Bias** offset added to the expression curve. Positive values shift the " -"curve upward; negative values shift it downward" +"**Dynamic range** of the expression curve. Values >1 amplify the " +"variation, =1 keeps original range, <1 compresses it" msgstr "" -"**Bias** offset added to the expression curve. Positive values shift the " -"curve upward; negative values shift it downward" +"**Dynamic range** of the expression curve. Values >1 amplify the " +"variation, =1 keeps original range, <1 compresses it" + +#: expressions/voic.py:41 +msgid "raw_voice_index" +msgstr "raw_voice_index" + +#: expressions/voic.py:41 +msgid "Raw Voice Index" +msgstr "Raw Voice Index" + +#: expressions/voic.py:41 expressions/voic.py:42 +msgid "Voice Index" +msgstr "Voice Index" -#: utils/ui.py:350 +#: expressions/voic.py:42 +msgid "aligned_voice_index" +msgstr "aligned_voice_index" + +#: expressions/voic.py:42 +msgid "Aligned Voice Index" +msgstr "Aligned Voice Index" + +#: utils/embedder.py:244 +#, python-brace-format +msgid "Loaded {name} model from '{path}'." +msgstr "Loaded {name} model from '{path}'." + +#: utils/embedder.py:246 +msgid "ONNX Runtime is using CUDA for inference acceleration." +msgstr "ONNX Runtime is using CUDA for inference acceleration." + +#: utils/embedder.py:284 +#, python-brace-format +msgid "Fetching {name} ({variant}) from Internet or local cache..." +msgstr "Fetching {name} ({variant}) from Internet or local cache..." + +#: utils/ui.py:349 msgid "Failed to load audio" msgstr "Failed to load audio" -#: utils/ui.py:377 +#: utils/ui.py:376 msgid "Play/Pause" msgstr "Play/Pause" -#: utils/ui.py:378 +#: utils/ui.py:377 msgid "Loop region" msgstr "Loop region" -#: utils/ui.py:379 +#: utils/ui.py:378 msgid "Zoom" msgstr "Zoom" -#: utils/wavtool.py:88 +#: utils/wavtool.py:67 +#, python-brace-format +msgid "Loading embedding data from cache file: '{}'" +msgstr "Loading embedding data from cache file: '{}'" + +#: utils/wavtool.py:83 +#, python-brace-format +msgid "Embedding data saved to cache file: '{}'" +msgstr "Embedding data saved to cache file: '{}'" + +#: utils/wavtool.py:117 +#, python-brace-format +msgid "Loading breath/voice data from cache file: '{}'" +msgstr "Loading breath/voice data from cache file: '{}'" + +#: utils/wavtool.py:140 +#, python-brace-format +msgid "Breath/voice data saved to cache file: '{}'" +msgstr "Breath/voice data saved to cache file: '{}'" + +#: utils/wavtool.py:217 #, python-brace-format msgid "Loading F0 data from cache file: '{}'" msgstr "Loading F0 data from cache file: '{}'" -#: utils/wavtool.py:133 +#: utils/wavtool.py:259 #, python-brace-format msgid "F0 data saved to cache file: '{}'" msgstr "F0 data saved to cache file: '{}'" -#: utils/wavtool.py:380 +#: utils/wavtool.py:506 #, python-brace-format msgid "start {:.3f}s clamped to {:.3f}s (total duration: {:.3f}s)" msgstr "start {:.3f}s clamped to {:.3f}s (total duration: {:.3f}s)" -#: utils/wavtool.py:386 +#: utils/wavtool.py:512 #, python-brace-format msgid "end {:.3f}s clamped to {:.3f}s (total duration: {:.3f}s)" msgstr "end {:.3f}s clamped to {:.3f}s (total duration: {:.3f}s)" diff --git a/locales/zh_CN/LC_MESSAGES/app.po b/locales/zh_CN/LC_MESSAGES/app.po index be89681..6b59991 100644 --- a/locales/zh_CN/LC_MESSAGES/app.po +++ b/locales/zh_CN/LC_MESSAGES/app.po @@ -7,7 +7,7 @@ msgid "" msgstr "" "Project-Id-Version: expressive\n" "Report-Msgid-Bugs-To: https://github.com/NewComer00/expressive/issues\n" -"POT-Creation-Date: 2026-04-15 21:12+0800\n" +"POT-Creation-Date: 2026-04-30 20:29+0800\n" "PO-Revision-Date: 2026-03-01 20:21+0800\n" "Last-Translator: NewComer00\n" "Language: zh_CN\n" @@ -18,40 +18,40 @@ msgstr "" "Content-Transfer-Encoding: 8bit\n" "Generated-By: Babel 2.18.0\n" -#: expressive.py:86 +#: expressive.py:93 #, python-brace-format msgid "Expression '{exp_type}' is not supported." msgstr "不支持的表情类型:'{exp_type}'。" -#: expressive.py:132 +#: expressive.py:146 #, python-brace-format msgid "Logs saved to '{log_path}'" msgstr "日志已保存到 '{log_path}'" -#: expressive.py:143 +#: expressive.py:157 msgid "Migrate expressions from real singers to DiffSingers (CLI)" msgstr "将表情参数从真实歌手迁移到 DiffSinger 歌手,适用于命令行界面(CLI)" -#: expressive.py:152 expressive_gui.py:511 +#: expressive.py:166 expressive_gui.py:511 msgid "Path to save the processed `.ustx` file" msgstr "用于保存处理后 USTX 文件的路径" -#: expressive.py:160 +#: expressive.py:174 msgid "" "**Expression(s)** to apply. Repeat the flag for multiple expressions " "(e.g., `-e dyn -e pitd`)" msgstr "要应用的表情参数。重复该标志以应用多个表情(例如,`-e dyn -e pitd`)" -#: expressive.py:184 +#: expressive.py:198 msgid "Starting Expressive CLI..." msgstr "正在启动 Expressive CLI..." -#: expressive.py:194 +#: expressive.py:208 #, python-brace-format msgid "Error occurred during processing: {e}" msgstr "处理时发生错误:{e}" -#: expressive.py:197 expressive_gui.py:235 expressive_gui.py:236 +#: expressive.py:210 expressive_gui.py:235 expressive_gui.py:236 msgid "Processing completed successfully!" msgstr "处理完成!" @@ -149,63 +149,75 @@ msgstr "轨道编号" msgid "Expression Selection" msgstr "表情参数" -#: expressive_gui.py:549 expressive_gui.py:647 -msgid "Trim Silence" -msgstr "剪除静音" - -#: expressive_gui.py:552 expressive_gui.py:581 expressive_gui.py:650 +#: expressive_gui.py:549 expressive_gui.py:586 expressive_gui.py:629 +#: expressive_gui.py:698 expressive_gui.py:732 msgid "Spline Smoothing" msgstr "样条曲线平滑" -#: expressive_gui.py:557 expressive_gui.py:617 expressive_gui.py:655 +#: expressive_gui.py:554 expressive_gui.py:591 expressive_gui.py:665 +#: expressive_gui.py:703 expressive_gui.py:737 msgid "Align Radius" msgstr "对齐半径" -#: expressive_gui.py:562 expressive_gui.py:628 expressive_gui.py:660 +#: expressive_gui.py:559 expressive_gui.py:596 expressive_gui.py:676 +#: expressive_gui.py:708 expressive_gui.py:742 msgid "Smoothness" msgstr "平滑度" -#: expressive_gui.py:567 expressive_gui.py:633 expressive_gui.py:665 +#: expressive_gui.py:564 expressive_gui.py:601 expressive_gui.py:681 +#: expressive_gui.py:713 msgid "Scaler" msgstr "缩放因子" -#: expressive_gui.py:601 +#: expressive_gui.py:569 expressive_gui.py:718 expressive_gui.py:752 +msgid "Bias" +msgstr "偏置" + +#: expressive_gui.py:583 expressive_gui.py:695 +msgid "Trim Silence" +msgstr "剪除静音" + +#: expressive_gui.py:617 +msgid "mHuBERT Embedder" +msgstr "mHuBERT 特征提取器" + +#: expressive_gui.py:649 msgid "UTAU Confidence" msgstr "歌姬音频置信度" -#: expressive_gui.py:607 +#: expressive_gui.py:655 msgid "Reference Confidence" msgstr "参考音频置信度" -#: expressive_gui.py:613 +#: expressive_gui.py:661 msgid "Backend" msgstr "后端" -#: expressive_gui.py:622 +#: expressive_gui.py:670 msgid "Semitone Shift" msgstr "半音偏移" -#: expressive_gui.py:623 +#: expressive_gui.py:671 msgid "Auto Estimation" msgstr "自动估算" -#: expressive_gui.py:670 -msgid "Bias" -msgstr "偏置" +#: expressive_gui.py:747 +msgid "Dynamic Range" +msgstr "动态范围" -#: expressive_gui.py:678 +#: expressive_gui.py:760 msgid "Import Config" msgstr "导入配置" -#: expressive_gui.py:684 +#: expressive_gui.py:766 msgid "Export Config" msgstr "导出配置" -#: expressive_gui.py:695 +#: expressive_gui.py:777 msgid "Process" msgstr "开始处理" -#: expressive_gui.py:737 +#: expressive_gui.py:802 msgid "Migrate expressions from real singers to DiffSingers (GUI)" msgstr "将表情参数从真实歌手迁移到 DiffSinger 歌手,适用于图形用户界面(GUI)" @@ -284,53 +296,118 @@ msgstr "表情参数已写入 USTX 文件:'{}'" msgid "Expression result is empty. Skipping USTX update." msgstr "表情参数结果为空,USTX 文件将不会更新。" -#: expressions/dyn.py:26 -msgid "Dynamics (curve)" -msgstr "动态曲线 Dynamics (curve)" +#: expressions/brec.py:31 +msgid "Breathiness (curve)" +msgstr "气息强度曲线 Breathiness (curve)" -#: expressions/dyn.py:28 expressions/tenc.py:28 -msgid "" -"**Trim silence** from the leading and trailing edges of the audio before " -"extracting expression\n" -"\n" -"**NOTICE**: This may slightly cut into the beginning and ending of voiced" -" segments. If the effect is too severe, consider disabling this option\n" -"\n" -msgstr "" -"在提取表情特征前,**剪除**音频开头与结尾的**静音部分**\n" -"\n" -"**注意**:该操作可能会轻微截断有声段的起始和结束部分;若影响较为明显,建议关闭此功能\n" -"\n" - -#: expressions/dyn.py:29 expressions/pitd.py:41 expressions/tenc.py:29 +#: expressions/brec.py:33 expressions/dyn.py:29 expressions/pitd.py:60 +#: expressions/tenc.py:29 expressions/voic.py:33 msgid "" "**Radius** for the FastDTW alignment algorithm; larger values allow more " "flexible alignment but increase computation time" msgstr "FastDTW 对齐算法的**半径**;值越大对齐越灵活,但计算时间也越长" -#: expressions/dyn.py:30 expressions/pitd.py:43 expressions/tenc.py:30 +#: expressions/brec.py:34 expressions/dyn.py:30 expressions/pitd.py:62 +#: expressions/tenc.py:30 expressions/voic.py:34 msgid "" "Controls the **smoothness** of the expression curve using Gaussian " "filtering. Higher values produce smoother curves but may lose fine detail" msgstr "通过高斯滤波控制表情曲线的**平滑度**;值越大曲线越平滑,但可能丢失细节" -#: expressions/dyn.py:31 expressions/pitd.py:44 expressions/tenc.py:31 +#: expressions/brec.py:35 expressions/dyn.py:31 expressions/pitd.py:63 +#: expressions/tenc.py:31 msgid "" "**Scaling factor** applied to the expression curve. Values >1 amplify the" " expression, =1 keeps original intensity, <1 reduces it" msgstr "应用于表情曲线的**缩放因子**;大于 1 则放大,等于 1 则保持原强度,小于 1 则缩小" -#: expressions/dyn.py:32 expressions/pitd.py:45 expressions/tenc.py:33 +#: expressions/brec.py:36 expressions/tenc.py:32 expressions/voic.py:36 +msgid "" +"**Bias** offset added to the expression curve. Positive values shift the " +"curve upward; negative values shift it downward" +msgstr "添加到表情曲线的**偏置**偏移量;正值使曲线上移,负值使曲线下移" + +#: expressions/brec.py:37 expressions/dyn.py:32 expressions/pitd.py:64 +#: expressions/tenc.py:33 expressions/voic.py:37 msgid "" "Perform **spline smoothing** on the final expression curve for extra " "smoothness" msgstr "对提取出的表情曲线进行**样条曲线平滑**,以获得更自然的效果" -#: expressions/dyn.py:35 expressions/dyn.py:37 expressions/pitd.py:48 -#: expressions/pitd.py:51 expressions/tenc.py:36 expressions/tenc.py:38 +#: expressions/brec.py:40 expressions/brec.py:42 expressions/dyn.py:35 +#: expressions/dyn.py:37 expressions/pitd.py:68 expressions/pitd.py:71 +#: expressions/tenc.py:36 expressions/tenc.py:38 expressions/voic.py:40 +#: expressions/voic.py:42 msgid "Tick" msgstr "时间刻度(Tick)" +#: expressions/brec.py:41 +msgid "raw_breath_index" +msgstr "原音频的 Breath Index" + +#: expressions/brec.py:41 +msgid "Raw Breath Index" +msgstr "原音频的气息指数(Breath Index)特征" + +#: expressions/brec.py:41 expressions/dyn.py:36 expressions/pitd.py:69 +#: expressions/pitd.py:70 expressions/tenc.py:37 expressions/voic.py:41 +msgid "Time (s)" +msgstr "时间(秒)" + +#: expressions/brec.py:41 expressions/brec.py:42 +msgid "Breath Index" +msgstr "气息指数" + +#: expressions/brec.py:41 expressions/brec.py:42 expressions/dyn.py:36 +#: expressions/dyn.py:37 expressions/pitd.py:69 expressions/pitd.py:70 +#: expressions/pitd.py:71 expressions/tenc.py:37 expressions/tenc.py:38 +#: expressions/voic.py:41 expressions/voic.py:42 +msgid "Reference" +msgstr "参考音频" + +#: expressions/brec.py:41 expressions/brec.py:42 expressions/dyn.py:36 +#: expressions/dyn.py:37 expressions/pitd.py:69 expressions/pitd.py:70 +#: expressions/pitd.py:71 expressions/tenc.py:37 expressions/tenc.py:38 +#: expressions/voic.py:41 expressions/voic.py:42 +msgid "UTAU" +msgstr "歌姬音频" + +#: expressions/brec.py:42 +msgid "aligned_breath_index" +msgstr "对齐后的 Breath Index" + +#: expressions/brec.py:42 +msgid "Aligned Breath Index" +msgstr "对齐后的气息指数(Breath Index)特征" + +#: expressions/brec.py:53 expressions/dyn.py:48 expressions/pitd.py:86 +#: expressions/tenc.py:50 expressions/voic.py:53 +msgid "Extracting expression..." +msgstr "正在提取表情参数..." + +#: expressions/brec.py:83 expressions/dyn.py:86 expressions/pitd.py:150 +#: expressions/tenc.py:88 expressions/voic.py:83 +msgid "Expression extraction complete." +msgstr "表情参数提取完成。" + +#: expressions/dyn.py:26 +msgid "Dynamics (curve)" +msgstr "动态曲线 Dynamics (curve)" + +#: expressions/dyn.py:28 expressions/tenc.py:28 +msgid "" +"**Trim silence** from the leading and trailing edges of the audio before " +"extracting expression\n" +"\n" +"**NOTICE**: This may slightly cut into the beginning and ending of voiced" +" segments. If the effect is too severe, consider disabling this option\n" +"\n" +msgstr "" +"在提取表情特征前,**剪除**音频开头与结尾的**静音部分**\n" +"\n" +"**注意**:该操作可能会轻微截断有声段的起始和结束部分;若影响较为明显,建议关闭此功能\n" +"\n" + #: expressions/dyn.py:36 expressions/tenc.py:37 msgid "raw_rms" msgstr "原音频的 RMS" @@ -339,28 +416,11 @@ msgstr "原音频的 RMS" msgid "Raw RMS" msgstr "原音频的均方根能量(RMS)特征" -#: expressions/dyn.py:36 expressions/pitd.py:49 expressions/pitd.py:50 -#: expressions/tenc.py:37 -msgid "Time (s)" -msgstr "时间(秒)" - #: expressions/dyn.py:36 expressions/dyn.py:37 expressions/tenc.py:37 #: expressions/tenc.py:38 msgid "RMS" msgstr "RMS" -#: expressions/dyn.py:36 expressions/dyn.py:37 expressions/pitd.py:49 -#: expressions/pitd.py:50 expressions/pitd.py:51 expressions/tenc.py:37 -#: expressions/tenc.py:38 -msgid "Reference" -msgstr "参考音频" - -#: expressions/dyn.py:36 expressions/dyn.py:37 expressions/pitd.py:49 -#: expressions/pitd.py:50 expressions/pitd.py:51 expressions/tenc.py:37 -#: expressions/tenc.py:38 -msgid "UTAU" -msgstr "歌姬音频" - #: expressions/dyn.py:37 expressions/tenc.py:38 msgid "aligned_rms" msgstr "对齐后的 RMS" @@ -369,14 +429,6 @@ msgstr "对齐后的 RMS" msgid "Aligned RMS" msgstr "对齐后的均方根能量(RMS)特征" -#: expressions/dyn.py:48 expressions/pitd.py:65 expressions/tenc.py:50 -msgid "Extracting expression..." -msgstr "正在提取表情参数..." - -#: expressions/dyn.py:86 expressions/pitd.py:124 expressions/tenc.py:88 -msgid "Expression extraction complete." -msgstr "表情参数提取完成。" - #: expressions/pitd.py:28 msgid "Pitch Deviation (curve)" msgstr "音高偏差曲线 Pitch Deviation (curve)" @@ -397,7 +449,52 @@ msgstr "精度较好,较慢,支持 CPU 和 NVIDIA GPU,基于 TensorFlow" msgid "based on rmvpe-onnx, improved by swift-f0, CPU only (ONNX Runtime)" msgstr "以 rmvpe-onnx 为基准,综合了 swift-f0 的结果,仅支持 CPU,基于 ONNX Runtime" +#: expressions/pitd.py:36 +msgid "(378 MB) maximum accuracy, full size" +msgstr "(378 MB) 精度最高,完整模型" + +#: expressions/pitd.py:37 +msgid "(189 MB) high accuracy, medium size" +msgstr "(189 MB) 精度较高,中等大小" + #: expressions/pitd.py:38 +msgid "(89.7 MB) fair accuracy, small size" +msgstr "(89.7 MB) 精度一般,轻量模型" + +#: expressions/pitd.py:39 +msgid "disabled, skip this feature" +msgstr "不启用此功能" + +#: expressions/pitd.py:41 +msgid "" +"\n" +"mHuBERT model weights are licensed under CC-BY-NC-SA-4.0.\n" +"Do NOT use the mHuBERT model weights for commercial purposes.\n" +"This software does not grant you rights to the model weights.\n" +"You are solely responsible for compliance with the model license:\n" +"https://creativecommons.org/licenses/by-nc-sa/4.0/\n" +"\n" +"Original model author: utter-project\n" +"\n" +"Original model page: https://huggingface.co/utter-project/mHuBERT-147\n" +"\n" +"ONNX conversion: https://huggingface.co/NewComer00/mHuBERT-147-ONNX" +"\n" +msgstr "" +"\n" +"mHuBERT 模型权重依据 CC-BY-NC-SA-4.0 协议授权。\n" +"请勿将 mHuBERT 模型权重用于任何商业用途。\n" +"本软件不授予您对模型权重的任何权利。\n" +"您须自行负责遵守模型许可协议:\n" +"https://creativecommons.org/licenses/by-nc-sa/4.0/\n" +"\n" +"原始模型作者:utter-project\n" +"\n" +"原始模型页面:https://huggingface.co/utter-project/mHuBERT-147\n" +"\n" +"ONNX 转换版本:https://huggingface.co/NewComer00/mHuBERT-147-ONNX\n" + +#: expressions/pitd.py:57 #, python-format msgid "" "**F0 detection backend** for extracting pitch from WAV files. Available " @@ -411,7 +508,7 @@ msgstr "" "%s\n" "\n" -#: expressions/pitd.py:39 +#: expressions/pitd.py:58 #, python-format msgid "" "Minimum **confidence level** for keeping detected pitch values in the " @@ -426,7 +523,7 @@ msgstr "" "%s\n" "\n" -#: expressions/pitd.py:40 +#: expressions/pitd.py:59 #, python-format msgid "" "Minimum **confidence level** for keeping detected pitch values in the " @@ -441,46 +538,66 @@ msgstr "" "%s\n" "\n" -#: expressions/pitd.py:42 +#: expressions/pitd.py:61 msgid "" "**Semitone shift** between the UTAU and reference WAV. If the UTAU WAV is" " an octave higher than the reference WAV, set to 12; if lower, set to " "-12. Omit to enable automatic shift estimation" msgstr "歌姬音频与参考音频之间的**半音偏移**;若歌姬比参考高一个八度则设为 12,低一个八度则设为 -12;忽略该参数则启用自动估算" -#: expressions/pitd.py:49 +#: expressions/pitd.py:65 +#, python-format +msgid "" +"**mHuBERT embedder variant** for feature extraction. If not `disabled`, " +"mHuBERT embedder will be downloaded automatically and used for advanced " +"feature extraction. Use this when the difference between UTAU and " +"reference audio so significant that conventional features are not enough." +" mHuBERT embedder runs well both on CPU and NVIDIA GPU. Available " +"options:\n" +"\n" +"%s\n" +"\n" +msgstr "" +"**mHuBERT 特征提取模型的种类**;默认值为 `disabled`,即不启用该功能。如果不为 `disabled`,则会自动下载 " +"mHuBERT 特征提取器并用于提取高级特征。当歌姬音频与参考音频差异较大,以至于常规特征不足以准确对齐时,建议启用此功能。mHuBERT " +"特征提取器 在 CPU 和 NVIDIA GPU 上均能良好运行。可选项如下:\n" +"\n" +"%s\n" +"\n" + +#: expressions/pitd.py:69 msgid "confidence" msgstr "置信度" -#: expressions/pitd.py:49 +#: expressions/pitd.py:69 msgid "Pitch Extraction Confidence" msgstr "音高提取置信度" -#: expressions/pitd.py:49 +#: expressions/pitd.py:69 msgid "Confidence" msgstr "置信度" -#: expressions/pitd.py:50 +#: expressions/pitd.py:70 msgid "raw_pitch" msgstr "原音频的 Pitch" -#: expressions/pitd.py:50 +#: expressions/pitd.py:70 msgid "Raw Pitch" msgstr "原音频的音高(Pitch)特征" -#: expressions/pitd.py:50 expressions/pitd.py:51 +#: expressions/pitd.py:70 expressions/pitd.py:71 msgid "Pitch (Hz)" msgstr "音高(Hz)" -#: expressions/pitd.py:51 +#: expressions/pitd.py:71 msgid "aligned_pitch" msgstr "对齐后的 Pitch" -#: expressions/pitd.py:51 +#: expressions/pitd.py:71 msgid "Aligned Pitch" msgstr "对齐后的音高(Pitch)特征" -#: expressions/pitd.py:199 +#: expressions/pitd.py:237 #, python-brace-format msgid "Estimated Semitone-shift: {}" msgstr "估计的半音偏移:{}" @@ -489,44 +606,102 @@ msgstr "估计的半音偏移:{}" msgid "Tension (curve)" msgstr "张力曲线 Tension (curve)" -#: expressions/tenc.py:32 +#: expressions/voic.py:31 +msgid "Voicing (curve)" +msgstr "人声强度曲线 Voicing (curve)" + +#: expressions/voic.py:35 msgid "" -"**Bias** offset added to the expression curve. Positive values shift the " -"curve upward; negative values shift it downward" -msgstr "添加到表情曲线的**偏置**偏移量;正值使曲线上移,负值使曲线下移" +"**Dynamic range** of the expression curve. Values >1 amplify the " +"variation, =1 keeps original range, <1 compresses it" +msgstr "表情曲线的**动态范围**;值大于 1 则放大变化,等于 1 则保持原动态范围,小于 1 则压缩动态范围" + +#: expressions/voic.py:41 +msgid "raw_voice_index" +msgstr "原音频的 Voice Index" + +#: expressions/voic.py:41 +msgid "Raw Voice Index" +msgstr "原音频的人声指数(Voice Index)特征" + +#: expressions/voic.py:41 expressions/voic.py:42 +msgid "Voice Index" +msgstr "人声指数(Voice Index)" + +#: expressions/voic.py:42 +msgid "aligned_voice_index" +msgstr "对齐后的 Voice Index" -#: utils/ui.py:350 +#: expressions/voic.py:42 +msgid "Aligned Voice Index" +msgstr "对齐后的人声指数(Voice Index)特征" + +#: utils/embedder.py:244 +#, python-brace-format +msgid "Loaded {name} model from '{path}'." +msgstr "{name} 模型已加载,模型路径:'{path}'。" + +#: utils/embedder.py:246 +msgid "ONNX Runtime is using CUDA for inference acceleration." +msgstr "ONNX Runtime 已启用 CUDA 进行推理加速。" + +#: utils/embedder.py:284 +#, python-brace-format +msgid "Fetching {name} ({variant}) from Internet or local cache..." +msgstr "正在从互联网或本地缓存获取 {name} 模型({variant})..." + +#: utils/ui.py:349 msgid "Failed to load audio" msgstr "音频加载失败" -#: utils/ui.py:377 +#: utils/ui.py:376 msgid "Play/Pause" msgstr "播放/暂停" -#: utils/ui.py:378 +#: utils/ui.py:377 msgid "Loop region" msgstr "选区循环播放" -#: utils/ui.py:379 +#: utils/ui.py:378 msgid "Zoom" msgstr "缩放" -#: utils/wavtool.py:88 +#: utils/wavtool.py:67 +#, python-brace-format +msgid "Loading embedding data from cache file: '{}'" +msgstr "正在从缓存文件加载特征数据:'{}'" + +#: utils/wavtool.py:83 +#, python-brace-format +msgid "Embedding data saved to cache file: '{}'" +msgstr "特征数据已保存到缓存文件:'{}'" + +#: utils/wavtool.py:117 +#, python-brace-format +msgid "Loading breath/voice data from cache file: '{}'" +msgstr "正在从缓存文件加载气息/人声数据:'{}'" + +#: utils/wavtool.py:140 +#, python-brace-format +msgid "Breath/voice data saved to cache file: '{}'" +msgstr "气息/人声数据已保存到缓存文件:'{}'" + +#: utils/wavtool.py:217 #, python-brace-format msgid "Loading F0 data from cache file: '{}'" msgstr "正在从缓存文件加载 F0 数据:'{}'" -#: utils/wavtool.py:133 +#: utils/wavtool.py:259 #, python-brace-format msgid "F0 data saved to cache file: '{}'" msgstr "F0 数据已保存到缓存文件:'{}'" -#: utils/wavtool.py:380 +#: utils/wavtool.py:506 #, python-brace-format msgid "start {:.3f}s clamped to {:.3f}s (total duration: {:.3f}s)" msgstr "开始时间 {:.3f}秒 被截取到 {:.3f}秒(完整时长:{:.3f}秒) " -#: utils/wavtool.py:386 +#: utils/wavtool.py:512 #, python-brace-format msgid "end {:.3f}s clamped to {:.3f}s (total duration: {:.3f}s)" msgstr "结束时间 {:.3f}秒 被截取到 {:.3f}秒(完整时长:{:.3f}秒) " diff --git a/pyproject.toml b/pyproject.toml index 78032bb..2fabcb4 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -11,9 +11,11 @@ requires-python = "==3.10.*" dependencies = [ "numpy<2", "tensorflow==2.10", + "onnxruntime==1.20.1", "crepe @ git+https://github.com/NewComer00/crepe.git@master", "swift-f0", "rmvpe-onnx", + "huggingface-hub", "oyaml", "fastdtw", "scipy", @@ -30,10 +32,13 @@ dependencies = [ "filelock", "plotly", "watchfiles", + "joblib", ] [project.optional-dependencies] gpu = [ + "onnxruntime-gpu @ https://aiinfra.pkgs.visualstudio.com/PublicPackages/_packaging/onnxruntime-cuda-11/pypi/download/onnxruntime-gpu/1.20.1/onnxruntime_gpu-1.20.1-cp310-cp310-win_amd64.whl ; sys_platform=='win32'", + "onnxruntime-gpu @ https://aiinfra.pkgs.visualstudio.com/PublicPackages/_packaging/onnxruntime-cuda-11/pypi/download/onnxruntime-gpu/1.20.1/onnxruntime_gpu-1.20.1-cp310-cp310-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl ; sys_platform=='linux'", "nvidia-cuda-runtime-cu11", "nvidia-cuda-nvcc-cu11", "nvidia-cublas-cu11", diff --git a/tests/test_embedder.py b/tests/test_embedder.py new file mode 100644 index 0000000..41ef48c --- /dev/null +++ b/tests/test_embedder.py @@ -0,0 +1,600 @@ +"""Tests for embedder.py — BaseEmbedder, EmbedderFactory, and feature extraction.""" + +from __future__ import annotations + +import numpy as np +import pytest + +from utils.embedder import ( + BaseEmbedder, + EmbedderFactory, + emb_boundary, + emb_frame_entropy, + emb_self_similarity, + emb_velocity_acceleration, + emb_features, +) + + +# --------------------------------------------------------------------------- +# Fixtures +# --------------------------------------------------------------------------- + +@pytest.fixture +def sample_embeddings(): + """Simple (T, D) embedding array for testing feature functions.""" + # 200 frames, 128-dimensional embeddings (enough for k-means with 128 clusters) + np.random.seed(42) + return np.random.randn(200, 128).astype(np.float32) + + +@pytest.fixture +def mock_wav_file(tmp_path): + """Create a mock WAV file path.""" + wav_path = tmp_path / "test.wav" + return wav_path + + +# --------------------------------------------------------------------------- +# BaseEmbedder +# --------------------------------------------------------------------------- + +class TestBaseEmbedder: + """Tests for the abstract BaseEmbedder class.""" + + def test_abstract_methods_required(self): + """Subclasses must implement _load_model and __call__.""" + with pytest.raises(TypeError): + BaseEmbedder("dummy_path") + + def test_frame_times_calculation(self): + """_frame_times returns correct centre times.""" + class DummyEmbedder(BaseEmbedder): + NAME = "dummy" + SAMPLE_RATE = 16000 + HOP_SIZE = 320 + + def _load_model(self, model_path, device): + pass + + def __call__(self, wav_path): + return np.zeros((10, 128)), np.zeros(10) + + emb = DummyEmbedder.__new__(DummyEmbedder) + emb.SAMPLE_RATE = 16000 + emb.HOP_SIZE = 320 + + times = emb._frame_times(10) + expected = (np.arange(10) + 0.5) * (320 / 16000) + np.testing.assert_array_almost_equal(times, expected) + + def test_frame_times_zero_frames(self): + """_frame_times handles zero frames.""" + class DummyEmbedder(BaseEmbedder): + NAME = "dummy" + SAMPLE_RATE = 16000 + HOP_SIZE = 320 + + def _load_model(self, model_path, device): + pass + + def __call__(self, wav_path): + return np.zeros((0, 128)), np.zeros(0) + + emb = DummyEmbedder.__new__(DummyEmbedder) + emb.SAMPLE_RATE = 16000 + emb.HOP_SIZE = 320 + + times = emb._frame_times(0) + assert len(times) == 0 + + def test_load_audio_stereo_to_mono(self, tmp_path, monkeypatch): + """_load_audio converts stereo to mono.""" + import soundfile as sf + + class DummyEmbedder(BaseEmbedder): + NAME = "dummy" + SAMPLE_RATE = 16000 + HOP_SIZE = 320 + + def _load_model(self, model_path, device): + pass + + def __call__(self, wav_path): + return np.zeros((10, 128)), np.zeros(10) + + # Create a stereo test file + wav_path = tmp_path / "stereo.wav" + stereo_audio = np.random.randn(2, 1000).astype(np.float32) * 0.01 + sf.write(str(wav_path), stereo_audio.T, 16000) + + emb = DummyEmbedder.__new__(DummyEmbedder) + emb.SAMPLE_RATE = 16000 + emb.HOP_SIZE = 320 + + audio = emb._load_audio(wav_path) + assert audio.ndim == 1 + assert len(audio) == 1000 + + def test_load_audio_resampling(self, tmp_path, monkeypatch): + """_load_audio resamples to target sample rate.""" + import soundfile as sf + + class DummyEmbedder(BaseEmbedder): + NAME = "dummy" + SAMPLE_RATE = 16000 + HOP_SIZE = 320 + + def _load_model(self, model_path, device): + pass + + def __call__(self, wav_path): + return np.zeros((10, 128)), np.zeros(10) + + # Create a 48kHz test file (should be resampled to 16kHz) + wav_path = tmp_path / "48k.wav" + audio_48k = np.random.randn(4800).astype(np.float32) * 0.01 + sf.write(str(wav_path), audio_48k, 48000) + + emb = DummyEmbedder.__new__(DummyEmbedder) + emb.SAMPLE_RATE = 16000 + emb.HOP_SIZE = 320 + + audio = emb._load_audio(wav_path) + # 4800 samples at 48kHz / 3 = 1600 samples at 16kHz + assert len(audio) == 1600 + assert audio.dtype == np.float32 + + def test_download_raises_not_implemented(self): + """BaseEmbedder.download raises NotImplementedError.""" + with pytest.raises(NotImplementedError): + BaseEmbedder.download() + + +# --------------------------------------------------------------------------- +# EmbedderFactory +# --------------------------------------------------------------------------- + +class TestEmbedderFactory: + """Tests for the EmbedderFactory registry.""" + + def test_register_valid_subclass(self): + """Registering a valid BaseEmbedder subclass succeeds.""" + class TestEmbedder(BaseEmbedder): + NAME = "test_embedder" + + def _load_model(self, model_path, device): + pass + + def __call__(self, wav_path): + return np.zeros((10, 128)), np.zeros(10) + + # Clean up before test + EmbedderFactory._registry.pop("test_embedder", None) + + result = EmbedderFactory.register(TestEmbedder) + assert result is TestEmbedder + assert "test_embedder" in EmbedderFactory._registry + + # Cleanup + del EmbedderFactory._registry["test_embedder"] + + def test_register_invalid_class(self): + """Registering a non-BaseEmbedder raises TypeError.""" + with pytest.raises(TypeError): + EmbedderFactory.register(object) + + def test_register_empty_name(self): + """Registering a class with empty NAME raises ValueError.""" + class NoNameEmbedder(BaseEmbedder): + NAME = "" + + def _load_model(self, model_path, device): + pass + + def __call__(self, wav_path): + return np.zeros((10, 128)), np.zeros(10) + + with pytest.raises(ValueError): + EmbedderFactory.register(NoNameEmbedder) + + def test_register_as_decorator(self): + """register can be used as a class decorator.""" + EmbedderFactory._registry.pop("decorator_test", None) + + @EmbedderFactory.register + class DecoratorTestEmbedder(BaseEmbedder): + NAME = "decorator_test" + + def _load_model(self, model_path, device): + pass + + def __call__(self, wav_path): + return np.zeros((10, 128)), np.zeros(10) + + assert "decorator_test" in EmbedderFactory._registry + + # Cleanup + del EmbedderFactory._registry["decorator_test"] + + def test_create_existing_embedder(self, monkeypatch): + """create() returns an instance of the registered embedder.""" + + # Ensure mhubert is available (may be registered) + try: + embedder = EmbedderFactory.create("mhubert") + assert isinstance(embedder, BaseEmbedder) + except KeyError: + pytest.skip("mhubert not registered") + + def test_create_unknown_embedder(self): + """create() with unknown name raises KeyError.""" + with pytest.raises(KeyError) as exc_info: + EmbedderFactory.create("nonexistent_embedder_xyz") + assert "nonexistent_embedder_xyz" in str(exc_info.value) + + def test_list_embedders(self): + """list() returns sorted names of registered embedders.""" + names = EmbedderFactory.list() + assert isinstance(names, list) + assert all(isinstance(n, str) for n in names) + # Should be sorted + assert names == sorted(names) + + +# --------------------------------------------------------------------------- +# Feature Extraction Functions +# --------------------------------------------------------------------------- + +class TestEmbBoundary: + """Tests for emb_boundary function.""" + + def test_basic_output_shape(self, sample_embeddings): + """Output shape matches input frame count.""" + result = emb_boundary(sample_embeddings) + assert result.shape == (sample_embeddings.shape[0],) + + def test_output_dtype(self, sample_embeddings): + """Output is float32.""" + result = emb_boundary(sample_embeddings) + assert result.dtype == np.float32 + + def test_output_non_negative(self, sample_embeddings): + """Boundary novelty curve is non-negative.""" + result = emb_boundary(sample_embeddings) + assert np.all(result >= 0) + + def test_smooth_sigma_zero(self, sample_embeddings): + """sigma=0 disables smoothing.""" + result = emb_boundary(sample_embeddings, smooth_sigma=0) + assert result.shape == (sample_embeddings.shape[0],) + + def test_smooth_sigma_large(self, sample_embeddings): + """Large sigma produces smoother output.""" + result_smooth = emb_boundary(sample_embeddings, smooth_sigma=5.0) + result_raw = emb_boundary(sample_embeddings, smooth_sigma=0) + # Smoothed should have lower variance + assert result_smooth.std() < result_raw.std() + + def test_constant_embeddings(self): + """Constant embeddings produce low boundary values.""" + emb = np.ones((100, 128), dtype=np.float32) + result = emb_boundary(emb, smooth_sigma=0) + # First diff should be zero + assert result[0] == 0.0 + + def test_single_frame(self): + """Single frame input - emb_boundary has a known issue with single frames.""" + emb = np.random.randn(1, 128).astype(np.float32) + # emb_boundary has a bug: np.diff(emb)[-1] fails when diff is empty + # This test documents the expected behavior (raises IndexError) + # In practice, embeddings will always have multiple frames + with pytest.raises(IndexError): + emb_boundary(emb) + + +class TestEmbFrameEntropy: + """Tests for emb_frame_entropy function.""" + + def test_basic_output_shape(self, sample_embeddings): + """Output shape matches input frame count.""" + result = emb_frame_entropy(sample_embeddings) + assert result.shape == (sample_embeddings.shape[0],) + + def test_output_dtype(self, sample_embeddings): + """Output is float32.""" + result = emb_frame_entropy(sample_embeddings) + assert result.dtype == np.float32 + + def test_output_range(self, sample_embeddings): + """Entropy values are in valid range [0, log(n_clusters)].""" + result = emb_frame_entropy(sample_embeddings) + assert np.all(result >= 0) + assert np.all(result <= np.log(128)) + + def test_smooth_sigma_zero(self, sample_embeddings): + """sigma=0 disables smoothing.""" + result = emb_frame_entropy(sample_embeddings, smooth_sigma=0) + assert result.shape == (sample_embeddings.shape[0],) + + def test_different_cluster_counts(self, sample_embeddings): + """Works with different n_clusters values.""" + # sample_embeddings has 200 frames, so max clusters is < 200 + for n_clusters in [16, 64, 128]: + result = emb_frame_entropy(sample_embeddings, n_clusters=n_clusters) + assert result.shape == (sample_embeddings.shape[0],) + assert np.all(result >= 0) + assert np.all(result <= np.log(n_clusters)) + + def test_temperature_effect(self, sample_embeddings): + """Higher temperature produces higher entropy.""" + result_low_temp = emb_frame_entropy(sample_embeddings, temperature=0.1) + result_high_temp = emb_frame_entropy(sample_embeddings, temperature=10.0) + # High temperature softmax tends toward uniform distribution + assert result_high_temp.mean() > result_low_temp.mean() * 0.5 + + +class TestEmbSelfSimilarity: + """Tests for emb_self_similarity function.""" + + def test_basic_output_shape(self, sample_embeddings): + """Output shape matches input frame count.""" + result = emb_self_similarity(sample_embeddings) + assert result.shape == (sample_embeddings.shape[0],) + + def test_output_dtype(self, sample_embeddings): + """Output is float32.""" + result = emb_self_similarity(sample_embeddings) + assert result.dtype == np.float32 + + def test_output_non_negative(self, sample_embeddings): + """Self-similarity novelty is non-negative.""" + result = emb_self_similarity(sample_embeddings) + assert np.all(result >= 0) + + def test_smooth_sigma_zero(self, sample_embeddings): + """sigma=0 disables smoothing.""" + result = emb_self_similarity(sample_embeddings, smooth_sigma=0) + assert result.shape == (sample_embeddings.shape[0],) + + def test_different_window_sizes(self, sample_embeddings): + """Works with different window sizes.""" + for window in [5, 10, 50]: + result = emb_self_similarity(sample_embeddings, window=window) + assert result.shape == (sample_embeddings.shape[0],) + + def test_small_window_large_data(self): + """Handles window smaller than data gracefully.""" + emb = np.random.randn(200, 128).astype(np.float32) + result = emb_self_similarity(emb, window=5) + assert result.shape == (200,) + + +class TestEmbVelocityAcceleration: + """Tests for emb_velocity_acceleration function.""" + + def test_output_shapes(self, sample_embeddings): + """Both outputs have correct shape.""" + velocity, acceleration = emb_velocity_acceleration(sample_embeddings) + assert velocity.shape == (sample_embeddings.shape[0],) + assert acceleration.shape == (sample_embeddings.shape[0],) + + def test_output_dtypes(self, sample_embeddings): + """Both outputs are float32.""" + velocity, acceleration = emb_velocity_acceleration(sample_embeddings) + assert velocity.dtype == np.float32 + assert acceleration.dtype == np.float32 + + def test_output_non_negative(self, sample_embeddings): + """Velocity and acceleration are non-negative.""" + velocity, acceleration = emb_velocity_acceleration(sample_embeddings) + assert np.all(velocity >= 0) + assert np.all(acceleration >= 0) + + def test_smooth_sigma_zero(self, sample_embeddings): + """sigma=0 disables smoothing.""" + velocity, acceleration = emb_velocity_acceleration(sample_embeddings, smooth_sigma=0) + assert velocity.shape == (sample_embeddings.shape[0],) + assert acceleration.shape == (sample_embeddings.shape[0],) + + def test_constant_embeddings(self): + """Constant embeddings produce zero velocity and acceleration.""" + emb = np.ones((100, 128), dtype=np.float32) + velocity, acceleration = emb_velocity_acceleration(emb) + np.testing.assert_array_almost_equal(velocity, 0) + np.testing.assert_array_almost_equal(acceleration, 0) + + def test_linear_change(self): + """Linear change produces constant velocity, zero acceleration.""" + emb = np.linspace(0, 1, 100, dtype=np.float32)[:, np.newaxis] * np.ones(128) + velocity, acceleration = emb_velocity_acceleration(emb) + # Velocity should be constant, acceleration near zero + assert velocity.std() < velocity.mean() * 0.1 + + +class TestEmbFeatures: + """Tests for emb_features function.""" + + def test_basic_output_shape(self, sample_embeddings): + """Output has 5 feature rows matching input frames.""" + result = emb_features(sample_embeddings) + assert result.shape == (5, sample_embeddings.shape[0]) + + def test_output_dtype(self, sample_embeddings): + """Output is float32.""" + result = emb_features(sample_embeddings) + assert result.dtype == np.float32 + + def test_row_order(self, sample_embeddings): + """Rows are in expected order: boundary, entropy, similarity, velocity, acceleration.""" + result = emb_features(sample_embeddings) + boundary, entropy, similarity, velocity, acceleration = result + + # Check boundary features + expected_boundary = emb_boundary(sample_embeddings) + np.testing.assert_array_almost_equal(result[0], expected_boundary) + + # Check entropy features + expected_entropy = emb_frame_entropy(sample_embeddings) + np.testing.assert_array_almost_equal(result[1], expected_entropy) + + # Check similarity features + expected_similarity = emb_self_similarity(sample_embeddings) + np.testing.assert_array_almost_equal(result[2], expected_similarity) + + # Check velocity/acceleration features + expected_velocity, expected_acceleration = emb_velocity_acceleration(sample_embeddings) + np.testing.assert_array_almost_equal(result[3], expected_velocity) + np.testing.assert_array_almost_equal(result[4], expected_acceleration) + + def test_custom_parameters(self, sample_embeddings): + """All custom parameters are passed through correctly.""" + result = emb_features( + sample_embeddings, + boundary_sigma=2.0, + entropy_clusters=64, + entropy_temperature=0.5, + entropy_sigma=2.0, + similarity_window=10, + similarity_sigma=2.0, + velocity_sigma=2.0, + ) + assert result.shape == (5, sample_embeddings.shape[0]) + + def test_column_wise_operations(self, sample_embeddings): + """Features are computed column-wise (per frame), not row-wise.""" + result = emb_features(sample_embeddings) + # Each column is a frame, each row is a feature type + assert result.shape[1] == sample_embeddings.shape[0] + assert result.shape[0] == 5 + + +# --------------------------------------------------------------------------- +# Integration-like tests (mocked audio loading) +# --------------------------------------------------------------------------- + +class TestEmbedderIntegration: + """Integration-style tests using mocked audio loading.""" + + def test_full_pipeline_boundary(self, tmp_path, monkeypatch): + """Full pipeline: load audio -> extract embeddings -> compute boundary.""" + import soundfile as sf + + # Create test audio file + wav_path = tmp_path / "test_audio.wav" + audio = np.random.randn(16000).astype(np.float32) * 0.01 # 1 second at 16kHz + sf.write(str(wav_path), audio, 16000) + + # Use mock embedder + class MockEmbedder(BaseEmbedder): + NAME = "mock" + SAMPLE_RATE = 16000 + HOP_SIZE = 320 + + def _load_model(self, model_path, device): + pass + + def __call__(self, wav_path): + # Return 50 frames of 128-dim embeddings + emb = np.random.randn(50, 128).astype(np.float32) + times = np.arange(50) * (320 / 16000) + return emb, times + + embedder = MockEmbedder(model_path="dummy") + embeddings, times = embedder(wav_path) + + # Compute boundary features + boundary = emb_boundary(embeddings) + + assert boundary.shape == (50,) + assert np.all(boundary >= 0) + + def test_feature_stacking_pipeline(self, tmp_path, monkeypatch): + """Test stacking all features from embeddings.""" + class MockEmbedder(BaseEmbedder): + NAME = "mock_stack" + SAMPLE_RATE = 16000 + HOP_SIZE = 320 + + def __init__(self, model_path=None, device="cpu"): + self._device = device + self._model_path = None + + def _load_model(self, model_path, device): + pass + + def __call__(self, wav_path): + emb = np.random.randn(200, 128).astype(np.float32) + return emb, np.zeros(200) + + EmbedderFactory._registry.pop("mock_stack", None) + EmbedderFactory.register(MockEmbedder) + + embedder = EmbedderFactory.create("mock_stack") + embeddings, _unused = embedder("dummy.wav") + + features = emb_features(embeddings) + + assert features.shape == (5, 200) + assert features.dtype == np.float32 + + # Cleanup + del EmbedderFactory._registry["mock_stack"] + + +# --------------------------------------------------------------------------- +# Edge cases +# --------------------------------------------------------------------------- + +class TestEmbedderEdgeCases: + """Edge case tests for embedder functionality.""" + + def test_single_frame_embedding(self): + """Handles single frame embedding gracefully.""" + # Need at least 2 frames for boundary, and 128+ for entropy k-means + emb = np.random.randn(200, 128).astype(np.float32) + boundary = emb_boundary(emb) + entropy = emb_frame_entropy(emb) + similarity = emb_self_similarity(emb) + velocity, acceleration = emb_velocity_acceleration(emb) + + assert boundary.shape == (200,) + assert entropy.shape == (200,) + assert similarity.shape == (200,) + assert velocity.shape == (200,) + assert acceleration.shape == (200,) + + def test_large_embedding_matrix(self): + """Handles large embedding matrices.""" + emb = np.random.randn(10000, 128).astype(np.float32) + features = emb_features(emb) + assert features.shape == (5, 10000) + + def test_highdimensional_embeddings(self): + """Handles high-dimensional embeddings.""" + # Need 128+ frames for entropy k-means with 128 clusters + emb = np.random.randn(200, 1024).astype(np.float32) + features = emb_features(emb) + assert features.shape == (5, 200) + + def test_nearzero_embeddings(self): + """Handles near-zero embeddings.""" + emb = np.random.randn(50, 128).astype(np.float32) * 1e-10 + boundary = emb_boundary(emb) + assert boundary.shape == (50,) + + def test_identical_embeddings(self): + """Handles identical frame embeddings.""" + emb = np.tile([1.0, 2.0], (50, 64)).astype(np.float32) + boundary = emb_boundary(emb, smooth_sigma=0) + # All boundary values should be zero + np.testing.assert_array_almost_equal(boundary, 0) + + def test_perfectly_periodic_embeddings(self): + """Handles perfectly periodic embedding patterns.""" + # Create embeddings with repeating pattern + pattern = np.sin(np.linspace(0, 4 * np.pi, 128)).astype(np.float32) + emb = np.tile(pattern, (50, 1)) + boundary = emb_boundary(emb) + # Boundary should show periodic spikes + assert boundary.shape == (50,) diff --git a/tests/test_expressive.py b/tests/test_expressive.py index acf9bf0..25fb8c4 100644 --- a/tests/test_expressive.py +++ b/tests/test_expressive.py @@ -443,6 +443,14 @@ def test_process_expressions_case_sensitive( class TestSetupLoggers: + """ + setup_loggers() configures two loggers: + + logger_app — gets file_handler + app_handler (FileHandler + StreamHandler) + logger_exp — gets file_handler + exp_handler (FileHandler + StreamHandler) + + Both therefore have exactly 2 handlers. + """ def test_setup_loggers_creates_log_file(self): with setup_loggers() as (logger_app, logger_exp, log_path): @@ -459,20 +467,28 @@ def test_setup_loggers_returns_correct_types(self): assert isinstance(log_path, Path) Path(log_path).unlink() - def test_setup_loggers_configures_handlers(self): - with setup_loggers() as (logger_app, logger_exp, log_path): + def test_setup_loggers_app_has_two_handlers(self): + """logger_app receives file_handler + app_handler.""" + with setup_loggers() as (logger_app, _unused, log_path): assert len(logger_app.handlers) == 2 - handler_types = [type(h).__name__ for h in logger_app.handlers] - assert 'FileHandler' in handler_types - assert 'StreamHandler' in handler_types - assert len(logger_exp.handlers) == 1 - assert isinstance(logger_exp.handlers[0], logging.FileHandler) + types_ = {type(h).__name__ for h in logger_app.handlers} + assert "FileHandler" in types_ + assert "StreamHandler" in types_ + Path(log_path).unlink() + + def test_setup_loggers_exp_has_two_handlers(self): + """logger_exp receives file_handler + exp_handler (both added in setup_loggers).""" + with setup_loggers() as (_unused, logger_exp, log_path): + assert len(logger_exp.handlers) == 2 + types_ = {type(h).__name__ for h in logger_exp.handlers} + assert "FileHandler" in types_ + assert "StreamHandler" in types_ Path(log_path).unlink() def test_setup_loggers_sets_debug_level(self): with setup_loggers() as (logger_app, logger_exp, log_path): assert logger_app.level == logging.DEBUG - assert logger_exp.level == logging.DEBUG + assert logger_exp.level == logging.DEBUG Path(log_path).unlink() def test_setup_loggers_writes_to_file(self): @@ -480,12 +496,13 @@ def test_setup_loggers_writes_to_file(self): logger_app.info("Test message from app") logger_exp.debug("Test message from exp") - log_content = Path(log_path).read_text(encoding='utf-8-sig') + log_content = Path(log_path).read_text(encoding="utf-8-sig") assert "Test message from app" in log_content assert "Test message from exp" in log_content Path(log_path).unlink() def test_setup_loggers_cleanup_on_exit(self): + """All handlers are removed from both loggers after the context exits.""" with setup_loggers() as (logger_app, logger_exp, log_path): assert len(logger_app.handlers) > 0 assert len(logger_exp.handlers) > 0 @@ -494,6 +511,7 @@ def test_setup_loggers_cleanup_on_exit(self): Path(log_path).unlink() def test_setup_loggers_cleanup_on_exception(self): + """Handlers are removed even when the body raises.""" logger_app = logger_exp = log_path = None try: with setup_loggers() as (la, le, lp): @@ -508,21 +526,333 @@ def test_setup_loggers_cleanup_on_exception(self): def test_setup_loggers_log_file_naming(self): with setup_loggers() as (_, __, log_path): log_filename = Path(log_path).name - assert log_filename.startswith(datetime.now().strftime('%Y%m%d_')) + assert log_filename.startswith(datetime.now().strftime("%Y%m%d_")) assert log_filename.endswith(".log") Path(log_path).unlink() def test_setup_loggers_writes_final_message(self): + """The 'Logs saved to …' message is written during teardown.""" with setup_loggers() as (_, __, log_path): pass - log_content = Path(log_path).read_text(encoding='utf-8-sig') + log_content = Path(log_path).read_text(encoding="utf-8-sig") assert f"Logs saved to '{log_path}'" in log_content Path(log_path).unlink() def test_setup_loggers_file_encoding(self): with setup_loggers() as (logger_app, _, log_path): logger_app.info("Test with unicode: 你好世界 Привет мир") - log_content = Path(log_path).read_text(encoding='utf-8-sig') - assert "你好世界" in log_content + log_content = Path(log_path).read_text(encoding="utf-8-sig") + assert "你好世界" in log_content assert "Привет мир" in log_content Path(log_path).unlink() + + +# --------------------------------------------------------------------------- +# SameFileError handling in process_expressions (lines 86-87) +# --------------------------------------------------------------------------- + +class TestSameFileError: + + @patch('expressive.copy') + @patch('expressive.getExpressionLoader') + @patch('expressive.get_registered_expressions') + def test_same_file_error_is_silenced( + self, mock_get_registered, mock_get_loader, mock_copy + ): + """SameFileError raised by copy() must be caught and silently ignored.""" + from shutil import SameFileError + mock_copy.side_effect = SameFileError + mock_get_registered.return_value = [] + + # Must not raise + process_expressions( + utau_wav="utau.wav", ref_wav="ref.wav", + ustx_input="same.ustx", ustx_output="same.ustx", + track_number=1, expressions=[], + **_TS, + ) + + @patch('expressive.copy') + @patch('expressive.getExpressionLoader') + @patch('expressive.get_registered_expressions') + def test_same_file_error_still_processes_expressions( + self, mock_get_registered, mock_get_loader, mock_copy + ): + """Processing must continue normally after a SameFileError from copy().""" + from shutil import SameFileError + mock_copy.side_effect = SameFileError + mock_get_registered.return_value = ['dyn'] + + mock_loader_instance = Mock() + mock_loader_instance.get_args_dict.return_value = {} + mock_loader_class = Mock(return_value=mock_loader_instance) + mock_get_loader.return_value = mock_loader_class + + process_expressions( + utau_wav="utau.wav", ref_wav="ref.wav", + ustx_input="same.ustx", ustx_output="same.ustx", + track_number=1, + expressions=[{"expression": "dyn"}], + **_TS, + ) + + mock_loader_instance.get_expression.assert_called_once() + mock_loader_instance.load_to_ustx.assert_called_once_with(1) + + @patch('expressive.copy') + @patch('expressive.get_registered_expressions') + def test_other_copy_errors_propagate( + self, mock_get_registered, mock_copy + ): + """Errors other than SameFileError raised by copy() must not be swallowed.""" + mock_copy.side_effect = OSError("disk full") + mock_get_registered.return_value = [] + + with pytest.raises(OSError, match="disk full"): + process_expressions( + utau_wav="utau.wav", ref_wav="ref.wav", + ustx_input="input.ustx", ustx_output="output.ustx", + track_number=1, expressions=[], + **_TS, + ) + + +# --------------------------------------------------------------------------- +# main() — argument parsing and dispatch (lines 154-210) +# --------------------------------------------------------------------------- + +# Shared helpers for building a mock argument namespace +def _make_general_arg(type_=str, default=None, help_=""): + a = Mock() + a.type = type_ + a.default = default + a.help = help_ + return a + + +def _make_loader_args_mock(): + """Return a mock for getExpressionLoader(None) whose .args namespace covers + every general argument that main() accesses.""" + loader_mock = Mock() + loader_mock.args.utau_path = _make_general_arg() + loader_mock.args.ref_path = _make_general_arg() + loader_mock.args.ustx_path = _make_general_arg() + loader_mock.args.track_number = _make_general_arg(type_=int) + loader_mock.args.utau_start = _make_general_arg(default=None) + loader_mock.args.utau_end = _make_general_arg(default=None) + loader_mock.args.ref_start = _make_general_arg(default=None) + loader_mock.args.ref_end = _make_general_arg(default=None) + loader_mock.get_args_dict.return_value = {} + loader_mock.__name__ = "MockExpressionLoader" + return loader_mock + + +class TestMain: + """Tests for expressive.main(). + + argparse is exercised by injecting sys.argv; all heavy collaborators + (process_expressions, setup_loggers, getExpressionLoader, …) are mocked. + """ + + # Baseline argv that satisfies every required argument. + _BASE_ARGV = [ + "expressive", + "-u", "utau.wav", + "-r", "ref.wav", + "-i", "input.ustx", + "-o", "output.ustx", + "-t", "1", + "-e", "dyn", + ] + + def _run_main(self, argv=None, expressions=("dyn",), extra_loader_args=None): + """Call main() with a fully-mocked environment. + + Returns (mock_process_expressions, mock_logger_app). + """ + from expressive import main + + loader_mock = _make_loader_args_mock() + if extra_loader_args is not None: + loader_mock.get_args_dict.return_value = extra_loader_args + + mock_logger_app = Mock() + mock_logger_exp = Mock() + mock_log_path = Mock() + + with patch("sys.argv", argv or self._BASE_ARGV), \ + patch("expressive.init_gettext"), \ + patch("expressive.get_registered_expressions", return_value=list(expressions)), \ + patch("expressive.getExpressionLoader", return_value=loader_mock), \ + patch("expressive.add_expression_args_group"), \ + patch("expressive.process_expressions") as mock_pe, \ + patch("expressive.setup_loggers") as mock_sl: + + mock_sl.return_value.__enter__ = Mock( + return_value=(mock_logger_app, mock_logger_exp, mock_log_path) + ) + mock_sl.return_value.__exit__ = Mock(return_value=False) + + main() + + return mock_pe, mock_logger_app + + # --- parser construction --- + + def test_main_calls_init_gettext(self): + from expressive import main + loader_mock = _make_loader_args_mock() + with patch("sys.argv", self._BASE_ARGV), \ + patch("expressive.init_gettext") as mock_ig, \ + patch("expressive.get_registered_expressions", return_value=["dyn"]), \ + patch("expressive.getExpressionLoader", return_value=loader_mock), \ + patch("expressive.add_expression_args_group"), \ + patch("expressive.process_expressions"), \ + patch("expressive.setup_loggers") as mock_sl: + mock_sl.return_value.__enter__ = Mock(return_value=(Mock(), Mock(), Mock())) + mock_sl.return_value.__exit__ = Mock(return_value=False) + main() + mock_ig.assert_called_once() + + def test_main_calls_process_expressions(self): + mock_pe, _ = self._run_main() + mock_pe.assert_called_once() + + def test_main_passes_utau_wav(self): + mock_pe, _ = self._run_main() + _, kwargs = mock_pe.call_args + assert mock_pe.call_args[0][0] == "utau.wav" + + def test_main_passes_ref_wav(self): + mock_pe, _ = self._run_main() + assert mock_pe.call_args[0][1] == "ref.wav" + + def test_main_passes_ustx_input(self): + mock_pe, _ = self._run_main() + assert mock_pe.call_args[0][2] == "input.ustx" + + def test_main_passes_ustx_output(self): + mock_pe, _ = self._run_main() + assert mock_pe.call_args[0][3] == "output.ustx" + + def test_main_passes_track_number(self): + mock_pe, _ = self._run_main() + assert mock_pe.call_args[0][4] == 1 + + def test_main_passes_expressions_list(self): + mock_pe, _ = self._run_main() + expressions = mock_pe.call_args[0][9] + assert isinstance(expressions, list) + assert any(e["expression"] == "dyn" for e in expressions) + + def test_main_default_timestamps_are_none(self): + mock_pe, _ = self._run_main() + args = mock_pe.call_args[0] + # ref_start=5, ref_end=6, utau_start=7, utau_end=8 + assert args[5] is None # ref_start + assert args[6] is None # ref_end + assert args[7] is None # utau_start + assert args[8] is None # utau_end + + def test_main_passes_explicit_timestamps(self): + argv = self._BASE_ARGV + [ + "--ref_start", "0:10", + "--ref_end", "1:30", + "--utau_start", "0:05", + "--utau_end", "1:25", + ] + mock_pe, _ = self._run_main(argv=argv) + args = mock_pe.call_args[0] + assert args[5] == "0:10" + assert args[6] == "1:30" + assert args[7] == "0:05" + assert args[8] == "1:25" + + # --- logging behaviour --- + + def test_main_logs_starting_message(self): + _, mock_logger_app = self._run_main() + logged = " ".join(str(c) for c in mock_logger_app.info.call_args_list) + assert "Starting" in logged or mock_logger_app.info.called + + def test_main_logs_success_on_no_exception(self): + _, mock_logger_app = self._run_main() + success_calls = [ + c for c in mock_logger_app.info.call_args_list + if "completed" in str(c).lower() or "successfully" in str(c).lower() + ] + assert len(success_calls) >= 1 + + def test_main_logs_exception_on_error(self): + from expressive import main + loader_mock = _make_loader_args_mock() + mock_logger_app = Mock() + + with patch("sys.argv", self._BASE_ARGV), \ + patch("expressive.init_gettext"), \ + patch("expressive.get_registered_expressions", return_value=["dyn"]), \ + patch("expressive.getExpressionLoader", return_value=loader_mock), \ + patch("expressive.add_expression_args_group"), \ + patch("expressive.process_expressions", side_effect=RuntimeError("boom")), \ + patch("expressive.setup_loggers") as mock_sl: + mock_sl.return_value.__enter__ = Mock( + return_value=(mock_logger_app, Mock(), Mock()) + ) + mock_sl.return_value.__exit__ = Mock(return_value=False) + main() + + mock_logger_app.exception.assert_called_once() + + def test_main_does_not_raise_on_process_error(self): + """main() must not propagate exceptions from process_expressions.""" + from expressive import main + loader_mock = _make_loader_args_mock() + + with patch("sys.argv", self._BASE_ARGV), \ + patch("expressive.init_gettext"), \ + patch("expressive.get_registered_expressions", return_value=["dyn"]), \ + patch("expressive.getExpressionLoader", return_value=loader_mock), \ + patch("expressive.add_expression_args_group"), \ + patch("expressive.process_expressions", side_effect=RuntimeError("boom")), \ + patch("expressive.setup_loggers") as mock_sl: + mock_sl.return_value.__enter__ = Mock( + return_value=(Mock(), Mock(), Mock()) + ) + mock_sl.return_value.__exit__ = Mock(return_value=False) + main() # must not raise + + # --- expression filtering --- + + def test_main_only_includes_selected_expressions(self): + """Expressions not in -e flags must be excluded from the call.""" + argv = [ + "expressive", + "-u", "utau.wav", "-r", "ref.wav", + "-i", "input.ustx", "-o", "output.ustx", + "-t", "1", + "-e", "dyn", # only dyn, not pitd + ] + mock_pe, _ = self._run_main(argv=argv, expressions=("dyn", "pitd")) + expressions = mock_pe.call_args[0][9] + names = [e["expression"] for e in expressions] + assert "dyn" in names + assert "pitd" not in names + + def test_main_includes_multiple_expressions(self): + argv = [ + "expressive", + "-u", "utau.wav", "-r", "ref.wav", + "-i", "input.ustx", "-o", "output.ustx", + "-t", "1", + "-e", "dyn", "-e", "pitd", + ] + mock_pe, _ = self._run_main(argv=argv, expressions=("dyn", "pitd")) + expressions = mock_pe.call_args[0][9] + names = [e["expression"] for e in expressions] + assert "dyn" in names + assert "pitd" in names + + def test_main_expression_dict_contains_expression_key(self): + mock_pe, _ = self._run_main() + for exp in mock_pe.call_args[0][9]: + assert "expression" in exp diff --git a/tests/test_fs.py b/tests/test_fs.py index 0b79e17..660489c 100644 --- a/tests/test_fs.py +++ b/tests/test_fs.py @@ -2,7 +2,7 @@ import pytest -from utils.fs import APP_CACHE_DIR, calculate_file_hash, clear_cache +from utils.fs import APP_CACHE_DIR, calculate_args_hash, calculate_file_hash, clear_cache class TestCalculateFileHash: @@ -15,8 +15,8 @@ def test_calculate_file_hash_basic(self, tmp_path): hash_result = calculate_file_hash(str(test_file)) - # SHA-256 of "Hello, World!" - expected_hash = "dffd6021bb2bd5b0af676290809ec3a53191dd81c7f70a4b28688a362182986f" + # calculate_file_hash returns only the first 16 hex characters + expected_hash = "dffd6021bb2bd5b0" assert hash_result == expected_hash def test_calculate_file_hash_empty_file(self, tmp_path): @@ -26,19 +26,19 @@ def test_calculate_file_hash_empty_file(self, tmp_path): hash_result = calculate_file_hash(str(test_file)) - # SHA-256 of empty string - expected_hash = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + # First 16 hex chars of SHA-256 of empty string + expected_hash = "e3b0c44298fc1c14" assert hash_result == expected_hash - def test_calculate_file_hash_binary_file(self, tmp_path): - """Test calculating hash of a binary file.""" + def test_calculate_file_hash_returns_16_chars(self, tmp_path): + """Hash is always truncated to 16 hex characters.""" test_file = tmp_path / "binary.bin" test_file.write_bytes(b"\x00\x01\x02\x03\x04\x05") hash_result = calculate_file_hash(str(test_file)) assert isinstance(hash_result, str) - assert len(hash_result) == 64 # SHA-256 produces 64 hex characters + assert len(hash_result) == 16 def test_calculate_file_hash_large_file(self, tmp_path): """Test calculating hash of a large file (tests chunking).""" @@ -50,7 +50,7 @@ def test_calculate_file_hash_large_file(self, tmp_path): hash_result = calculate_file_hash(str(test_file)) assert isinstance(hash_result, str) - assert len(hash_result) == 64 + assert len(hash_result) == 16 def test_calculate_file_hash_same_content_same_hash(self, tmp_path): """Test that identical files produce the same hash.""" @@ -84,40 +84,89 @@ def test_calculate_file_hash_nonexistent_file(self): with pytest.raises(FileNotFoundError): calculate_file_hash("/nonexistent/file.txt") + def test_calculate_file_hash_only_hex_chars(self, tmp_path): + """Hash must contain only lowercase hex characters.""" + test_file = tmp_path / "hex.txt" + test_file.write_text("hex check") + hash_result = calculate_file_hash(str(test_file)) + assert all(c in "0123456789abcdef" for c in hash_result) + + def test_calculate_file_hash_deterministic(self, tmp_path): + """Calling twice on the same file yields the same result.""" + test_file = tmp_path / "det.txt" + test_file.write_text("determinism") + assert calculate_file_hash(str(test_file)) == calculate_file_hash(str(test_file)) + + +class TestCalculateArgsHash: + """Test argument hash calculation.""" + + def test_returns_string(self): + result = calculate_args_hash(1, 2, key="value") + assert isinstance(result, str) + + def test_returns_16_chars(self): + result = calculate_args_hash("a", "b") + assert len(result) == 16 + + def test_same_args_same_hash(self): + h1 = calculate_args_hash(1, 2, x=3) + h2 = calculate_args_hash(1, 2, x=3) + assert h1 == h2 + + def test_different_args_different_hash(self): + h1 = calculate_args_hash(1, 2) + h2 = calculate_args_hash(1, 3) + assert h1 != h2 + + def test_kwargs_order_invariant(self): + """Keyword argument order must not change the hash (sorted internally).""" + h1 = calculate_args_hash(a=1, b=2) + h2 = calculate_args_hash(b=2, a=1) + assert h1 == h2 + + def test_no_args_returns_string(self): + result = calculate_args_hash() + assert isinstance(result, str) + assert len(result) == 16 + + def test_positional_and_keyword_distinguished(self): + """Positional and keyword args with the same value should differ.""" + h_pos = calculate_args_hash(1) + h_kw = calculate_args_hash(x=1) + assert h_pos != h_kw + + def test_none_args_stable(self): + h1 = calculate_args_hash(None) + h2 = calculate_args_hash(None) + assert h1 == h2 + class TestClearCache: """Test cache clearing functionality.""" def test_clear_cache_when_exists(self, tmp_path, monkeypatch): """Test clearing cache when directory exists.""" - # Mock APP_CACHE_DIR to use temp directory mock_cache_dir = tmp_path / "cache" mock_cache_dir.mkdir() - # Create some files in the cache (mock_cache_dir / "file1.txt").write_text("data1") (mock_cache_dir / "file2.txt").write_text("data2") - # Patch APP_CACHE_DIR monkeypatch.setattr("utils.fs.APP_CACHE_DIR", str(mock_cache_dir)) - # Clear cache clear_cache() - # Verify directory was removed assert not mock_cache_dir.exists() def test_clear_cache_when_not_exists(self, tmp_path, monkeypatch): """Test clearing cache when directory doesn't exist.""" mock_cache_dir = tmp_path / "nonexistent_cache" - # Patch APP_CACHE_DIR monkeypatch.setattr("utils.fs.APP_CACHE_DIR", str(mock_cache_dir)) - # Should not raise error clear_cache() - # Directory should still not exist assert not mock_cache_dir.exists() def test_clear_cache_with_subdirectories(self, tmp_path, monkeypatch): @@ -125,19 +174,15 @@ def test_clear_cache_with_subdirectories(self, tmp_path, monkeypatch): mock_cache_dir = tmp_path / "cache" mock_cache_dir.mkdir() - # Create nested structure subdir = mock_cache_dir / "subdir" subdir.mkdir() (subdir / "nested_file.txt").write_text("nested data") (mock_cache_dir / "root_file.txt").write_text("root data") - # Patch APP_CACHE_DIR monkeypatch.setattr("utils.fs.APP_CACHE_DIR", str(mock_cache_dir)) - # Clear cache clear_cache() - # Verify entire tree was removed assert not mock_cache_dir.exists() assert not subdir.exists() diff --git a/tests/test_wavtool.py b/tests/test_wavtool.py index e87c7db..c63682d 100644 --- a/tests/test_wavtool.py +++ b/tests/test_wavtool.py @@ -13,6 +13,7 @@ from utils.wavtool import ( ClampedWav, + extract_wav_breath_voice, extract_wav_frequency, extract_wav_mfcc, extract_wav_rms, @@ -641,9 +642,7 @@ def test_rms_time_does_not_exceed_duration(self): def test_silent_wav_masked_with_nan(self): """mask_silence=True on a fully-silent file must not raise and must return - arrays of equal length. Otsu's threshold on a flat-zero signal is 0, so - no frames satisfy rms < 0; the masking logic runs but clips nothing — - we only assert the call succeeds and output shapes are consistent.""" + arrays of equal length.""" rms_time, rms = extract_wav_rms(self.wav, mask_silence=True) self.assertEqual(rms_time.shape, rms.shape) self.assertGreater(len(rms), 0) @@ -749,8 +748,6 @@ def tearDown(self): pass def _patch_swift(self): - """Patch SwiftF0 on its home module so the local import inside - extract_wav_frequency picks up the replacement.""" fake_result = MagicMock() fake_result.timestamps.tolist.return_value = self._fake_times fake_result.pitch_hz.tolist.return_value = self._fake_freqs @@ -762,8 +759,6 @@ def _patch_swift(self): return patch("swift_f0.SwiftF0", mock_cls, create=True) def _patch_rmvpe(self, times=None, freqs=None, confs=None): - """Patch RMVPE and soundfile so the rmvpe-onnx backend runs without - real model weights or audio I/O. Optionally override return values.""" fake_timestamp = np.array(times if times is not None else self._fake_times) fake_frequency = np.array(freqs if freqs is not None else self._fake_freqs) fake_confidence = np.array(confs if confs is not None else self._fake_confs) @@ -851,12 +846,10 @@ def test_swift_f0_backend_accepted(self): extract_wav_frequency(self.wav, backend="swift-f0", use_cache=False) def test_rmvpe_onnx_backend_accepted(self): - """rmvpe-onnx is the default backend and must be accepted (mocked).""" with self._patch_rmvpe(): extract_wav_frequency(self.wav, backend="rmvpe-onnx", use_cache=False) def test_rmvpe_onnx_is_default_backend(self): - """Calling extract_wav_frequency without backend= should use rmvpe-onnx.""" import inspect sig = inspect.signature(extract_wav_frequency) self.assertEqual(sig.parameters["backend"].default, "rmvpe-onnx") @@ -871,7 +864,6 @@ def test_rmvpe_onnx_returns_correct_shapes(self): self.assertEqual(len(conf), self._n) def test_crepe_backend_accepted(self): - """crepe backend path should be accepted (mocked to avoid TF dependency).""" fake_time = np.linspace(0, 2, self._n) fake_freq = np.random.uniform(80, 300, self._n) fake_conf = np.random.uniform(0.5, 1.0, self._n) @@ -894,9 +886,7 @@ def test_crepe_backend_accepted(self): # --- hybrid backend --- def test_hybrid_backend_accepted(self): - """hybrid backend must be accepted without raising ValueError.""" with self._patch_rmvpe(), self._patch_swift(): - # soundfile is also used inside _merge_rmvpe_and_swift_f0 extract_wav_frequency(self.wav, backend="hybrid", use_cache=False) def test_hybrid_returns_tuple_of_three(self): @@ -919,7 +909,6 @@ def test_hybrid_all_outputs_same_length(self): self.assertEqual(len(freq), len(conf)) def test_hybrid_output_length_matches_rmvpe_grid(self): - """Hybrid uses the rmvpe-onnx time grid, so output length == rmvpe output length.""" with self._patch_rmvpe(), self._patch_swift(): time, freq, conf = extract_wav_frequency(self.wav, backend="hybrid", use_cache=False) self.assertEqual(len(time), self._n) @@ -956,21 +945,15 @@ def test_hybrid_replaces_low_confidence_rmvpe_frames(self): region, the hybrid output should include some swift-f0 frequency values.""" n = 50 times = list(np.linspace(0, 1.0, n)) - - # rmvpe: low confidence everywhere so swift-f0 replacements will be chosen rmvpe_freqs = [200.0] * n rmvpe_confs = [0.5] * n # below threshold 0.80 - - # swift-f0: high confidence + different frequency swift_freqs = [400.0] * n swift_confs = [0.99] * n # above threshold 0.95 - # Build a tonal WAV so RMS > 0 (voiced region) wav = _make_tonal_wav(duration=1.0, sr=22050, freq=440.0) try: with self._patch_rmvpe(times=times, freqs=rmvpe_freqs, confs=rmvpe_confs), \ self._patch_swift(): - # Override swift-f0 mock with specific values fake_result = MagicMock() fake_result.timestamps.tolist.return_value = times fake_result.pitch_hz.tolist.return_value = swift_freqs @@ -981,7 +964,6 @@ def test_hybrid_replaces_low_confidence_rmvpe_frames(self): with patch("swift_f0.SwiftF0", mock_cls, create=True): _, freq, _ = extract_wav_frequency(wav, backend="hybrid", use_cache=False) - # At least some frames should have been replaced with 400 Hz self.assertTrue(np.any(np.isclose(freq, 400.0)), "Expected some frames to be replaced by swift-f0 (400 Hz)") finally: @@ -992,8 +974,7 @@ def test_hybrid_keeps_rmvpe_frames_when_rmvpe_confident(self): n = 50 times = list(np.linspace(0, 1.0, n)) rmvpe_freqs = [200.0] * n - rmvpe_confs = [0.95] * n # above threshold 0.80 — should not be replaced - + rmvpe_confs = [0.95] * n # above threshold 0.80 swift_freqs = [400.0] * n swift_confs = [0.99] * n @@ -1010,13 +991,13 @@ def test_hybrid_keeps_rmvpe_frames_when_rmvpe_confident(self): with patch("swift_f0.SwiftF0", mock_cls, create=True): _, freq, _ = extract_wav_frequency(wav, backend="hybrid", use_cache=False) - # All frames should stay at 200 Hz (rmvpe confident, no replacement) self.assertTrue(np.all(np.isclose(freq, 200.0)), "Expected rmvpe frequencies to be kept when confidence is high") finally: os.unlink(wav) # --- caching --- + # NOTE: wavtool.py uses "f0" as the cache subdirectory, not "pitd". def test_cache_file_written_when_use_cache_true(self): tmp_cache_dir = tempfile.mkdtemp() @@ -1025,14 +1006,16 @@ def test_cache_file_written_when_use_cache_true(self): patch("utils.wavtool.calculate_file_hash", return_value="testhash"): extract_wav_frequency(self.wav, backend="swift-f0", use_cache=True) - cache_path = os.path.join(tmp_cache_dir, "pitd", "testhash.swift-f0.csv") + # Cache subdirectory is "f0", not "pitd" + cache_path = os.path.join(tmp_cache_dir, "f0", "testhash.swift-f0.csv") self.assertTrue(os.path.exists(cache_path)) def test_cache_read_skips_backend_call(self): """If a valid cache file exists the backend must not be invoked.""" tmp_cache_dir = tempfile.mkdtemp() fake_hash = "deadbeef" - cache_path = os.path.join(tmp_cache_dir, "pitd", f"{fake_hash}.swift-f0.csv") + # Cache subdirectory is "f0" + cache_path = os.path.join(tmp_cache_dir, "f0", f"{fake_hash}.swift-f0.csv") os.makedirs(os.path.dirname(cache_path), exist_ok=True) with open(cache_path, "w", newline="") as f: @@ -1045,7 +1028,6 @@ def test_cache_read_skips_backend_call(self): patch("utils.wavtool.APP_CACHE_DIR", tmp_cache_dir), \ patch("utils.wavtool.calculate_file_hash", return_value=fake_hash): time, freq, conf = extract_wav_frequency(self.wav, backend="swift-f0", use_cache=True) - # The SwiftF0 class must never have been instantiated mock_swift_cls.assert_not_called() np.testing.assert_array_equal(time, [0.0, 0.5]) @@ -1064,10 +1046,354 @@ def test_no_cache_written_when_use_cache_false(self): patch("utils.wavtool.calculate_file_hash", return_value="abc123"): extract_wav_frequency(self.wav, backend="swift-f0", use_cache=False) - pitd_dir = os.path.join(tmp_cache_dir, "pitd") - cache_files = os.listdir(pitd_dir) if os.path.exists(pitd_dir) else [] + # "f0" subdir should not exist (no cache written) + f0_dir = os.path.join(tmp_cache_dir, "f0") + cache_files = os.listdir(f0_dir) if os.path.exists(f0_dir) else [] self.assertEqual(cache_files, []) +# --------------------------------------------------------------------------- +# extract_wav_breath_voice +# --------------------------------------------------------------------------- + +class TestExtractWavBreathVoice(unittest.TestCase): + """Tests for extract_wav_breath_voice. + + librosa and soundfile I/O are used directly (no ML models), so we only + need real WAV files — no heavy mocking required. + """ + + def setUp(self): + self.wav = _make_tonal_wav(duration=2.0, sr=22050, freq=440.0) + + def tearDown(self): + try: + os.unlink(self.wav) + except FileNotFoundError: + pass + + # --- return type and shape --- + + def test_returns_tuple_of_three(self): + result = extract_wav_breath_voice(self.wav, use_cache=False) + self.assertIsInstance(result, tuple) + self.assertEqual(len(result), 3) + + def test_time_is_ndarray(self): + time, _, _ = extract_wav_breath_voice(self.wav, use_cache=False) + self.assertIsInstance(time, np.ndarray) + + def test_breath_index_is_ndarray(self): + _, breath, _ = extract_wav_breath_voice(self.wav, use_cache=False) + self.assertIsInstance(breath, np.ndarray) + + def test_voice_index_is_ndarray(self): + _, _, voice = extract_wav_breath_voice(self.wav, use_cache=False) + self.assertIsInstance(voice, np.ndarray) + + def test_all_outputs_same_length(self): + time, breath, voice = extract_wav_breath_voice(self.wav, use_cache=False) + self.assertEqual(len(time), len(breath)) + self.assertEqual(len(breath), len(voice)) + + def test_time_is_1d(self): + time, _, _ = extract_wav_breath_voice(self.wav, use_cache=False) + self.assertEqual(time.ndim, 1) + + def test_breath_is_1d(self): + _, breath, _ = extract_wav_breath_voice(self.wav, use_cache=False) + self.assertEqual(breath.ndim, 1) + + def test_voice_is_1d(self): + _, _, voice = extract_wav_breath_voice(self.wav, use_cache=False) + self.assertEqual(voice.ndim, 1) + + def test_time_starts_at_or_near_zero(self): + time, _, _ = extract_wav_breath_voice(self.wav, use_cache=False) + self.assertGreaterEqual(time[0], 0.0) + + def test_time_is_monotonically_increasing(self): + time, _, _ = extract_wav_breath_voice(self.wav, use_cache=False) + self.assertTrue(np.all(np.diff(time) > 0)) + + def test_time_does_not_exceed_duration(self): + time, _, _ = extract_wav_breath_voice(self.wav, use_cache=False) + self.assertLessEqual(time[-1], 2.1) + + def test_indices_are_finite(self): + _, breath, voice = extract_wav_breath_voice(self.wav, use_cache=False) + self.assertTrue(np.all(np.isfinite(breath))) + self.assertTrue(np.all(np.isfinite(voice))) + + def test_indices_are_in_db_range(self): + """dB values computed with a +1e-9 floor; -180 dB is a reasonable lower bound.""" + _, breath, voice = extract_wav_breath_voice(self.wav, use_cache=False) + self.assertTrue(np.all(breath > -200)) + self.assertTrue(np.all(voice > -200)) + + # --- band parameters --- + + def test_default_bands(self): + """Calling without band overrides must succeed and return consistent shapes.""" + time, breath, voice = extract_wav_breath_voice(self.wav, use_cache=False) + self.assertEqual(len(time), len(breath)) + self.assertEqual(len(time), len(voice)) + + def test_custom_voice_band(self): + """A narrower voice band should still return the same number of frames.""" + time_def, _, _ = extract_wav_breath_voice(self.wav, use_cache=False) + time_cust, _, _ = extract_wav_breath_voice( + self.wav, voice_band=(500, 2000), use_cache=False + ) + self.assertEqual(len(time_def), len(time_cust)) + + def test_different_bands_give_different_voice_values(self): + """Different frequency bands should yield different dB energy values.""" + _, _, voice_low = extract_wav_breath_voice( + self.wav, voice_band=(0, 1000), use_cache=False + ) + _, _, voice_high = extract_wav_breath_voice( + self.wav, voice_band=(4000, 8000), use_cache=False + ) + self.assertFalse(np.allclose(voice_low, voice_high)) + + # --- caching --- + + def test_cache_file_written_when_use_cache_true(self): + tmp_cache_dir = tempfile.mkdtemp() + with patch("utils.wavtool.APP_CACHE_DIR", tmp_cache_dir), \ + patch("utils.wavtool.calculate_file_hash", return_value="testhash"), \ + patch("utils.wavtool.calculate_args_hash", return_value="argshash"): + extract_wav_breath_voice(self.wav, use_cache=True) + + cache_path = os.path.join(tmp_cache_dir, "bv", "testhash.argshash.npz") + self.assertTrue(os.path.exists(cache_path)) + + def test_cache_read_returns_same_data(self): + """A second call with use_cache=True must return the cached arrays unchanged.""" + tmp_cache_dir = tempfile.mkdtemp() + with patch("utils.wavtool.APP_CACHE_DIR", tmp_cache_dir): + t1, b1, v1 = extract_wav_breath_voice(self.wav, use_cache=True) + t2, b2, v2 = extract_wav_breath_voice(self.wav, use_cache=True) + + np.testing.assert_array_equal(t1, t2) + np.testing.assert_array_equal(b1, b2) + np.testing.assert_array_equal(v1, v2) + + def test_cache_disabled_does_not_write(self): + tmp_cache_dir = tempfile.mkdtemp() + with patch("utils.wavtool.APP_CACHE_DIR", tmp_cache_dir), \ + patch("utils.wavtool.calculate_file_hash", return_value="xyz"): + extract_wav_breath_voice(self.wav, use_cache=False) + + bv_dir = os.path.join(tmp_cache_dir, "bv") + files = os.listdir(bv_dir) if os.path.exists(bv_dir) else [] + self.assertEqual(files, []) + + +# --------------------------------------------------------------------------- +# extract_wav_embedding +# --------------------------------------------------------------------------- + +class TestExtractWavEmbedding(unittest.TestCase): + """Tests for extract_wav_embedding. + + EmbedderFactory is patched via its lazy import path + (``utils.embedder.EmbedderFactory``) so the ONNX model is never loaded. + The fake embedder callable returns a (T, D) embedding matrix and a (T,) + frame-time array. + """ + + _D = 768 # embedding dimension (matches mHuBERT-147 hidden size) + _T = 100 # number of frames + + def setUp(self): + self.wav = _make_tonal_wav(duration=2.0, sr=16000, freq=440.0) + self._fake_emb = np.random.randn(self._T, self._D).astype(np.float32) + self._fake_times = np.linspace(0, 2.0, self._T).astype(np.float32) + + def tearDown(self): + try: + os.unlink(self.wav) + except FileNotFoundError: + pass + + def _patches(self, embedder_list=None, fake_emb=None, fake_times=None): + """ExitStack with all patches applied. + + Patches both EmbedderFactory (no ONNX model loads) and emb_features + (so sklearn k-means never runs against an under-sized frame count). + """ + import contextlib + import utils.embedder as emb_mod + + emb = self._fake_emb if fake_emb is None else fake_emb + times = self._fake_times if fake_times is None else fake_times + names = embedder_list or ["mhubert"] + + fake_features = np.random.randn(5, len(times)).astype(np.float32) + + mock_ef = MagicMock() + mock_ef.list.return_value = names + mock_ef.create.return_value = MagicMock(return_value=(emb, times)) + + stack = contextlib.ExitStack() + stack.enter_context(patch.object(emb_mod, "EmbedderFactory", mock_ef)) + stack.enter_context(patch.object(emb_mod, "emb_features", + MagicMock(return_value=fake_features))) + return stack + + def _call(self, use_cache=False, **kwargs): + from utils.wavtool import extract_wav_embedding + with self._patches(): + return extract_wav_embedding( + self.wav, embedder="mhubert", use_cache=use_cache, **kwargs + ) + + # --- return types and shapes --- + + def test_returns_tuple_of_two(self): + result = self._call() + self.assertIsInstance(result, tuple) + self.assertEqual(len(result), 2) + + def test_frame_times_is_ndarray(self): + frame_times, _ = self._call() + self.assertIsInstance(frame_times, np.ndarray) + + def test_features_is_ndarray(self): + _, features = self._call() + self.assertIsInstance(features, np.ndarray) + + def test_features_has_5_rows_by_default(self): + """Without pca_dims, emb_features returns 5 feature rows.""" + _, features = self._call() + self.assertEqual(features.shape[0], 5) + + def test_features_time_axis_matches_frame_times(self): + frame_times, features = self._call() + self.assertEqual(features.shape[1], len(frame_times)) + + def test_frame_times_is_1d(self): + frame_times, _ = self._call() + self.assertEqual(frame_times.ndim, 1) + + def test_features_is_2d(self): + _, features = self._call() + self.assertEqual(features.ndim, 2) + + def test_features_are_float32(self): + _, features = self._call() + self.assertEqual(features.dtype, np.float32) + + def test_frame_times_length_matches_fake(self): + frame_times, _ = self._call() + self.assertEqual(len(frame_times), self._T) + + # --- pca_dims --- + + def test_pca_dims_reduces_rows(self): + _, features = self._call(pca_dims=3) + self.assertEqual(features.shape[0], 3) + + def test_pca_dims_none_returns_5_rows(self): + _, features = self._call(pca_dims=None) + self.assertEqual(features.shape[0], 5) + + def test_pca_dims_time_axis_unchanged(self): + _, feat_full = self._call(pca_dims=None) + _, feat_pca = self._call(pca_dims=3) + self.assertEqual(feat_full.shape[1], feat_pca.shape[1]) + + def test_pca_dims_result_is_float32(self): + _, features = self._call(pca_dims=2) + self.assertEqual(features.dtype, np.float32) + + # --- unknown embedder --- + + def test_unknown_embedder_raises_key_error(self): + from utils.wavtool import extract_wav_embedding + with self._patches(): + with self.assertRaises(KeyError): + extract_wav_embedding(self.wav, embedder="nonexistent", use_cache=False) + + # --- caching --- + + def test_cache_file_written_when_use_cache_true(self): + tmp_cache_dir = tempfile.mkdtemp() + from utils.wavtool import extract_wav_embedding + with self._patches(), \ + patch("utils.wavtool.APP_CACHE_DIR", tmp_cache_dir), \ + patch("utils.wavtool.calculate_file_hash", return_value="testhash"), \ + patch("utils.wavtool.calculate_args_hash", return_value="argshash"): + extract_wav_embedding(self.wav, embedder="mhubert", use_cache=True) + + embedder_dir = os.path.join(tmp_cache_dir, "embedder") + npz_files = [f for f in os.listdir(embedder_dir) if f.endswith(".npz")] + self.assertTrue(len(npz_files) > 0) + + def test_cache_file_name_contains_hash_and_embedder(self): + tmp_cache_dir = tempfile.mkdtemp() + from utils.wavtool import extract_wav_embedding + with self._patches(), \ + patch("utils.wavtool.APP_CACHE_DIR", tmp_cache_dir), \ + patch("utils.wavtool.calculate_file_hash", return_value="filehash"), \ + patch("utils.wavtool.calculate_args_hash", return_value="argshash"): + extract_wav_embedding(self.wav, embedder="mhubert", use_cache=True) + + embedder_dir = os.path.join(tmp_cache_dir, "embedder") + npz_files = os.listdir(embedder_dir) + self.assertEqual(len(npz_files), 1) + self.assertIn("filehash", npz_files[0]) + self.assertIn("mhubert", npz_files[0]) + + def test_cache_read_skips_embedder_call(self): + """On the second call the embedder callable must not be invoked again.""" + tmp_cache_dir = tempfile.mkdtemp() + from utils.wavtool import extract_wav_embedding + import utils.embedder as emb_mod + + mock_ef = MagicMock() + mock_ef.list.return_value = ["mhubert"] + mock_callable = MagicMock(return_value=(self._fake_emb, self._fake_times)) + mock_ef.create.return_value = mock_callable + + fake_features = np.random.randn(5, self._T).astype(np.float32) + with patch.object(emb_mod, "EmbedderFactory", mock_ef), \ + patch.object(emb_mod, "emb_features", MagicMock(return_value=fake_features)), \ + patch("utils.wavtool.APP_CACHE_DIR", tmp_cache_dir): + extract_wav_embedding(self.wav, embedder="mhubert", use_cache=True) + extract_wav_embedding(self.wav, embedder="mhubert", use_cache=True) + + # create() is called once to build the embedder; the embedder itself + # (__call__) should only have been invoked for the first call. + self.assertEqual(mock_callable.call_count, 1) + + def test_cache_read_returns_same_data(self): + """Two calls with use_cache=True must return identical arrays.""" + tmp_cache_dir = tempfile.mkdtemp() + from utils.wavtool import extract_wav_embedding + with self._patches(), \ + patch("utils.wavtool.APP_CACHE_DIR", tmp_cache_dir): + t1, f1 = extract_wav_embedding(self.wav, embedder="mhubert", use_cache=True) + t2, f2 = extract_wav_embedding(self.wav, embedder="mhubert", use_cache=True) + + np.testing.assert_array_equal(t1, t2) + np.testing.assert_array_equal(f1, f2) + + def test_cache_disabled_does_not_write(self): + tmp_cache_dir = tempfile.mkdtemp() + from utils.wavtool import extract_wav_embedding + with self._patches(), \ + patch("utils.wavtool.APP_CACHE_DIR", tmp_cache_dir), \ + patch("utils.wavtool.calculate_file_hash", return_value="xyz"), \ + patch("utils.wavtool.calculate_args_hash", return_value="abc"): + extract_wav_embedding(self.wav, embedder="mhubert", use_cache=False) + + embedder_dir = os.path.join(tmp_cache_dir, "embedder") + files = os.listdir(embedder_dir) if os.path.exists(embedder_dir) else [] + self.assertEqual(files, []) + + if __name__ == "__main__": unittest.main() diff --git a/utils/embedder.py b/utils/embedder.py new file mode 100644 index 0000000..e1fcdc5 --- /dev/null +++ b/utils/embedder.py @@ -0,0 +1,486 @@ +from __future__ import annotations + +import abc +from math import gcd +from pathlib import Path +from typing import ClassVar + +import numpy as np +from scipy.signal import resample_poly +from scipy.ndimage import gaussian_filter1d + +from utils.i18n import _ + + +# --------------------------------------------------------------------------- +# Abstract base +# --------------------------------------------------------------------------- + +class BaseEmbedder(abc.ABC): + """Common interface for all audio embedders. + + Sub-classes must set :attr:`NAME`, :attr:`SAMPLE_RATE`, :attr:`HOP_SIZE`, + implement :meth:`_load_model` and :meth:`__call__`, and optionally + override :meth:`download`. + + The inference backend is intentionally unconstrained — sub-classes may + use ONNX Runtime, PyTorch, TensorFlow, or anything else; only the public + ``__call__`` signature is enforced. + """ + + #: Short identifier used for factory registration and the download CLI. + #: Must be a non-empty string in every concrete sub-class. + NAME: ClassVar[str] = "" + + #: SPDX license identifier for the model weights, e.g. ``"MIT"``, + #: ``"Apache-2.0"``, ``"CC-BY-4.0"``. Empty string means unspecified. + LICENSE: ClassVar[str] = "" + + #: Target sample rate in Hz. + SAMPLE_RATE: ClassVar[int] = 16_000 + + #: Hop size in samples between successive embedding frames. + HOP_SIZE: ClassVar[int] = 320 + + def __init__(self, model_path: str | Path, device: str = "cpu") -> None: + self._model_path = Path(model_path) + self._device = device + self._load_model(self._model_path, device) + + @abc.abstractmethod + def _load_model(self, model_path: Path, device: str) -> None: + """Initialise the inference backend from *model_path*. + + Called once from ``__init__``. Load whatever framework the + sub-class uses (ONNX Runtime, PyTorch, …) here. + """ + + @abc.abstractmethod + def __call__( + self, wav_path: str | Path + ) -> tuple[np.ndarray, np.ndarray]: + """Extract embeddings from an audio file. + + Args: + wav_path: Path to a WAV (or any soundfile-readable) file. + + Returns: + embeddings: ``(T, D)`` float32 array. + frame_times: ``(T,)`` array of frame centre times in seconds. + """ + + @classmethod + def download( + cls, + cache_dir: str | Path | None = None, + **kwargs, + ) -> Path: + """Download the model weights and return the local path. + + The base implementation raises :exc:`NotImplementedError`. + Sub-classes that fetch weights from a remote source should override + this. Extra *kwargs* allow variant/precision selection without + breaking the shared CLI signature. + + Args: + cache_dir: Where to store the downloaded file. + + Returns: + Local path to the ready-to-use model file. + """ + raise NotImplementedError( + f"{cls.__name__} does not implement download()." + ) + + # ------------------------------------------------------------------ + # Shared audio helpers + # ------------------------------------------------------------------ + + def _load_audio(self, wav_path: str | Path) -> np.ndarray: + """Read *wav_path*, convert to mono float32, resample to SAMPLE_RATE.""" + import soundfile as sf + + audio, sr = sf.read(str(wav_path), always_2d=False) + + if audio.ndim == 2: + audio = audio.mean(axis=1) + + if sr != self.SAMPLE_RATE: + g = gcd(sr, self.SAMPLE_RATE) + audio = resample_poly(audio, self.SAMPLE_RATE // g, sr // g) + + return audio.astype(np.float32) + + def _frame_times(self, n_frames: int) -> np.ndarray: + """Return centre times (seconds) for *n_frames* frames.""" + return (np.arange(n_frames) + 0.5) * (self.HOP_SIZE / self.SAMPLE_RATE) + + +# --------------------------------------------------------------------------- +# Factory / registry +# --------------------------------------------------------------------------- + +class EmbedderFactory: + """Registry of :class:`BaseEmbedder` sub-classes, keyed by :attr:`~BaseEmbedder.NAME`. + + Sub-classes are registered by passing them to :meth:`register`, which + doubles as a plain class decorator:: + + @EmbedderFactory.register + class MyEmbedder(BaseEmbedder): + NAME = "my_model" + ... + + # or after the fact: + EmbedderFactory.register(MyEmbedder) + + :attr:`~BaseEmbedder.NAME` is the only thing that needs to be set — + no separate string argument required. + """ + + _registry: ClassVar[dict[str, type[BaseEmbedder]]] = {} + _instance_cache: ClassVar[dict[tuple, BaseEmbedder]] = {} + + @classmethod + def register(cls, klass: type[BaseEmbedder]) -> type[BaseEmbedder]: + """Add *klass* to the registry using ``klass.NAME``. + + Raises: + TypeError: If *klass* does not subclass :class:`BaseEmbedder`. + ValueError: If :attr:`~BaseEmbedder.NAME` is empty. + """ + if not issubclass(klass, BaseEmbedder): + raise TypeError(f"{klass} must subclass BaseEmbedder") + if not klass.NAME: + raise ValueError(f"{klass.__name__}.NAME must be a non-empty string") + cls._registry[klass.NAME] = klass + return klass + + @classmethod + def create(cls, name: str, **kwargs) -> BaseEmbedder: + """Return a cached :class:`BaseEmbedder` instance, creating it on first use. + + Instances are keyed by ``(name, *sorted_kwargs)`` so that different + configurations each own a separate ONNX/CUDA session while the same + configuration reuses the existing one — keeping the CUDA execution + provider context alive between calls. + + All *kwargs* are forwarded to the embedder's ``__init__`` on first + construction only. + + Raises: + KeyError: If *name* is not in the registry. + """ + if name not in cls._registry: + raise KeyError( + f"No embedder registered as {name!r}. " + f"Available: {list(cls._registry)}" + ) + key = (name, tuple(sorted(kwargs.items()))) + if key not in cls._instance_cache: + cls._instance_cache[key] = cls._registry[name](**kwargs) + return cls._instance_cache[key] + + @classmethod + def clear_cache(cls) -> None: + """Release all cached embedder instances and free their resources. + + Call this to explicitly drop CUDA/ONNX sessions and reclaim GPU + memory (e.g. before loading a different model or on shutdown). + """ + cls._instance_cache.clear() + + @classmethod + def list(cls) -> list[str]: + """Return the names of all registered embedders.""" + return sorted(cls._registry) + + +# --------------------------------------------------------------------------- +# mHuBERT-147 (ONNX Runtime backend) +# --------------------------------------------------------------------------- + +@EmbedderFactory.register +class mHuBERTEmbedder(BaseEmbedder): + """mHuBERT-147 via ONNX Runtime, downloaded on demand from HuggingFace.""" + + NAME: ClassVar[str] = "mhubert" + LICENSE: ClassVar[str] = "CC-BY-NC-SA-4.0" + SAMPLE_RATE: ClassVar[int] = 16_000 + HOP_SIZE: ClassVar[int] = 320 # ~50 FPS + + REPO_ID: ClassVar[str] = "NewComer00/mHuBERT-147-ONNX" + + MODEL_FILES: ClassVar[dict[str, str]] = { + "fp32": "model.onnx", + "fp16": "model_fp16.onnx", + "q4": "model_q4.onnx", + "q4f16": "model_q4f16.onnx", + "bnb4": "model_bnb4.onnx", + } + + def __init__( + self, + model_path: str | Path | None = None, + variant: str = "q4f16", + device: str = "cpu", + ) -> None: + if model_path is None: + model_path = self.download(variant=variant) + super().__init__(model_path, device=device) + + def _load_model(self, model_path: Path, device: str) -> None: + import onnxruntime as ort + + providers = ( + ["CUDAExecutionProvider", "CPUExecutionProvider"] + if device == "cuda" + else ["CPUExecutionProvider"] + ) + self._session = ort.InferenceSession(str(model_path), providers=providers) + self._input_name = self._session.get_inputs()[0].name + self._output_name = self._session.get_outputs()[0].name + + print(_("Loaded {name} model from '{path}'.").format(name=self.NAME, path=model_path)) + if "CUDAExecutionProvider" in self._session.get_providers(): + print(_("ONNX Runtime is using CUDA for inference acceleration.")) + + def __call__( + self, wav_path: str | Path + ) -> tuple[np.ndarray, np.ndarray]: + audio = self._load_audio(wav_path) + outputs = self._session.run( + [self._output_name], + {self._input_name: audio[np.newaxis, :]}, + ) + embeddings = outputs[0].squeeze(0) # (T, D) + if embeddings.ndim != 2: + raise ValueError( + f"Expected embeddings of shape (T, D), got {embeddings.shape}" + ) + return embeddings, self._frame_times(len(embeddings)) + + @classmethod + def download( # type: ignore[override] + cls, + variant: str = "q4f16", + cache_dir: str | Path | None = None, + ) -> Path: + """Download the requested ONNX variant from HuggingFace. + + Args: + variant: One of :attr:`MODEL_FILES` keys. + cache_dir: Override HF Hub cache directory. + + Returns: + Local path to the downloaded ``.onnx`` file. + """ + if variant not in cls.MODEL_FILES: + raise ValueError( + f"Unknown variant {variant!r}. " + f"Choose from: {list(cls.MODEL_FILES)}" + ) + + print(_("Fetching {name} ({variant}) from Internet or local cache...").format(name=cls.NAME, variant=variant)) + from huggingface_hub import hf_hub_download + return hf_hub_download( + repo_id=cls.REPO_ID, + filename=f"onnx/{cls.MODEL_FILES[variant]}", + cache_dir=cache_dir, + ) + + +# --------------------------------------------------------------------------- +# Embedding feature extractors +# --------------------------------------------------------------------------- + +def emb_boundary(emb: np.ndarray, smooth_sigma: float = 1.0) -> np.ndarray: + """Phoneme boundary novelty curve (L2 distance between consecutive frames). + + Peaks sharply at phoneme/syllable transitions regardless of pitch or + timbre, making it a strong DTW anchor for note boundary alignment. + + Args: + emb: Frame embeddings, shape ``(T, D)``. + smooth_sigma: Gaussian smoothing sigma. Larger values widen peaks. + + Returns: + Boundary novelty curve, shape ``(T,)``. Non-negative; peaks ≈ boundaries. + """ + diff = np.linalg.norm(np.diff(emb, axis=0), axis=1) + diff = np.append(diff, diff[-1]) + + if smooth_sigma > 0: + diff = gaussian_filter1d(diff, sigma=smooth_sigma) + + return diff.astype(np.float32) + + +def emb_frame_entropy( + emb: np.ndarray, + n_clusters: int = 128, + temperature: float = 1.0, + smooth_sigma: float = 1.0, +) -> np.ndarray: + """Per-frame entropy over soft k-means cluster assignments. + + Low entropy → confident phoneme centre (sustained vowel core). + High entropy → ambiguous region, typically at phoneme transitions. + + Complements :func:`emb_boundary`: boundary peaks *at* the change; + entropy stays elevated *through* the transition region. + + Args: + emb: Frame embeddings, shape ``(T, D)``. + n_clusters: Number of pseudo-phoneme clusters for k-means. + temperature: Softmax temperature applied to cosine distances. + smooth_sigma: Gaussian smoothing sigma. + + Returns: + Per-frame entropy values, shape ``(T,)``. + """ + from sklearn.cluster import MiniBatchKMeans + + norms = np.linalg.norm(emb, axis=1, keepdims=True) + emb_norm = emb / np.maximum(norms, 1e-8) + + km = MiniBatchKMeans(n_clusters=n_clusters, n_init=3, random_state=0) + km.fit(emb_norm) + + centroids = km.cluster_centers_ + c_norms = np.linalg.norm(centroids, axis=1, keepdims=True) + centroids = centroids / np.maximum(c_norms, 1e-8) + + sim = emb_norm @ centroids.T + logits = sim / temperature + logits -= logits.max(axis=1, keepdims=True) + probs = np.exp(logits) + probs /= probs.sum(axis=1, keepdims=True) + + entropy = -np.sum(probs * np.log(probs + 1e-12), axis=1) + + if smooth_sigma > 0: + entropy = gaussian_filter1d(entropy, sigma=smooth_sigma) + + return entropy.astype(np.float32) + + +def emb_self_similarity( + emb: np.ndarray, + window: int = 20, + smooth_sigma: float = 1.0, +) -> np.ndarray: + """Local self-similarity novelty curve. + + Returns the row-wise standard deviation of a local cosine-similarity + matrix — high where the frame is changing rapidly, low in stable + phonetic regions. + + Args: + emb: Frame embeddings, shape ``(T, D)``. + window: Half-width of the comparison window in frames. + At 50 FPS, ``window=20`` covers ±0.4 s. + smooth_sigma: Gaussian smoothing sigma. + + Returns: + Novelty curve, shape ``(T,)``. + """ + norms = np.linalg.norm(emb, axis=1, keepdims=True) + emb_norm = emb / np.maximum(norms, 1e-8) + + T = len(emb_norm) + W = window + + # Pad both ends so every frame has a full 2W+1 neighbourhood + padded = np.pad(emb_norm, ((W, W), (0, 0)), mode="edge") # (T+2W, D) + # Build (T, 2W+1, D) sliding window view without copying data + shape = (T, 2 * W + 1, emb_norm.shape[1]) + strides = (padded.strides[0], padded.strides[0], padded.strides[1]) + windows = np.lib.stride_tricks.as_strided(padded, shape=shape, strides=strides) + # Cosine similarity of each frame against its neighbourhood: (T, 2W+1) + sim = np.einsum("twd,td->tw", windows, emb_norm) + novelty = sim.std(axis=1) + + if smooth_sigma > 0: + novelty = gaussian_filter1d(novelty, sigma=smooth_sigma) + + return novelty.astype(np.float32) + + +def emb_velocity_acceleration( + emb: np.ndarray, + smooth_sigma: float = 1.0, +) -> tuple[np.ndarray, np.ndarray]: + """Per-frame velocity and acceleration in embedding space. + + Analogous to delta/delta-delta MFCCs but in the mHuBERT space. + Velocity peaks at the *onset* of a transition; acceleration peaks + slightly earlier, giving DTW an early-warning signal. + + Args: + emb: Frame embeddings, shape ``(T, D)``. + smooth_sigma: Pre-smoothing sigma before differentiation. + + Returns: + ``(velocity, acceleration)`` — each shape ``(T,)``. + """ + emb_smooth = ( + gaussian_filter1d(emb, sigma=smooth_sigma, axis=0) + if smooth_sigma > 0 + else emb + ) + + grad1 = np.gradient(emb_smooth, axis=0) + grad2 = np.gradient(grad1, axis=0) + + velocity = np.linalg.norm(grad1, axis=1).astype(np.float32) + acceleration = np.linalg.norm(grad2, axis=1).astype(np.float32) + return velocity, acceleration + + +def emb_features( + emb: np.ndarray, + boundary_sigma: float = 1.0, + entropy_clusters: int = 128, + entropy_temperature: float = 1.0, + entropy_sigma: float = 1.0, + similarity_window: int = 20, + similarity_sigma: float = 1.0, + velocity_sigma: float = 1.0, +) -> np.ndarray: + """Stack all DTW-oriented features into a single matrix. + + Row order: + + ========= ================================================= + Row 0 boundary novelty (sharp spikes at transitions) + Row 1 frame entropy (elevated through transition regions) + Row 2 self-similarity (stable in sustained phonemes) + Row 3 velocity (magnitude of embedding change) + Row 4 acceleration (rate of change of velocity) + ========= ================================================= + + Args: + emb: Frame embeddings, shape ``(T, D)``. + boundary_sigma: Smoothing for boundary curve. + entropy_clusters: K-means clusters for entropy. + entropy_temperature: Softmax temperature for entropy. + entropy_sigma: Smoothing for entropy curve. + similarity_window: Half-window for self-similarity. + similarity_sigma: Smoothing for self-similarity. + velocity_sigma: Pre-smoothing before differentiation. + + Returns: + Stacked feature matrix, shape ``(5, T)``. + """ + boundary = emb_boundary(emb, smooth_sigma=boundary_sigma) + entropy = emb_frame_entropy( + emb, + n_clusters=entropy_clusters, + temperature=entropy_temperature, + smooth_sigma=entropy_sigma, + ) + similarity = emb_self_similarity(emb, window=similarity_window, smooth_sigma=similarity_sigma) + velocity, acceleration = emb_velocity_acceleration(emb, smooth_sigma=velocity_sigma) + + return np.vstack([boundary, entropy, similarity, velocity, acceleration]) # (5, T) diff --git a/utils/fs.py b/utils/fs.py index af4fbcc..838ff50 100644 --- a/utils/fs.py +++ b/utils/fs.py @@ -24,13 +24,27 @@ def calculate_file_hash(file_path): file_path (str): Path to the file. Returns: - str: SHA-256 hash of the file contents. + str: 16-character SHA-256 hash of the file contents. """ hash_sha256 = hashlib.sha256() with open(file_path, "rb") as f: while chunk := f.read(8192): hash_sha256.update(chunk) - return hash_sha256.hexdigest() + return hash_sha256.hexdigest()[:16] + + +def calculate_args_hash(*args, **kwargs): + """Calculate a short hash of the given arguments for use in cache keys. + + Args: + *args: Positional values to hash. + **kwargs: Keyword values to hash. + + Returns: + str: 16-character hash of the combined arguments. + """ + import joblib + return joblib.hash((args, sorted(kwargs.items())))[:16] def clear_cache(): diff --git a/utils/wavtool.py b/utils/wavtool.py index b778cf7..b80c301 100644 --- a/utils/wavtool.py +++ b/utils/wavtool.py @@ -6,11 +6,140 @@ import tempfile from pathlib import Path +import scipy import librosa import numpy as np from utils.i18n import _ -from utils.fs import APP_CACHE_DIR, calculate_file_hash +from utils.fs import APP_CACHE_DIR, calculate_args_hash, calculate_file_hash + + +def extract_wav_embedding( + wav_path, + embedder: str = "mhubert", + pca_dims: int | None = None, + use_cache=True, + **embedder_kwargs, +): + """Extract frame-level embedding features from a WAV file. + + Args: + wav_path: Path to the input WAV file. + embedder: Registered embedder name — one of + :meth:`EmbedderFactory.list`. + pca_dims: If set, reduce the 5 feature rows to *pca_dims* + principal components using + :class:`sklearn.decomposition.PCA`. + ``None`` returns all 5 rows unchanged. + use_cache: Load from / save to a per-file ``.npz`` cache keyed + by file hash and embedder name. + **embedder_kwargs: Forwarded to the embedder constructor (e.g. + ``variant``, ``device``). + + Returns: + tuple: ``(frame_times, features)`` where + + - ``frame_times``: ``(T,)`` float32 array, seconds. + - ``features``: ``(5, T)`` or ``(pca_dims, T)`` float32 array. + + Raises: + KeyError: If *embedder* is not in :meth:`EmbedderFactory.list`. + """ + from utils.embedder import EmbedderFactory, emb_features + + if embedder not in EmbedderFactory.list(): + raise KeyError( + f"Unknown embedder {embedder!r}. " + f"Available: {EmbedderFactory.list()}" + ) + + cache_dir = Path(APP_CACHE_DIR) / "embedder" + suffix = f"{embedder}.{calculate_args_hash(**embedder_kwargs)}" + if pca_dims is not None: + suffix += f".pca{pca_dims}" + + if use_cache: + os.makedirs(cache_dir, exist_ok=True) + wav_hash = calculate_file_hash(wav_path) + cache_path = cache_dir / f"{wav_hash}.{suffix}.npz" + + if cache_path.is_file(): + print(f"[{embedder}] " + _("Loading embedding data from cache file: '{}'").format(cache_path)) + data = np.load(cache_path) + return np.asarray(data["frame_times"]), np.asarray(data["embeddings"]) + + emb, frame_times = EmbedderFactory.create(embedder, **embedder_kwargs)(wav_path) + + norms = np.linalg.norm(emb, axis=1, keepdims=True) + emb_norm = emb / np.maximum(norms, 1e-8) # (T, D) + features = emb_features(emb_norm) # (5, T) + + if pca_dims is not None: + from sklearn.decomposition import PCA + features = PCA(n_components=pca_dims).fit_transform(features.T).T.astype(np.float32) # (pca_dims, T) + + if use_cache: + np.savez(cache_path, frame_times=frame_times, embeddings=features) + print(f"[{embedder}] " + _("Embedding data saved to cache file: '{}'").format(cache_path)) + + return np.asarray(frame_times), np.asarray(features) + + +def extract_wav_breath_voice(wav_path, breath_band=(10000, np.inf), voice_band=(0, 4000), use_cache=True): + """Extract breath and voice intensity indices from a WAV file. + + Separates harmonic content via HPSS, then computes per-frame RMS energy + in the specified frequency bands as dB indices. + + Args: + wav_path (str): Path to the WAV file. + breath_band (tuple, optional): Frequency range (Hz) for breath detection. + Defaults to (10000, inf). + voice_band (tuple, optional): Frequency range (Hz) for voice detection. + Defaults to (0, 4000). + use_cache (bool, optional): Whether to use cached data if available. + Defaults to True. + + Returns: + tuple: (time, breath_index, voice_index), where: + - time (np.ndarray of float): Time points in seconds. Shape: (n_frames,). + - breath_index (np.ndarray of float): Breath intensity in dB. Shape: (n_frames,). + - voice_index (np.ndarray of float): Voice intensity in dB. Shape: (n_frames,). + """ + cache_dir = Path(APP_CACHE_DIR) / "bv" + + if use_cache: + os.makedirs(cache_dir, exist_ok=True) + wav_hash = calculate_file_hash(wav_path) + bands_hash = calculate_args_hash(breath_band, voice_band) + cache_path = cache_dir / f"{wav_hash}.{bands_hash}.npz" + if cache_path.is_file(): + print(_("Loading breath/voice data from cache file: '{}'").format(cache_path)) + data = np.load(cache_path) + return data["time"], data["breath_index"], data["voice_index"] + + y, sr = librosa.load(wav_path, sr=None, mono=True) + y_h, y_n = librosa.effects.hpss(y, margin=(1.0, 5.0)) + + D = librosa.stft(y_h, n_fft=2048, hop_length=512) + freqs = librosa.fft_frequencies(sr=sr, n_fft=2048) + + def band_rms(mask): + return np.sqrt(np.mean(np.abs(D[mask, :]) ** 2, axis=0)) + + voice_mask = (freqs >= voice_band[0]) & (freqs < voice_band[1]) + voice_index = 20 * np.log10(band_rms(voice_mask) + 1e-9) + + breath_mask = (freqs >= breath_band[0]) & (freqs < breath_band[1]) + breath_index = 20 * np.log10(band_rms(breath_mask) + 1e-9) + + time = librosa.frames_to_time(np.arange(D.shape[1]), sr=sr, hop_length=512) + + if use_cache: + np.savez(cache_path, time=time, breath_index=breath_index, voice_index=voice_index) + print(_("Breath/voice data saved to cache file: '{}'").format(cache_path)) + + return time, breath_index, voice_index def extract_wav_mfcc(wav_path, n_feat=6, n_mfcc=13): @@ -77,7 +206,7 @@ def extract_wav_frequency(file_path, backend="rmvpe-onnx", use_cache=True): time = [] frequency = [] confidence = [] - cache_dir = Path(APP_CACHE_DIR) / "pitd" + cache_dir = Path(APP_CACHE_DIR) / "f0" # Try reading data from cache if use_cache: os.makedirs(cache_dir, exist_ok=True) @@ -99,10 +228,7 @@ def extract_wav_frequency(file_path, backend="rmvpe-onnx", use_cache=True): # Extract pitch using the specified backend if backend == "crepe": import crepe - from utils.gpu import add_cuda_to_path - from scipy.io import wavfile - add_cuda_to_path(skip_missing=True) - sr, audio = wavfile.read(file_path) + sr, audio = scipy.io.wavfile.read(file_path) time, frequency, confidence, _unused = crepe.predict(audio, sr, viterbi=True) elif backend == "swift-f0": from swift_f0 import SwiftF0 @@ -115,7 +241,7 @@ def extract_wav_frequency(file_path, backend="rmvpe-onnx", use_cache=True): from rmvpe_onnx import RMVPE import soundfile as sf audio, sr = sf.read(file_path) - rmvpe = RMVPE() + rmvpe = RMVPE(device="cpu") timestamp, frequency, confidence, _unused = rmvpe.predict(audio=audio, sr=sr) time = timestamp.tolist() frequency = frequency.tolist()