Research zipvoice
#1
by inoryQwQ - opened
- .gitattributes +12 -23
- .gitignore +0 -24
- README.md +14 -76
- bin/zipvoice_ax630c +0 -3
- bin/zipvoice_ax650 +0 -3
- bin/zipvoice_axcl +0 -3
- configuration.json +0 -1
- infer_zipvoice_axera.py +8 -18
- models/vocoder/vocos_full.axmodel +0 -3
- models/vocoder/vocos_full_ax630c.axmodel +0 -3
- models/zipvoice_distill_ax630C/decoder4_split_manifest.json +66 -102
- models/zipvoice_distill_ax630C/decoder_part0.axmodel +2 -2
- models/zipvoice_distill_ax630C/decoder_part1.axmodel +2 -2
- models/zipvoice_distill_ax630C/decoder_part2.axmodel +2 -2
- models/zipvoice_distill_ax630C/decoder_part3_0.axmodel +0 -3
- models/zipvoice_distill_ax630C/decoder_part3_1.axmodel +0 -3
- models/zipvoice_distill_ax630C/decoder_part3_2.axmodel +0 -3
- models/zipvoice_distill_ax630C/decoder_part3_3.axmodel +0 -3
- models/zipvoice_distill_ax630C/encoder.axmodel +2 -2
- run_ax630c.sh +0 -51
- run_ax650.sh +0 -57
- run_axcl.sh +0 -57
- scripts/__pycache__/__init__.cpython-310.pyc +0 -0
- scripts/__pycache__/local_tokenizer.cpython-310.pyc +0 -0
- scripts/__pycache__/text_processing.cpython-310.pyc +0 -0
- scripts/common_infer.py +0 -71
- scripts/zipvoice_decoder4_runtime.py +0 -2
- scripts/zipvoice_decoder4_runtime_encoder_onnx.py +0 -87
- scripts/zipvoice_decoder4_runtime_part0_onnx.py +0 -94
- scripts/zipvoice_decoder4_runtime_part3_onnx.py +0 -93
.gitattributes
CHANGED
|
@@ -35,16 +35,6 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
assets/moss_prompts/en_4_4p5s.wav filter=lfs diff=lfs merge=lfs -text
|
| 37 |
assets/moss_prompts/zh_1_4p5s.wav filter=lfs diff=lfs merge=lfs -text
|
| 38 |
-
cpp/install/ax650/zipvoice filter=lfs diff=lfs merge=lfs -text
|
| 39 |
-
cpp/install/ax650/zipvoice_axera filter=lfs diff=lfs merge=lfs -text
|
| 40 |
-
cpp/install/axcl/zipvoice filter=lfs diff=lfs merge=lfs -text
|
| 41 |
-
cpp/vocoder/vocos_full.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 42 |
-
cpp/vocoder/vocos_full_ax630c.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 43 |
-
models/axmodels/decoder_part0.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 44 |
-
models/axmodels/decoder_part1.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 45 |
-
models/axmodels/decoder_part2.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 46 |
-
models/axmodels/decoder_part3.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 47 |
-
models/axmodels/encoder.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 48 |
models/zipvoice_ax650/decoder_part0.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 49 |
models/zipvoice_ax650/decoder_part1.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 50 |
models/zipvoice_ax650/decoder_part2.axmodel filter=lfs diff=lfs merge=lfs -text
|
|
@@ -54,26 +44,25 @@ models/zipvoice_distill_ax630C/decoder_part0.axmodel filter=lfs diff=lfs merge=l
|
|
| 54 |
models/zipvoice_distill_ax630C/decoder_part1.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 55 |
models/zipvoice_distill_ax630C/decoder_part2.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 56 |
models/zipvoice_distill_ax630C/decoder_part3.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 57 |
-
models/zipvoice_distill_ax630C/decoder_part3_0.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 58 |
-
models/zipvoice_distill_ax630C/decoder_part3_1.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 59 |
-
models/zipvoice_distill_ax630C/decoder_part3_2.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 60 |
-
models/zipvoice_distill_ax630C/decoder_part3_3.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 61 |
models/zipvoice_distill_ax630C/encoder.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 62 |
models/zipvoice_distill_ax650/decoder_part0.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 63 |
models/zipvoice_distill_ax650/decoder_part1.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 64 |
models/zipvoice_distill_ax650/decoder_part2.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 65 |
models/zipvoice_distill_ax650/decoder_part3.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 66 |
models/zipvoice_distill_ax650/encoder.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 67 |
-
outputs/
|
| 68 |
outputs/en_long_paragraph_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
| 69 |
-
outputs/
|
| 70 |
outputs/en_sentence_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
| 71 |
-
outputs/
|
| 72 |
outputs/zh_long_paragraph_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
| 73 |
-
outputs/
|
| 74 |
outputs/zh_sentence_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
| 75 |
-
|
| 76 |
-
|
| 77 |
-
|
| 78 |
-
|
| 79 |
-
|
|
|
|
|
|
|
|
|
|
|
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
assets/moss_prompts/en_4_4p5s.wav filter=lfs diff=lfs merge=lfs -text
|
| 37 |
assets/moss_prompts/zh_1_4p5s.wav filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 38 |
models/zipvoice_ax650/decoder_part0.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 39 |
models/zipvoice_ax650/decoder_part1.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 40 |
models/zipvoice_ax650/decoder_part2.axmodel filter=lfs diff=lfs merge=lfs -text
|
|
|
|
| 44 |
models/zipvoice_distill_ax630C/decoder_part1.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 45 |
models/zipvoice_distill_ax630C/decoder_part2.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 46 |
models/zipvoice_distill_ax630C/decoder_part3.axmodel filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
| 47 |
models/zipvoice_distill_ax630C/encoder.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 48 |
models/zipvoice_distill_ax650/decoder_part0.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 49 |
models/zipvoice_distill_ax650/decoder_part1.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 50 |
models/zipvoice_distill_ax650/decoder_part2.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 51 |
models/zipvoice_distill_ax650/decoder_part3.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 52 |
models/zipvoice_distill_ax650/encoder.axmodel filter=lfs diff=lfs merge=lfs -text
|
| 53 |
+
outputs/en_long_paragraph.wav filter=lfs diff=lfs merge=lfs -text
|
| 54 |
outputs/en_long_paragraph_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
| 55 |
+
outputs/en_sentence.wav filter=lfs diff=lfs merge=lfs -text
|
| 56 |
outputs/en_sentence_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
| 57 |
+
outputs/zh_long_paragraph.wav filter=lfs diff=lfs merge=lfs -text
|
| 58 |
outputs/zh_long_paragraph_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
| 59 |
+
outputs/zh_sentence.wav filter=lfs diff=lfs merge=lfs -text
|
| 60 |
outputs/zh_sentence_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
| 61 |
+
en_long_paragraph_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
| 62 |
+
en_long_paragraph_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
| 63 |
+
en_sentence_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
| 64 |
+
en_sentence_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
| 65 |
+
zh_long_paragraph_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
| 66 |
+
zh_long_paragraph_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
| 67 |
+
zh_sentence_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
| 68 |
+
zh_sentence_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
|
.gitignore
DELETED
|
@@ -1,24 +0,0 @@
|
|
| 1 |
-
# HuggingFace CLI 缓存
|
| 2 |
-
.msc/
|
| 3 |
-
.mv/
|
| 4 |
-
._____temp/
|
| 5 |
-
.cache/
|
| 6 |
-
|
| 7 |
-
# 运行产物
|
| 8 |
-
outputs/
|
| 9 |
-
reference/
|
| 10 |
-
__pycache__/
|
| 11 |
-
*.pyc
|
| 12 |
-
|
| 13 |
-
# 编译产物/源码(HUG 仓只放可执行文件,不放源码)
|
| 14 |
-
cpp/src/
|
| 15 |
-
cpp/third_party/
|
| 16 |
-
cpp/toolchains/
|
| 17 |
-
cpp/cmake/
|
| 18 |
-
cpp/scripts/
|
| 19 |
-
cpp/CMakeLists.txt
|
| 20 |
-
cpp/build_*.sh
|
| 21 |
-
cpp/download_bsp.sh
|
| 22 |
-
cpp/*.cpp
|
| 23 |
-
cpp/*.hpp
|
| 24 |
-
build/
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
README.md
CHANGED
|
@@ -1,39 +1,18 @@
|
|
| 1 |
# ZipVoice.AXERA
|
| 2 |
|
| 3 |
-
[ZipVoice]((https://github.com/k2-fsa/ZipVoice)) AXERA 板端
|
| 4 |
|
| 5 |
## 功能
|
| 6 |
|
| 7 |
- 支持中文和英文语音生成。
|
| 8 |
- 支持语音克隆。
|
| 9 |
- 支持 ZipVoice、ZipVoice Distill
|
| 10 |
-
- 支持 C++ 和 Python 推理
|
| 11 |
-
- 支持 AX650、AX630C
|
| 12 |
|
| 13 |
## 模型说明
|
| 14 |
|
| 15 |
-
ZipVoice Distill 是 ZipVoice 的蒸馏版本,主要优势是在较小性能损失下提升推理速度。
|
| 16 |
-
|
| 17 |
-
## 性能对比
|
| 18 |
-
|
| 19 |
-
### Distill 模型 (num_step=4)
|
| 20 |
-
|
| 21 |
-
| 场景 | 音频时长 | Python AX650<br>耗时 RTF | C++ AX650<br>耗时 RTF | C++ AXCL<br>耗时 RTF | Python AX630C<br>耗时 RTF | C++ AX630C<br>耗时 RTF |
|
| 22 |
-
|------|---------|-------------------------|----------------------|---------------------|--------------------------|-----------------------|
|
| 23 |
-
| 中文句子 | 6.41s | 1.992s/0.311 | 1.057s/0.165 | 0.918s/**0.143** | 10.296s/1.606 | 4.650s/0.725 |
|
| 24 |
-
| 中文段落 | 44.97s | 13.457s/0.301 | 7.357s/0.164 | 6.443s/**0.143** | 71.574s/1.600 | 32.590s/0.725 |
|
| 25 |
-
| 英文句子 | 6.41s | 2.045s/0.319 | 1.067s/0.166 | 0.919s/**0.143** | 10.686s/1.667 | 4.654s/0.726 |
|
| 26 |
-
| 英文段落 | 59.16s | 19.715s/0.305 | 10.529s/0.178 | 9.207s/**0.156** | 106.183s/1.640 | 46.541s/0.787 |
|
| 27 |
-
|
| 28 |
-
### 普通模型 (num_step=10)
|
| 29 |
-
|
| 30 |
-
| 场景 | 音频时长 | Python AX650 (耗时/RTF) | C++ AX650 (耗时/RTF) | C++ AXCL (耗时/RTF) |
|
| 31 |
-
|------|---------|------------------------|---------------------|---------------------|
|
| 32 |
-
| 中文句子 | 6.41s | 5.781s / 0.902 | 5.709s / 0.891 | 5.452s / **0.850** |
|
| 33 |
-
| 中文段落 | 44.97s | 40.292s / 0.901 | 39.714s / 0.883 | 38.138s / **0.848** |
|
| 34 |
-
| 英文句子 | 6.41s | 5.711s / 0.891 | 5.693s / 0.888 | 5.447s / **0.850** |
|
| 35 |
-
| 英文段落 | 59.16s | 62.161s / 0.960 | 56.759s / 0.957 | 54.451s / **0.920** |
|
| 36 |
|
|
|
|
| 37 |
|
| 38 |
## 模型转换
|
| 39 |
|
|
@@ -49,8 +28,6 @@ ZipVoice Distill 是 ZipVoice 的蒸馏版本,主要优势是在较小性能
|
|
| 49 |
- [M4N-Dock(爱芯派Pro)](https://wiki.sipeed.com/hardware/zh/maixIV/m4ndock/m4ndock.html)
|
| 50 |
- [M.2 Accelerator Card](https://docs.m5stack.com/zh_CN/ai_hardware/LLM-8850_Card)
|
| 51 |
|
| 52 |
-
- AX630C
|
| 53 |
-
|
| 54 |
|
| 55 |
## 目录结构
|
| 56 |
|
|
@@ -62,30 +39,20 @@ ZipVoice.AXERA/
|
|
| 62 |
├── models/
|
| 63 |
│ ├── zipvoice_ax650/
|
| 64 |
│ ├── zipvoice_distill_ax650/
|
| 65 |
-
│
|
| 66 |
-
│ └── vocoder/ # vocos_full.axmodel(AX650/AX630C)
|
| 67 |
├── resources/
|
| 68 |
-
│ ├── vocos-mel-24khz/
|
| 69 |
-
│ └── zipvoice_hf/
|
| 70 |
-
├── bin/ # C++ 推理可执行文件(源码见 GitHub)
|
| 71 |
-
│ ├── zipvoice_ax650
|
| 72 |
-
│ ├── zipvoice_ax630c
|
| 73 |
-
│ └── zipvoice_axcl
|
| 74 |
├── scripts/
|
| 75 |
-
├── infer_zipvoice_axera.py
|
| 76 |
-
├── run_ax650.sh # AX650 板端
|
| 77 |
-
├── run_ax630c.sh # AX630C 板端
|
| 78 |
-
├── run_axcl.sh # AXCL 算力卡
|
| 79 |
├── requirements.txt
|
| 80 |
└── README.md
|
| 81 |
```
|
| 82 |
|
| 83 |
-
##
|
| 84 |
-
|
| 85 |
-
### 环境
|
| 86 |
|
| 87 |
安装 pyaxengine:
|
| 88 |
-
|
| 89 |
```bash
|
| 90 |
pip3 install axengine-x.x.x-py3-none-any.whl
|
| 91 |
```
|
|
@@ -98,7 +65,7 @@ conda activate ZipVoice
|
|
| 98 |
pip3 install -r requirements.txt
|
| 99 |
```
|
| 100 |
|
| 101 |
-
##
|
| 102 |
|
| 103 |
进入目录:
|
| 104 |
|
|
@@ -106,7 +73,7 @@ pip3 install -r requirements.txt
|
|
| 106 |
cd ZipVoice.AXERA
|
| 107 |
```
|
| 108 |
|
| 109 |
-
###
|
| 110 |
|
| 111 |
中文句子:
|
| 112 |
|
|
@@ -196,7 +163,7 @@ RTF: 0.9600
|
|
| 196 |
音频:[outputs/en_long_paragraph_ax650.wav](outputs/en_long_paragraph_ax650.wav)
|
| 197 |
提示音:[assets/moss_prompts/en_4_4p5s.wav](assets/moss_prompts/en_4_4p5s.wav)
|
| 198 |
|
| 199 |
-
###
|
| 200 |
|
| 201 |
中文句子:
|
| 202 |
|
|
@@ -286,44 +253,15 @@ RTF: 0.3045
|
|
| 286 |
音频:[outputs/en_long_paragraph_distill_ax650.wav](outputs/en_long_paragraph_distill_ax650.wav)
|
| 287 |
提示音:[assets/moss_prompts/en_4_4p5s.wav](assets/moss_prompts/en_4_4p5s.wav)
|
| 288 |
|
| 289 |
-
#### AX630C ZipVoice Distill
|
| 290 |
-
|
| 291 |
-
普通模型推理太慢,这边仅提供 Distill 模型,参数与 AX650 一致,仅 `--model-name` 改为 `zipvoice_distill_ax630C`:
|
| 292 |
-
|
| 293 |
-
```bash
|
| 294 |
-
python3 infer_zipvoice_axera.py \
|
| 295 |
-
--model-name zipvoice_distill_ax630C \
|
| 296 |
-
--text "今天午后天气很好,我打开窗户,听见远处有人聊天,水杯也轻轻晃了一下。" \
|
| 297 |
-
--prompt-text "不管怎么样我和汤姆还是要感谢贝尔卡金的援手" \
|
| 298 |
-
--prompt-wav assets/moss_prompts/zh_1_4p5s.wav --seed 42
|
| 299 |
-
```
|
| 300 |
|
| 301 |
-
##
|
| 302 |
|
| 303 |
-
- `--model-name`:`zipvoice_ax650`
|
| 304 |
- `--prompt-wav`:参考音频,用于控制音色,建议 3-5s。
|
| 305 |
- `--prompt-text`:参考音频对应文本,必须尽量和 `prompt-wav` 内容一致。
|
| 306 |
- `--num-step`:采样步数。默认从模型目录的 `runtime_config.json` 读取。
|
| 307 |
- `--max-feat-len`:decoder 固定 feature 长度,当前模型均为 1024。
|
| 308 |
|
| 309 |
-
## C++
|
| 310 |
-
|
| 311 |
-
C++ 推理可执行文件在 `bin/` 下,**无需编译**,直接运行:
|
| 312 |
-
|
| 313 |
-
```bash
|
| 314 |
-
# AX650 板端
|
| 315 |
-
bash run_ax650.sh distill zh sentence # distill/standard, zh/en, sentence/paragraph
|
| 316 |
-
# AX630C 板端
|
| 317 |
-
bash run_ax630c.sh zh sentence
|
| 318 |
-
# AXCL 算力卡
|
| 319 |
-
bash run_axcl.sh distill zh paragraph
|
| 320 |
-
```
|
| 321 |
-
|
| 322 |
-
C++ 源码见 [GitHub: AXERA-TECH/ZipVoice.AXERA/cpp](https://github.com/AXERA-TECH/ZipVoice.AXERA/tree/main/cpp)。
|
| 323 |
-
|
| 324 |
-
|
| 325 |
## 参考
|
| 326 |
|
| 327 |
- [ZipVoice](https://github.com/k2-fsa/ZipVoice)
|
| 328 |
-
- [Pulsar2 Docs](https://pulsar2-docs.readthedocs.io/en/latest/pulsar2/introduction.html)
|
| 329 |
-
- [模型量化工程](https://github.com/AXERA-TECH/ZipVoice.AXERA)
|
|
|
|
| 1 |
# ZipVoice.AXERA
|
| 2 |
|
| 3 |
+
[ZipVoice]((https://github.com/k2-fsa/ZipVoice)) AXERA 板端推理 demo。
|
| 4 |
|
| 5 |
## 功能
|
| 6 |
|
| 7 |
- 支持中文和英文语音生成。
|
| 8 |
- 支持语音克隆。
|
| 9 |
- 支持 ZipVoice、ZipVoice Distill
|
|
|
|
|
|
|
| 10 |
|
| 11 |
## 模型说明
|
| 12 |
|
| 13 |
+
ZipVoice Distill 是 ZipVoice 的蒸馏版本,主要优势是在较小性能损失下提升推理速度。初步测试,AX650 ZipVoice Distill 在长文本场景下相比基础版模型约有 3 倍速度提升,RTF 在 0.3 左右,效果没有明显下降。
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 14 |
|
| 15 |
+
AX630C 版本当前推理结果差,RTF 约为 1.5 左右,需要继续调优。
|
| 16 |
|
| 17 |
## 模型转换
|
| 18 |
|
|
|
|
| 28 |
- [M4N-Dock(爱芯派Pro)](https://wiki.sipeed.com/hardware/zh/maixIV/m4ndock/m4ndock.html)
|
| 29 |
- [M.2 Accelerator Card](https://docs.m5stack.com/zh_CN/ai_hardware/LLM-8850_Card)
|
| 30 |
|
|
|
|
|
|
|
| 31 |
|
| 32 |
## 目录结构
|
| 33 |
|
|
|
|
| 39 |
├── models/
|
| 40 |
│ ├── zipvoice_ax650/
|
| 41 |
│ ├── zipvoice_distill_ax650/
|
| 42 |
+
│ └── zipvoice_distill_ax630C/
|
|
|
|
| 43 |
├── resources/
|
| 44 |
+
│ ├── vocos-mel-24khz/
|
| 45 |
+
│ └── zipvoice_hf/
|
|
|
|
|
|
|
|
|
|
|
|
|
| 46 |
├── scripts/
|
| 47 |
+
├── infer_zipvoice_axera.py
|
|
|
|
|
|
|
|
|
|
| 48 |
├── requirements.txt
|
| 49 |
└── README.md
|
| 50 |
```
|
| 51 |
|
| 52 |
+
## 环境
|
|
|
|
|
|
|
| 53 |
|
| 54 |
安装 pyaxengine:
|
| 55 |
+
|
| 56 |
```bash
|
| 57 |
pip3 install axengine-x.x.x-py3-none-any.whl
|
| 58 |
```
|
|
|
|
| 65 |
pip3 install -r requirements.txt
|
| 66 |
```
|
| 67 |
|
| 68 |
+
## 推理命令
|
| 69 |
|
| 70 |
进入目录:
|
| 71 |
|
|
|
|
| 73 |
cd ZipVoice.AXERA
|
| 74 |
```
|
| 75 |
|
| 76 |
+
### AX650 ZipVoice
|
| 77 |
|
| 78 |
中文句子:
|
| 79 |
|
|
|
|
| 163 |
音频:[outputs/en_long_paragraph_ax650.wav](outputs/en_long_paragraph_ax650.wav)
|
| 164 |
提示音:[assets/moss_prompts/en_4_4p5s.wav](assets/moss_prompts/en_4_4p5s.wav)
|
| 165 |
|
| 166 |
+
### AX650 ZipVoice Distill
|
| 167 |
|
| 168 |
中文句子:
|
| 169 |
|
|
|
|
| 253 |
音频:[outputs/en_long_paragraph_distill_ax650.wav](outputs/en_long_paragraph_distill_ax650.wav)
|
| 254 |
提示音:[assets/moss_prompts/en_4_4p5s.wav](assets/moss_prompts/en_4_4p5s.wav)
|
| 255 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 256 |
|
| 257 |
+
## 参数说明
|
| 258 |
|
| 259 |
+
- `--model-name`:选择模型目录。可选 `zipvoice_ax650`、`zipvoice_distill_ax650`、`zipvoice_distill_ax630C`。
|
| 260 |
- `--prompt-wav`:参考音频,用于控制音色,建议 3-5s。
|
| 261 |
- `--prompt-text`:参考音频对应文本,必须尽量和 `prompt-wav` 内容一致。
|
| 262 |
- `--num-step`:采样步数。默认从模型目录的 `runtime_config.json` 读取。
|
| 263 |
- `--max-feat-len`:decoder 固定 feature 长度,当前模型均为 1024。
|
| 264 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 265 |
## 参考
|
| 266 |
|
| 267 |
- [ZipVoice](https://github.com/k2-fsa/ZipVoice)
|
|
|
|
|
|
bin/zipvoice_ax630c
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:adbc16b5766ca2eede9b14b405d5b534e03b66b29043d179b41aa0a5df77d721
|
| 3 |
-
size 527832
|
|
|
|
|
|
|
|
|
|
|
|
bin/zipvoice_ax650
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:8653e75af7dc2571df341a6172a2342c19f6c99761ae640261c798c86087fe67
|
| 3 |
-
size 527848
|
|
|
|
|
|
|
|
|
|
|
|
bin/zipvoice_axcl
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:de730337c6c5ecf11e805f7267447ff387cf5ee80131d1ad4875129586285fc2
|
| 3 |
-
size 581640
|
|
|
|
|
|
|
|
|
|
|
|
configuration.json
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
{"framework": "pytorch", "task": "others", "allow_remote": true}
|
|
|
|
|
|
infer_zipvoice_axera.py
CHANGED
|
@@ -58,7 +58,6 @@ def parse_args() -> argparse.Namespace:
|
|
| 58 |
help="Allow raw features_len up to this ratio of max_feat_len before splitting",
|
| 59 |
)
|
| 60 |
p.add_argument("--silence-ms", type=int, default=140)
|
| 61 |
-
p.add_argument("--vocoder-model", default=None, help="Path to vocos axmodel. If set, use axmodel vocoder instead of PyTorch.")
|
| 62 |
return p.parse_args()
|
| 63 |
|
| 64 |
|
|
@@ -159,13 +158,7 @@ def main() -> None:
|
|
| 159 |
t_shift=args.t_shift,
|
| 160 |
)
|
| 161 |
logging.debug("Loading vocoder...")
|
| 162 |
-
|
| 163 |
-
from scripts.common_infer import load_axmodel_vocoder, axmodel_vocoder_decode
|
| 164 |
-
vocoder = load_axmodel_vocoder(args.vocoder_model)
|
| 165 |
-
vocoder_is_axmodel = True
|
| 166 |
-
else:
|
| 167 |
-
vocoder = load_vocoder(repo_dir)
|
| 168 |
-
vocoder_is_axmodel = False
|
| 169 |
logging.info("模型加载完成")
|
| 170 |
|
| 171 |
audios: list[np.ndarray] = []
|
|
@@ -190,16 +183,13 @@ def main() -> None:
|
|
| 190 |
seed=args.seed + index - 1,
|
| 191 |
)
|
| 192 |
|
| 193 |
-
|
| 194 |
-
|
| 195 |
-
|
| 196 |
-
|
| 197 |
-
|
| 198 |
-
|
| 199 |
-
|
| 200 |
-
vocoder, pred_features,
|
| 201 |
-
feat_scale=args.feat_scale, target_rms=args.target_rms,
|
| 202 |
-
prompt_rms=prompt_rms)
|
| 203 |
segment_sec = time.perf_counter() - t_segment_start
|
| 204 |
audio_sec = len(audio) / 24000
|
| 205 |
|
|
|
|
| 58 |
help="Allow raw features_len up to this ratio of max_feat_len before splitting",
|
| 59 |
)
|
| 60 |
p.add_argument("--silence-ms", type=int, default=140)
|
|
|
|
| 61 |
return p.parse_args()
|
| 62 |
|
| 63 |
|
|
|
|
| 158 |
t_shift=args.t_shift,
|
| 159 |
)
|
| 160 |
logging.debug("Loading vocoder...")
|
| 161 |
+
vocoder = load_vocoder(repo_dir)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 162 |
logging.info("模型加载完成")
|
| 163 |
|
| 164 |
audios: list[np.ndarray] = []
|
|
|
|
| 183 |
seed=args.seed + index - 1,
|
| 184 |
)
|
| 185 |
|
| 186 |
+
audio = vocoder_decode_loaded(
|
| 187 |
+
vocoder,
|
| 188 |
+
pred_features,
|
| 189 |
+
feat_scale=args.feat_scale,
|
| 190 |
+
target_rms=args.target_rms,
|
| 191 |
+
prompt_rms=prompt_rms,
|
| 192 |
+
)
|
|
|
|
|
|
|
|
|
|
| 193 |
segment_sec = time.perf_counter() - t_segment_start
|
| 194 |
audio_sec = len(audio) / 24000
|
| 195 |
|
models/vocoder/vocos_full.axmodel
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:70a7a897f6beb0b6e8536413a0ea6ee2132bec1257e51208e80eca506bda4c40
|
| 3 |
-
size 15413002
|
|
|
|
|
|
|
|
|
|
|
|
models/vocoder/vocos_full_ax630c.axmodel
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:4b704aa370e7875099615ebbb2772c24e7d97529df81f11b713ca04e4261903e
|
| 3 |
-
size 16208038
|
|
|
|
|
|
|
|
|
|
|
|
models/zipvoice_distill_ax630C/decoder4_split_manifest.json
CHANGED
|
@@ -1,104 +1,68 @@
|
|
| 1 |
{
|
| 2 |
-
|
| 3 |
-
|
| 4 |
-
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
| 8 |
-
|
| 9 |
-
|
| 10 |
-
|
| 11 |
-
|
| 12 |
-
|
| 13 |
-
},
|
| 14 |
-
"decoder_parts": [
|
| 15 |
-
{
|
| 16 |
-
"name": "fm_decoder_part0",
|
| 17 |
-
"file": "decoder_part0.axmodel",
|
| 18 |
-
"inputs": [
|
| 19 |
-
"t",
|
| 20 |
-
"x",
|
| 21 |
-
"text_condition",
|
| 22 |
-
"speech_condition",
|
| 23 |
-
"guidance_scale",
|
| 24 |
-
"padding_mask"
|
| 25 |
-
],
|
| 26 |
-
"outputs": [
|
| 27 |
-
"decoder_hidden_p0",
|
| 28 |
-
"time_hidden"
|
| 29 |
-
]
|
| 30 |
-
},
|
| 31 |
-
{
|
| 32 |
-
"name": "fm_decoder_part1",
|
| 33 |
-
"file": "decoder_part1.axmodel",
|
| 34 |
-
"inputs": [
|
| 35 |
-
"decoder_hidden_p0",
|
| 36 |
-
"time_hidden",
|
| 37 |
-
"padding_mask"
|
| 38 |
-
],
|
| 39 |
-
"outputs": [
|
| 40 |
-
"decoder_hidden_p1"
|
| 41 |
-
]
|
| 42 |
-
},
|
| 43 |
-
{
|
| 44 |
-
"name": "fm_decoder_part2",
|
| 45 |
-
"file": "decoder_part2.axmodel",
|
| 46 |
-
"inputs": [
|
| 47 |
-
"decoder_hidden_p1",
|
| 48 |
-
"time_hidden",
|
| 49 |
-
"padding_mask"
|
| 50 |
-
],
|
| 51 |
-
"outputs": [
|
| 52 |
-
"decoder_hidden_p2"
|
| 53 |
-
]
|
| 54 |
-
},
|
| 55 |
-
{
|
| 56 |
-
"name": "fm_decoder_part3_0",
|
| 57 |
-
"file": "decoder_part3_0.axmodel",
|
| 58 |
-
"inputs": [
|
| 59 |
-
"decoder_hidden_p2",
|
| 60 |
-
"time_hidden",
|
| 61 |
-
"padding_mask"
|
| 62 |
-
],
|
| 63 |
-
"outputs": [
|
| 64 |
-
"/fm_decoder/4/0/bypass/Add_output_0"
|
| 65 |
-
]
|
| 66 |
},
|
| 67 |
-
|
| 68 |
-
|
| 69 |
-
|
| 70 |
-
|
| 71 |
-
|
| 72 |
-
|
| 73 |
-
|
| 74 |
-
|
| 75 |
-
|
| 76 |
-
|
| 77 |
-
|
| 78 |
-
|
| 79 |
-
|
| 80 |
-
|
| 81 |
-
|
| 82 |
-
|
| 83 |
-
|
| 84 |
-
|
| 85 |
-
|
| 86 |
-
|
| 87 |
-
|
| 88 |
-
|
| 89 |
-
|
| 90 |
-
|
| 91 |
-
|
| 92 |
-
|
| 93 |
-
|
| 94 |
-
|
| 95 |
-
|
| 96 |
-
|
| 97 |
-
|
| 98 |
-
|
| 99 |
-
|
| 100 |
-
|
| 101 |
-
|
| 102 |
-
|
| 103 |
-
|
| 104 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
{
|
| 2 |
+
"version": 1,
|
| 3 |
+
"model_type": "zipvoice_distill",
|
| 4 |
+
"encoder": {
|
| 5 |
+
"name": "encoder_core",
|
| 6 |
+
"file": "encoder.axmodel",
|
| 7 |
+
"inputs": [
|
| 8 |
+
"cat_tokens"
|
| 9 |
+
],
|
| 10 |
+
"outputs": [
|
| 11 |
+
"encoded"
|
| 12 |
+
]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 13 |
},
|
| 14 |
+
"decoder_parts": [
|
| 15 |
+
{
|
| 16 |
+
"name": "fm_decoder_part0",
|
| 17 |
+
"file": "decoder_part0.axmodel",
|
| 18 |
+
"inputs": [
|
| 19 |
+
"t",
|
| 20 |
+
"x",
|
| 21 |
+
"text_condition",
|
| 22 |
+
"speech_condition",
|
| 23 |
+
"guidance_scale",
|
| 24 |
+
"padding_mask"
|
| 25 |
+
],
|
| 26 |
+
"outputs": [
|
| 27 |
+
"decoder_hidden_p0",
|
| 28 |
+
"time_hidden"
|
| 29 |
+
]
|
| 30 |
+
},
|
| 31 |
+
{
|
| 32 |
+
"name": "fm_decoder_part1",
|
| 33 |
+
"file": "decoder_part1.axmodel",
|
| 34 |
+
"inputs": [
|
| 35 |
+
"decoder_hidden_p0",
|
| 36 |
+
"time_hidden",
|
| 37 |
+
"padding_mask"
|
| 38 |
+
],
|
| 39 |
+
"outputs": [
|
| 40 |
+
"decoder_hidden_p1"
|
| 41 |
+
]
|
| 42 |
+
},
|
| 43 |
+
{
|
| 44 |
+
"name": "fm_decoder_part2",
|
| 45 |
+
"file": "decoder_part2.axmodel",
|
| 46 |
+
"inputs": [
|
| 47 |
+
"decoder_hidden_p1",
|
| 48 |
+
"time_hidden",
|
| 49 |
+
"padding_mask"
|
| 50 |
+
],
|
| 51 |
+
"outputs": [
|
| 52 |
+
"decoder_hidden_p2"
|
| 53 |
+
]
|
| 54 |
+
},
|
| 55 |
+
{
|
| 56 |
+
"name": "fm_decoder_part3",
|
| 57 |
+
"file": "decoder_part3.axmodel",
|
| 58 |
+
"inputs": [
|
| 59 |
+
"decoder_hidden_p2",
|
| 60 |
+
"time_hidden",
|
| 61 |
+
"padding_mask"
|
| 62 |
+
],
|
| 63 |
+
"outputs": [
|
| 64 |
+
"v"
|
| 65 |
+
]
|
| 66 |
+
}
|
| 67 |
+
]
|
| 68 |
+
}
|
models/zipvoice_distill_ax630C/decoder_part0.axmodel
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
-
size
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1f0bd9974d94f349c0001d624d3c2c7140295abbe9c12c0db3556d5759d845a8
|
| 3 |
+
size 42595736
|
models/zipvoice_distill_ax630C/decoder_part1.axmodel
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
-
size
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cfe296e53b7b8f3842aea1bbc60e48f946c0b5f0ae57c639faa4d31d75ff9c5e
|
| 3 |
+
size 31881170
|
models/zipvoice_distill_ax630C/decoder_part2.axmodel
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
-
size
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e95d65c96ba38c7a3cee6ecc558567a4d79e1803ba427749442690afc3b99da9
|
| 3 |
+
size 35101628
|
models/zipvoice_distill_ax630C/decoder_part3_0.axmodel
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:0b4d6a68bbc3286ed14bbe1d7f783bcc8cd559fe709151a99079d6558c1ce7d9
|
| 3 |
-
size 10038370
|
|
|
|
|
|
|
|
|
|
|
|
models/zipvoice_distill_ax630C/decoder_part3_1.axmodel
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:744d3121d74670fec1f29f9a7ce5d158cb215e9b822464ad123fe8daeb8247f1
|
| 3 |
-
size 10038193
|
|
|
|
|
|
|
|
|
|
|
|
models/zipvoice_distill_ax630C/decoder_part3_2.axmodel
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:719c53fb3e4e0949b893c25814f705d33b12cd442fb84434d00dc3867c846ac1
|
| 3 |
-
size 10038201
|
|
|
|
|
|
|
|
|
|
|
|
models/zipvoice_distill_ax630C/decoder_part3_3.axmodel
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:7c44654f65f1398e611b02d9da2a23e931008ffa9ee6a53ebf027fd0da04bd17
|
| 3 |
-
size 10004459
|
|
|
|
|
|
|
|
|
|
|
|
models/zipvoice_distill_ax630C/encoder.axmodel
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
-
size
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ffbd78ded4a05cf472f78a66e34d9c4dac4c5525f6a7b1d2cc5dbb4d96e7538f
|
| 3 |
+
size 8394155
|
run_ax630c.sh
DELETED
|
@@ -1,51 +0,0 @@
|
|
| 1 |
-
#!/bin/bash
|
| 2 |
-
# ZipVoice AX630C 板端一键运行
|
| 3 |
-
# 用法: bash run_ax630c.sh [zh|en] [sentence|paragraph]
|
| 4 |
-
|
| 5 |
-
MODEL="${1:-zh}"
|
| 6 |
-
MODE="${2:-sentence}"
|
| 7 |
-
|
| 8 |
-
BIN="bin/zipvoice_ax630c"
|
| 9 |
-
MODEL_DIR="./models/zipvoice_distill_ax630C"
|
| 10 |
-
TOKEN="./resources/zipvoice_hf/zipvoice/tokens.txt"
|
| 11 |
-
VOCODER="./models/vocoder/vocos_full_ax630c.axmodel"
|
| 12 |
-
|
| 13 |
-
cd "$(dirname "$0")"
|
| 14 |
-
|
| 15 |
-
if [ "$MODEL" = "en" ]; then
|
| 16 |
-
PWAV="./assets/moss_prompts/en_4_4p5s.wav"
|
| 17 |
-
PTEXT="This is almost twice the current industry production level per train."
|
| 18 |
-
DTEXT="This morning, a small train left the station, carrying sleepy passengers toward a bright coastal town."
|
| 19 |
-
TFILE="./assets/paragraphs/en_scavenger.txt"
|
| 20 |
-
REPO="--repo-dir ."
|
| 21 |
-
else
|
| 22 |
-
PWAV="./assets/moss_prompts/zh_1_4p5s.wav"
|
| 23 |
-
PTEXT="不管怎么样我和汤姆还是要感谢贝尔卡金的援手"
|
| 24 |
-
DTEXT="今天午后天气很好,我打开窗户,听见远处有人聊天,水杯也轻轻晃了一下。"
|
| 25 |
-
TFILE="./assets/paragraphs/zh_ginkgo.txt"
|
| 26 |
-
REPO=""
|
| 27 |
-
fi
|
| 28 |
-
|
| 29 |
-
OUT="output_ax630c_${MODEL}_${MODE}.wav"
|
| 30 |
-
|
| 31 |
-
echo "=========================================="
|
| 32 |
-
echo " ZipVoice AX630C"
|
| 33 |
-
echo "=========================================="
|
| 34 |
-
echo " MODEL: $MODEL_DIR"
|
| 35 |
-
echo " LANG: $MODEL MODE: $MODE"
|
| 36 |
-
echo " OUTPUT: $OUT"
|
| 37 |
-
echo "=========================================="
|
| 38 |
-
|
| 39 |
-
if [ "$MODE" = "paragraph" ]; then
|
| 40 |
-
exec $BIN \
|
| 41 |
-
--model-dir "$MODEL_DIR" --token-file "$TOKEN" \
|
| 42 |
-
--prompt-wav "$PWAV" --prompt-text "$PTEXT" \
|
| 43 |
-
--text-file "$TFILE" \
|
| 44 |
-
--vocoder-model "$VOCODER" --output-wav "$OUT" --seed 42 $REPO
|
| 45 |
-
else
|
| 46 |
-
exec $BIN \
|
| 47 |
-
--model-dir "$MODEL_DIR" --token-file "$TOKEN" \
|
| 48 |
-
--prompt-wav "$PWAV" --prompt-text "$PTEXT" \
|
| 49 |
-
--text "$DTEXT" \
|
| 50 |
-
--vocoder-model "$VOCODER" --output-wav "$OUT" --seed 42 $REPO
|
| 51 |
-
fi
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
run_ax650.sh
DELETED
|
@@ -1,57 +0,0 @@
|
|
| 1 |
-
#!/bin/bash
|
| 2 |
-
# ZipVoice AXERA 板端一键运行
|
| 3 |
-
# 用法: bash run_ax650.sh [standard|distill] [zh|en] [sentence|paragraph]
|
| 4 |
-
|
| 5 |
-
MODEL="${1:-distill}"
|
| 6 |
-
LANG="${2:-zh}"
|
| 7 |
-
MODE="${3:-sentence}"
|
| 8 |
-
|
| 9 |
-
BIN="bin/zipvoice_ax650"
|
| 10 |
-
TOKEN="./resources/zipvoice_hf/zipvoice/tokens.txt"
|
| 11 |
-
VOCODER="./models/vocoder/vocos_full.axmodel"
|
| 12 |
-
|
| 13 |
-
cd "$(dirname "$0")"
|
| 14 |
-
|
| 15 |
-
if [ "$MODEL" = "distill" ]; then
|
| 16 |
-
MODEL_DIR="./models/zipvoice_distill_ax650"
|
| 17 |
-
else
|
| 18 |
-
MODEL_DIR="./models/zipvoice_ax650"
|
| 19 |
-
fi
|
| 20 |
-
|
| 21 |
-
if [ "$LANG" = "en" ]; then
|
| 22 |
-
PWAV="./assets/moss_prompts/en_4_4p5s.wav"
|
| 23 |
-
PTEXT="This is almost twice the current industry production level per train."
|
| 24 |
-
DTEXT="This morning, a small train left the station, carrying sleepy passengers toward a bright coastal town."
|
| 25 |
-
TFILE="./assets/paragraphs/en_scavenger.txt"
|
| 26 |
-
REPO="--repo-dir ."
|
| 27 |
-
else
|
| 28 |
-
PWAV="./assets/moss_prompts/zh_1_4p5s.wav"
|
| 29 |
-
PTEXT="不管怎么样我和汤姆还是要感谢贝尔卡金的援手"
|
| 30 |
-
DTEXT="今天午后天气很好,我打开窗户,听见远处有人聊天,水杯也轻轻晃了一下。"
|
| 31 |
-
TFILE="./assets/paragraphs/zh_ginkgo.txt"
|
| 32 |
-
REPO=""
|
| 33 |
-
fi
|
| 34 |
-
|
| 35 |
-
OUT="output_${LANG}_${MODEL}_${MODE}.wav"
|
| 36 |
-
|
| 37 |
-
echo "=========================================="
|
| 38 |
-
echo " ZipVoice AX650"
|
| 39 |
-
echo "=========================================="
|
| 40 |
-
echo " MODEL: $MODEL_DIR"
|
| 41 |
-
echo " LANG: $LANG MODE: $MODE"
|
| 42 |
-
echo " OUTPUT: $OUT"
|
| 43 |
-
echo "=========================================="
|
| 44 |
-
|
| 45 |
-
if [ "$MODE" = "paragraph" ]; then
|
| 46 |
-
exec $BIN \
|
| 47 |
-
--model-dir "$MODEL_DIR" --token-file "$TOKEN" \
|
| 48 |
-
--prompt-wav "$PWAV" --prompt-text "$PTEXT" \
|
| 49 |
-
--text-file "$TFILE" \
|
| 50 |
-
--vocoder-model "$VOCODER" --output-wav "$OUT" --seed 42 $REPO
|
| 51 |
-
else
|
| 52 |
-
exec $BIN \
|
| 53 |
-
--model-dir "$MODEL_DIR" --token-file "$TOKEN" \
|
| 54 |
-
--prompt-wav "$PWAV" --prompt-text "$PTEXT" \
|
| 55 |
-
--text "$DTEXT" \
|
| 56 |
-
--vocoder-model "$VOCODER" --output-wav "$OUT" --seed 42 $REPO
|
| 57 |
-
fi
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
run_axcl.sh
DELETED
|
@@ -1,57 +0,0 @@
|
|
| 1 |
-
#!/bin/bash
|
| 2 |
-
# ZipVoice AXCL 算力卡一键运行
|
| 3 |
-
# 用法: bash run_axcl.sh [standard|distill] [zh|en] [sentence|paragraph]
|
| 4 |
-
|
| 5 |
-
MODEL="${1:-distill}"
|
| 6 |
-
LANG="${2:-zh}"
|
| 7 |
-
MODE="${3:-sentence}"
|
| 8 |
-
|
| 9 |
-
BIN="bin/zipvoice_axcl"
|
| 10 |
-
TOKEN="./resources/zipvoice_hf/zipvoice/tokens.txt"
|
| 11 |
-
VOCODER="./models/vocoder/vocos_full.axmodel"
|
| 12 |
-
|
| 13 |
-
cd "$(dirname "$0")"
|
| 14 |
-
|
| 15 |
-
if [ "$MODEL" = "distill" ]; then
|
| 16 |
-
MODEL_DIR="./models/zipvoice_distill_ax650"
|
| 17 |
-
else
|
| 18 |
-
MODEL_DIR="./models/zipvoice_ax650"
|
| 19 |
-
fi
|
| 20 |
-
|
| 21 |
-
if [ "$LANG" = "en" ]; then
|
| 22 |
-
PWAV="./assets/moss_prompts/en_4_4p5s.wav"
|
| 23 |
-
PTEXT="This is almost twice the current industry production level per train."
|
| 24 |
-
DTEXT="This morning, a small train left the station, carrying sleepy passengers toward a bright coastal town."
|
| 25 |
-
TFILE="./assets/paragraphs/en_scavenger.txt"
|
| 26 |
-
REPO="--repo-dir ."
|
| 27 |
-
else
|
| 28 |
-
PWAV="./assets/moss_prompts/zh_1_4p5s.wav"
|
| 29 |
-
PTEXT="不管怎么样我和汤姆还是要感谢贝尔卡金的援手"
|
| 30 |
-
DTEXT="今天午后天气很好,我打开窗户,听见远处有人聊天,水杯也轻轻晃了一下。"
|
| 31 |
-
TFILE="./assets/paragraphs/zh_ginkgo.txt"
|
| 32 |
-
REPO=""
|
| 33 |
-
fi
|
| 34 |
-
|
| 35 |
-
OUT="output_${LANG}_${MODEL}_${MODE}.wav"
|
| 36 |
-
|
| 37 |
-
echo "=========================================="
|
| 38 |
-
echo " ZipVoice AXCL"
|
| 39 |
-
echo "=========================================="
|
| 40 |
-
echo " MODEL: $MODEL_DIR"
|
| 41 |
-
echo " LANG: $LANG MODE: $MODE"
|
| 42 |
-
echo " OUTPUT: $OUT"
|
| 43 |
-
echo "=========================================="
|
| 44 |
-
|
| 45 |
-
if [ "$MODE" = "paragraph" ]; then
|
| 46 |
-
exec $BIN \
|
| 47 |
-
--model-dir "$MODEL_DIR" --token-file "$TOKEN" \
|
| 48 |
-
--prompt-wav "$PWAV" --prompt-text "$PTEXT" \
|
| 49 |
-
--text-file "$TFILE" \
|
| 50 |
-
--vocoder-model "$VOCODER" --output-wav "$OUT" --seed 42 $REPO
|
| 51 |
-
else
|
| 52 |
-
exec $BIN \
|
| 53 |
-
--model-dir "$MODEL_DIR" --token-file "$TOKEN" \
|
| 54 |
-
--prompt-wav "$PWAV" --prompt-text "$PTEXT" \
|
| 55 |
-
--text "$DTEXT" \
|
| 56 |
-
--vocoder-model "$VOCODER" --output-wav "$OUT" --seed 42 $REPO
|
| 57 |
-
fi
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
scripts/__pycache__/__init__.cpython-310.pyc
DELETED
|
Binary file (155 Bytes)
|
|
|
scripts/__pycache__/local_tokenizer.cpython-310.pyc
DELETED
|
Binary file (7.28 kB)
|
|
|
scripts/__pycache__/text_processing.cpython-310.pyc
DELETED
|
Binary file (8.46 kB)
|
|
|
scripts/common_infer.py
CHANGED
|
@@ -66,74 +66,3 @@ def vocoder_decode_loaded(
|
|
| 66 |
if prompt_rms < target_rms:
|
| 67 |
wav = wav * prompt_rms / target_rms
|
| 68 |
return wav.squeeze().cpu().numpy()
|
| 69 |
-
|
| 70 |
-
|
| 71 |
-
def load_axmodel_vocoder(model_path: str):
|
| 72 |
-
"""Load vocoder axmodel for inference."""
|
| 73 |
-
import axengine as axe
|
| 74 |
-
|
| 75 |
-
session = axe.InferenceSession(model_path)
|
| 76 |
-
return session
|
| 77 |
-
|
| 78 |
-
|
| 79 |
-
def axmodel_vocoder_decode(session, features: np.ndarray, feat_scale: float,
|
| 80 |
-
target_rms: float, prompt_rms: float) -> np.ndarray:
|
| 81 |
-
"""Decode mel features to audio using axmodel vocoder + IRFFT."""
|
| 82 |
-
import math
|
| 83 |
-
|
| 84 |
-
# features shape: (1, T, 100) or (T, 100), squeeze batch dim
|
| 85 |
-
features = np.squeeze(features)
|
| 86 |
-
if features.ndim == 2:
|
| 87 |
-
features = features.T # (T, 100) → (100, T)
|
| 88 |
-
T = features.shape[1]
|
| 89 |
-
n_mels = features.shape[0]
|
| 90 |
-
n_fft, hop = 1024, 256
|
| 91 |
-
T_model = 620
|
| 92 |
-
|
| 93 |
-
# Undo feat_scale and pad to [1, n_mels, T_model]
|
| 94 |
-
inv_scale = 1.0 / feat_scale if feat_scale != 0 else 1.0
|
| 95 |
-
mel_input = np.zeros((1, n_mels, T_model), dtype=np.float32)
|
| 96 |
-
t_actual = min(T, T_model)
|
| 97 |
-
for t in range(t_actual):
|
| 98 |
-
mel_input[0, :, t] = features[:, t] * inv_scale
|
| 99 |
-
|
| 100 |
-
# Run axmodel: mel → (real, imag)
|
| 101 |
-
real, imag = session.run(None, {"mel": mel_input})
|
| 102 |
-
real, imag = real.squeeze(0), imag.squeeze(0) # [T_model, n_freqs]
|
| 103 |
-
|
| 104 |
-
# IRFFT + overlap-add (matching C++ vocoder)
|
| 105 |
-
n_freqs = n_fft // 2 + 1
|
| 106 |
-
window = 0.5 * (1.0 - np.cos(2.0 * math.pi * np.arange(n_fft) / (n_fft - 1)))
|
| 107 |
-
window_sq = window ** 2
|
| 108 |
-
out_len = (T - 1) * hop + n_fft
|
| 109 |
-
audio = np.zeros(out_len, dtype=np.float32)
|
| 110 |
-
envelope = np.zeros(out_len, dtype=np.float32)
|
| 111 |
-
|
| 112 |
-
for t_idx in range(T):
|
| 113 |
-
spec = np.zeros(n_fft, dtype=np.complex64)
|
| 114 |
-
spec[0] = real[t_idx, 0]
|
| 115 |
-
for k in range(1, n_freqs - 1):
|
| 116 |
-
spec[k] = real[t_idx, k] + 1j * imag[t_idx, k]
|
| 117 |
-
spec[n_fft - k] = real[t_idx, k] - 1j * imag[t_idx, k]
|
| 118 |
-
spec[n_freqs - 1] = real[t_idx, n_freqs - 1]
|
| 119 |
-
ifft_out = np.fft.irfft(spec, n=n_fft).real
|
| 120 |
-
|
| 121 |
-
pos = t_idx * hop
|
| 122 |
-
for n in range(n_fft):
|
| 123 |
-
p = pos + n
|
| 124 |
-
if p < out_len:
|
| 125 |
-
audio[p] += ifft_out[n] * window[n]
|
| 126 |
-
envelope[p] += window_sq[n]
|
| 127 |
-
|
| 128 |
-
audio /= np.maximum(envelope, 1e-10)
|
| 129 |
-
pad = n_fft // 2
|
| 130 |
-
audio = audio[pad:out_len - pad].astype(np.float32)
|
| 131 |
-
|
| 132 |
-
# RMS normalize (numpy version)
|
| 133 |
-
rms = np.sqrt(np.mean(audio ** 2))
|
| 134 |
-
if rms < target_rms and rms > 1e-10:
|
| 135 |
-
audio = audio * (target_rms / rms)
|
| 136 |
-
if prompt_rms < target_rms:
|
| 137 |
-
audio = audio * (prompt_rms / target_rms)
|
| 138 |
-
return audio
|
| 139 |
-
|
|
|
|
| 66 |
if prompt_rms < target_rms:
|
| 67 |
wav = wav * prompt_rms / target_rms
|
| 68 |
return wav.squeeze().cpu().numpy()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
scripts/zipvoice_decoder4_runtime.py
CHANGED
|
@@ -276,7 +276,6 @@ class Decoder4ZipVoiceBoardRuntime:
|
|
| 276 |
encoded = self.run_encoder(cat_tokens)
|
| 277 |
t_enc = time.perf_counter() - t_start
|
| 278 |
logging.debug(" encoder: %.3f s (output shape=%s)", t_enc, encoded.shape)
|
| 279 |
-
# DEBUG: dump intermediates
|
| 280 |
|
| 281 |
t_start = time.perf_counter()
|
| 282 |
text_condition, features_len = self.duration_expand(
|
|
@@ -288,7 +287,6 @@ class Decoder4ZipVoiceBoardRuntime:
|
|
| 288 |
)
|
| 289 |
t_dur = time.perf_counter() - t_start
|
| 290 |
logging.debug(" duration_expand: %.3f s (features_len=%d)", t_dur, features_len)
|
| 291 |
-
# DEBUG: dump
|
| 292 |
|
| 293 |
seq_len = self.decoder_seq_len or self.max_feat_len
|
| 294 |
if features_len > seq_len:
|
|
|
|
| 276 |
encoded = self.run_encoder(cat_tokens)
|
| 277 |
t_enc = time.perf_counter() - t_start
|
| 278 |
logging.debug(" encoder: %.3f s (output shape=%s)", t_enc, encoded.shape)
|
|
|
|
| 279 |
|
| 280 |
t_start = time.perf_counter()
|
| 281 |
text_condition, features_len = self.duration_expand(
|
|
|
|
| 287 |
)
|
| 288 |
t_dur = time.perf_counter() - t_start
|
| 289 |
logging.debug(" duration_expand: %.3f s (features_len=%d)", t_dur, features_len)
|
|
|
|
| 290 |
|
| 291 |
seq_len = self.decoder_seq_len or self.max_feat_len
|
| 292 |
if features_len > seq_len:
|
scripts/zipvoice_decoder4_runtime_encoder_onnx.py
DELETED
|
@@ -1,87 +0,0 @@
|
|
| 1 |
-
#!/usr/bin/env python3
|
| 2 |
-
|
| 3 |
-
from __future__ import annotations
|
| 4 |
-
|
| 5 |
-
import logging
|
| 6 |
-
from pathlib import Path
|
| 7 |
-
from typing import Dict, List
|
| 8 |
-
|
| 9 |
-
import numpy as np
|
| 10 |
-
import onnxruntime as ort
|
| 11 |
-
|
| 12 |
-
from scripts.zipvoice_decoder4_runtime import Decoder4ZipVoiceBoardRuntime
|
| 13 |
-
from scripts.zipvoice_runtime import AxeSession
|
| 14 |
-
|
| 15 |
-
|
| 16 |
-
class _OrtSession:
|
| 17 |
-
"""ONNX Runtime wrapper with the AxeSession interface used by split runtime."""
|
| 18 |
-
|
| 19 |
-
_TYPE_TO_DTYPE = {
|
| 20 |
-
"tensor(float)": np.float32,
|
| 21 |
-
"tensor(float32)": np.float32,
|
| 22 |
-
"tensor(double)": np.float64,
|
| 23 |
-
"tensor(int32)": np.int32,
|
| 24 |
-
"tensor(int64)": np.int64,
|
| 25 |
-
"tensor(uint8)": np.uint8,
|
| 26 |
-
"tensor(bool)": np.bool_,
|
| 27 |
-
}
|
| 28 |
-
|
| 29 |
-
def __init__(self, model_path: str | Path):
|
| 30 |
-
self.path = Path(model_path)
|
| 31 |
-
if not self.path.exists():
|
| 32 |
-
raise FileNotFoundError(f"ONNX model not found: {self.path}")
|
| 33 |
-
self._session = ort.InferenceSession(
|
| 34 |
-
str(self.path),
|
| 35 |
-
providers=["CPUExecutionProvider"],
|
| 36 |
-
)
|
| 37 |
-
self._inputs = self._session.get_inputs()
|
| 38 |
-
self._outputs = self._session.get_outputs()
|
| 39 |
-
|
| 40 |
-
@property
|
| 41 |
-
def input_names(self) -> List[str]:
|
| 42 |
-
return [item.name for item in self._inputs]
|
| 43 |
-
|
| 44 |
-
@property
|
| 45 |
-
def output_names(self) -> List[str]:
|
| 46 |
-
return [item.name for item in self._outputs]
|
| 47 |
-
|
| 48 |
-
def _coerce_one(self, name: str, value: np.ndarray) -> np.ndarray:
|
| 49 |
-
info = next((item for item in self._inputs if item.name == name), None)
|
| 50 |
-
array = np.asarray(value)
|
| 51 |
-
if info is None:
|
| 52 |
-
return np.ascontiguousarray(array)
|
| 53 |
-
|
| 54 |
-
dtype = self._TYPE_TO_DTYPE.get(str(info.type).lower())
|
| 55 |
-
if dtype is not None:
|
| 56 |
-
array = array.astype(dtype, copy=False)
|
| 57 |
-
|
| 58 |
-
if list(info.shape) == [] and array.shape == (1,):
|
| 59 |
-
array = array.reshape(())
|
| 60 |
-
|
| 61 |
-
return np.ascontiguousarray(array)
|
| 62 |
-
|
| 63 |
-
def run(self, feed_dict: Dict[str, np.ndarray]) -> Dict[str, np.ndarray]:
|
| 64 |
-
feed = {name: self._coerce_one(name, value) for name, value in feed_dict.items()}
|
| 65 |
-
outputs = self._session.run(None, feed)
|
| 66 |
-
return {name: value for name, value in zip(self.output_names, outputs)}
|
| 67 |
-
|
| 68 |
-
|
| 69 |
-
class Decoder4ZipVoiceBoardRuntimeEncoderOnnx(Decoder4ZipVoiceBoardRuntime):
|
| 70 |
-
"""Runs encoder with ONNX Runtime, all decoder parts with axmodel."""
|
| 71 |
-
|
| 72 |
-
def _load_models(self) -> None:
|
| 73 |
-
self.sessions = {}
|
| 74 |
-
|
| 75 |
-
encoder_name = self.encoder_info["name"]
|
| 76 |
-
encoder_path = self.models_dir / "encoder_core.onnx"
|
| 77 |
-
logging.info("encoder 使用 ONNX Runtime: %s", encoder_path)
|
| 78 |
-
self.sessions[encoder_name] = _OrtSession(encoder_path)
|
| 79 |
-
|
| 80 |
-
for info in self.decoder_parts:
|
| 81 |
-
name = info["name"]
|
| 82 |
-
path = self.models_dir / info["file"]
|
| 83 |
-
logging.debug("Loading %s from %s", name, path)
|
| 84 |
-
self.sessions[name] = AxeSession(path)
|
| 85 |
-
|
| 86 |
-
self.decoder_label = "decoder4(encoder_onnx+part0-3_axmodel)"
|
| 87 |
-
logging.debug("Loaded encoder ONNX + %d decoder axmodels", len(self.decoder_parts))
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
scripts/zipvoice_decoder4_runtime_part0_onnx.py
DELETED
|
@@ -1,94 +0,0 @@
|
|
| 1 |
-
#!/usr/bin/env python3
|
| 2 |
-
|
| 3 |
-
from __future__ import annotations
|
| 4 |
-
|
| 5 |
-
import logging
|
| 6 |
-
from pathlib import Path
|
| 7 |
-
from typing import Dict, List
|
| 8 |
-
|
| 9 |
-
import numpy as np
|
| 10 |
-
import onnxruntime as ort
|
| 11 |
-
|
| 12 |
-
from scripts.zipvoice_decoder4_runtime import Decoder4ZipVoiceBoardRuntime
|
| 13 |
-
from scripts.zipvoice_runtime import AxeSession
|
| 14 |
-
|
| 15 |
-
|
| 16 |
-
class _OrtSession:
|
| 17 |
-
"""ONNX Runtime wrapper with the small AxeSession interface used by the board runtime."""
|
| 18 |
-
|
| 19 |
-
_TYPE_TO_DTYPE = {
|
| 20 |
-
"tensor(float)": np.float32,
|
| 21 |
-
"tensor(float32)": np.float32,
|
| 22 |
-
"tensor(double)": np.float64,
|
| 23 |
-
"tensor(int32)": np.int32,
|
| 24 |
-
"tensor(int64)": np.int64,
|
| 25 |
-
"tensor(uint8)": np.uint8,
|
| 26 |
-
"tensor(bool)": np.bool_,
|
| 27 |
-
}
|
| 28 |
-
|
| 29 |
-
def __init__(self, model_path: str | Path):
|
| 30 |
-
self.path = Path(model_path)
|
| 31 |
-
if not self.path.exists():
|
| 32 |
-
raise FileNotFoundError(f"ONNX model not found: {self.path}")
|
| 33 |
-
self._session = ort.InferenceSession(
|
| 34 |
-
str(self.path),
|
| 35 |
-
providers=["CPUExecutionProvider"],
|
| 36 |
-
)
|
| 37 |
-
self._inputs = self._session.get_inputs()
|
| 38 |
-
self._outputs = self._session.get_outputs()
|
| 39 |
-
|
| 40 |
-
@property
|
| 41 |
-
def input_names(self) -> List[str]:
|
| 42 |
-
return [item.name for item in self._inputs]
|
| 43 |
-
|
| 44 |
-
@property
|
| 45 |
-
def output_names(self) -> List[str]:
|
| 46 |
-
return [item.name for item in self._outputs]
|
| 47 |
-
|
| 48 |
-
def _coerce_one(self, name: str, value: np.ndarray) -> np.ndarray:
|
| 49 |
-
info = next((item for item in self._inputs if item.name == name), None)
|
| 50 |
-
array = np.asarray(value)
|
| 51 |
-
if info is None:
|
| 52 |
-
return np.ascontiguousarray(array)
|
| 53 |
-
|
| 54 |
-
dtype = self._TYPE_TO_DTYPE.get(str(info.type).lower())
|
| 55 |
-
if dtype is not None:
|
| 56 |
-
array = array.astype(dtype, copy=False)
|
| 57 |
-
|
| 58 |
-
# The exported part0 ONNX keeps t/guidance_scale as scalar inputs ([]),
|
| 59 |
-
# while the axmodel path feeds them as shape [1]. Normalize only for ONNX.
|
| 60 |
-
if list(info.shape) == [] and array.shape == (1,):
|
| 61 |
-
array = array.reshape(())
|
| 62 |
-
|
| 63 |
-
return np.ascontiguousarray(array)
|
| 64 |
-
|
| 65 |
-
def run(self, feed_dict: Dict[str, np.ndarray]) -> Dict[str, np.ndarray]:
|
| 66 |
-
feed = {name: self._coerce_one(name, value) for name, value in feed_dict.items()}
|
| 67 |
-
outputs = self._session.run(None, feed)
|
| 68 |
-
return {name: value for name, value in zip(self.output_names, outputs)}
|
| 69 |
-
|
| 70 |
-
|
| 71 |
-
class Decoder4ZipVoiceBoardRuntimePart0Onnx(Decoder4ZipVoiceBoardRuntime):
|
| 72 |
-
"""Runs decoder part0 with ONNX Runtime, all other split models with axmodel."""
|
| 73 |
-
|
| 74 |
-
def _load_models(self) -> None:
|
| 75 |
-
self.sessions = {}
|
| 76 |
-
|
| 77 |
-
encoder_name = self.encoder_info["name"]
|
| 78 |
-
encoder_path = self.models_dir / self.encoder_info["file"]
|
| 79 |
-
logging.debug("Loading %s from %s", encoder_name, encoder_path)
|
| 80 |
-
self.sessions[encoder_name] = AxeSession(encoder_path)
|
| 81 |
-
|
| 82 |
-
for index, info in enumerate(self.decoder_parts):
|
| 83 |
-
name = info["name"]
|
| 84 |
-
if index == 0:
|
| 85 |
-
path = self.models_dir / "fm_decoder_part0.onnx"
|
| 86 |
-
logging.info("part0 使用 ONNX Runtime: %s", path)
|
| 87 |
-
self.sessions[name] = _OrtSession(path)
|
| 88 |
-
else:
|
| 89 |
-
path = self.models_dir / info["file"]
|
| 90 |
-
logging.debug("Loading %s from %s", name, path)
|
| 91 |
-
self.sessions[name] = AxeSession(path)
|
| 92 |
-
|
| 93 |
-
self.decoder_label = "decoder4(part0_onnx+part1-3_axmodel)"
|
| 94 |
-
logging.debug("Loaded encoder axmodel + part0 ONNX + %d decoder axmodels", len(self.decoder_parts) - 1)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
scripts/zipvoice_decoder4_runtime_part3_onnx.py
DELETED
|
@@ -1,93 +0,0 @@
|
|
| 1 |
-
#!/usr/bin/env python3
|
| 2 |
-
|
| 3 |
-
from __future__ import annotations
|
| 4 |
-
|
| 5 |
-
import logging
|
| 6 |
-
from pathlib import Path
|
| 7 |
-
from typing import Dict, List
|
| 8 |
-
|
| 9 |
-
import numpy as np
|
| 10 |
-
import onnxruntime as ort
|
| 11 |
-
|
| 12 |
-
from scripts.zipvoice_decoder4_runtime import Decoder4ZipVoiceBoardRuntime
|
| 13 |
-
from scripts.zipvoice_runtime import AxeSession
|
| 14 |
-
|
| 15 |
-
|
| 16 |
-
class _OrtSession:
|
| 17 |
-
"""ONNX Runtime wrapper with the AxeSession interface used by split runtime."""
|
| 18 |
-
|
| 19 |
-
_TYPE_TO_DTYPE = {
|
| 20 |
-
"tensor(float)": np.float32,
|
| 21 |
-
"tensor(float32)": np.float32,
|
| 22 |
-
"tensor(double)": np.float64,
|
| 23 |
-
"tensor(int32)": np.int32,
|
| 24 |
-
"tensor(int64)": np.int64,
|
| 25 |
-
"tensor(uint8)": np.uint8,
|
| 26 |
-
"tensor(bool)": np.bool_,
|
| 27 |
-
}
|
| 28 |
-
|
| 29 |
-
def __init__(self, model_path: str | Path):
|
| 30 |
-
self.path = Path(model_path)
|
| 31 |
-
if not self.path.exists():
|
| 32 |
-
raise FileNotFoundError(f"ONNX model not found: {self.path}")
|
| 33 |
-
self._session = ort.InferenceSession(
|
| 34 |
-
str(self.path),
|
| 35 |
-
providers=["CPUExecutionProvider"],
|
| 36 |
-
)
|
| 37 |
-
self._inputs = self._session.get_inputs()
|
| 38 |
-
self._outputs = self._session.get_outputs()
|
| 39 |
-
|
| 40 |
-
@property
|
| 41 |
-
def input_names(self) -> List[str]:
|
| 42 |
-
return [item.name for item in self._inputs]
|
| 43 |
-
|
| 44 |
-
@property
|
| 45 |
-
def output_names(self) -> List[str]:
|
| 46 |
-
return [item.name for item in self._outputs]
|
| 47 |
-
|
| 48 |
-
def _coerce_one(self, name: str, value: np.ndarray) -> np.ndarray:
|
| 49 |
-
info = next((item for item in self._inputs if item.name == name), None)
|
| 50 |
-
array = np.asarray(value)
|
| 51 |
-
if info is None:
|
| 52 |
-
return np.ascontiguousarray(array)
|
| 53 |
-
|
| 54 |
-
dtype = self._TYPE_TO_DTYPE.get(str(info.type).lower())
|
| 55 |
-
if dtype is not None:
|
| 56 |
-
array = array.astype(dtype, copy=False)
|
| 57 |
-
|
| 58 |
-
if list(info.shape) == [] and array.shape == (1,):
|
| 59 |
-
array = array.reshape(())
|
| 60 |
-
|
| 61 |
-
return np.ascontiguousarray(array)
|
| 62 |
-
|
| 63 |
-
def run(self, feed_dict: Dict[str, np.ndarray]) -> Dict[str, np.ndarray]:
|
| 64 |
-
feed = {name: self._coerce_one(name, value) for name, value in feed_dict.items()}
|
| 65 |
-
outputs = self._session.run(None, feed)
|
| 66 |
-
return {name: value for name, value in zip(self.output_names, outputs)}
|
| 67 |
-
|
| 68 |
-
|
| 69 |
-
class Decoder4ZipVoiceBoardRuntimePart3Onnx(Decoder4ZipVoiceBoardRuntime):
|
| 70 |
-
"""Runs decoder part3 with ONNX Runtime, encoder and part0-2 with axmodel."""
|
| 71 |
-
|
| 72 |
-
def _load_models(self) -> None:
|
| 73 |
-
self.sessions = {}
|
| 74 |
-
|
| 75 |
-
encoder_name = self.encoder_info["name"]
|
| 76 |
-
encoder_path = self.models_dir / self.encoder_info["file"]
|
| 77 |
-
logging.debug("Loading %s from %s", encoder_name, encoder_path)
|
| 78 |
-
self.sessions[encoder_name] = AxeSession(encoder_path)
|
| 79 |
-
|
| 80 |
-
last_index = len(self.decoder_parts) - 1
|
| 81 |
-
for index, info in enumerate(self.decoder_parts):
|
| 82 |
-
name = info["name"]
|
| 83 |
-
if index == last_index:
|
| 84 |
-
path = self.models_dir / "fm_decoder_part3.onnx"
|
| 85 |
-
logging.info("part3 使用 ONNX Runtime: %s", path)
|
| 86 |
-
self.sessions[name] = _OrtSession(path)
|
| 87 |
-
else:
|
| 88 |
-
path = self.models_dir / info["file"]
|
| 89 |
-
logging.debug("Loading %s from %s", name, path)
|
| 90 |
-
self.sessions[name] = AxeSession(path)
|
| 91 |
-
|
| 92 |
-
self.decoder_label = "decoder4(part0-2_axmodel+part3_onnx)"
|
| 93 |
-
logging.debug("Loaded encoder axmodel + %d decoder axmodels + part3 ONNX", last_index)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|