Research zipvoice

#1
.gitattributes CHANGED
@@ -35,16 +35,6 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
  assets/moss_prompts/en_4_4p5s.wav filter=lfs diff=lfs merge=lfs -text
37
  assets/moss_prompts/zh_1_4p5s.wav filter=lfs diff=lfs merge=lfs -text
38
- cpp/install/ax650/zipvoice filter=lfs diff=lfs merge=lfs -text
39
- cpp/install/ax650/zipvoice_axera filter=lfs diff=lfs merge=lfs -text
40
- cpp/install/axcl/zipvoice filter=lfs diff=lfs merge=lfs -text
41
- cpp/vocoder/vocos_full.axmodel filter=lfs diff=lfs merge=lfs -text
42
- cpp/vocoder/vocos_full_ax630c.axmodel filter=lfs diff=lfs merge=lfs -text
43
- models/axmodels/decoder_part0.axmodel filter=lfs diff=lfs merge=lfs -text
44
- models/axmodels/decoder_part1.axmodel filter=lfs diff=lfs merge=lfs -text
45
- models/axmodels/decoder_part2.axmodel filter=lfs diff=lfs merge=lfs -text
46
- models/axmodels/decoder_part3.axmodel filter=lfs diff=lfs merge=lfs -text
47
- models/axmodels/encoder.axmodel filter=lfs diff=lfs merge=lfs -text
48
  models/zipvoice_ax650/decoder_part0.axmodel filter=lfs diff=lfs merge=lfs -text
49
  models/zipvoice_ax650/decoder_part1.axmodel filter=lfs diff=lfs merge=lfs -text
50
  models/zipvoice_ax650/decoder_part2.axmodel filter=lfs diff=lfs merge=lfs -text
@@ -54,26 +44,25 @@ models/zipvoice_distill_ax630C/decoder_part0.axmodel filter=lfs diff=lfs merge=l
54
  models/zipvoice_distill_ax630C/decoder_part1.axmodel filter=lfs diff=lfs merge=lfs -text
55
  models/zipvoice_distill_ax630C/decoder_part2.axmodel filter=lfs diff=lfs merge=lfs -text
56
  models/zipvoice_distill_ax630C/decoder_part3.axmodel filter=lfs diff=lfs merge=lfs -text
57
- models/zipvoice_distill_ax630C/decoder_part3_0.axmodel filter=lfs diff=lfs merge=lfs -text
58
- models/zipvoice_distill_ax630C/decoder_part3_1.axmodel filter=lfs diff=lfs merge=lfs -text
59
- models/zipvoice_distill_ax630C/decoder_part3_2.axmodel filter=lfs diff=lfs merge=lfs -text
60
- models/zipvoice_distill_ax630C/decoder_part3_3.axmodel filter=lfs diff=lfs merge=lfs -text
61
  models/zipvoice_distill_ax630C/encoder.axmodel filter=lfs diff=lfs merge=lfs -text
62
  models/zipvoice_distill_ax650/decoder_part0.axmodel filter=lfs diff=lfs merge=lfs -text
63
  models/zipvoice_distill_ax650/decoder_part1.axmodel filter=lfs diff=lfs merge=lfs -text
64
  models/zipvoice_distill_ax650/decoder_part2.axmodel filter=lfs diff=lfs merge=lfs -text
65
  models/zipvoice_distill_ax650/decoder_part3.axmodel filter=lfs diff=lfs merge=lfs -text
66
  models/zipvoice_distill_ax650/encoder.axmodel filter=lfs diff=lfs merge=lfs -text
67
- outputs/en_long_paragraph_ax650.wav filter=lfs diff=lfs merge=lfs -text
68
  outputs/en_long_paragraph_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
69
- outputs/en_sentence_ax650.wav filter=lfs diff=lfs merge=lfs -text
70
  outputs/en_sentence_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
71
- outputs/zh_long_paragraph_ax650.wav filter=lfs diff=lfs merge=lfs -text
72
  outputs/zh_long_paragraph_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
73
- outputs/zh_sentence_ax650.wav filter=lfs diff=lfs merge=lfs -text
74
  outputs/zh_sentence_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
75
- bin/zipvoice_ax630c filter=lfs diff=lfs merge=lfs -text
76
- bin/zipvoice_ax650 filter=lfs diff=lfs merge=lfs -text
77
- bin/zipvoice_axcl filter=lfs diff=lfs merge=lfs -text
78
- models/vocoder/vocos_full.axmodel filter=lfs diff=lfs merge=lfs -text
79
- models/vocoder/vocos_full_ax630c.axmodel filter=lfs diff=lfs merge=lfs -text
 
 
 
 
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
  assets/moss_prompts/en_4_4p5s.wav filter=lfs diff=lfs merge=lfs -text
37
  assets/moss_prompts/zh_1_4p5s.wav filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
38
  models/zipvoice_ax650/decoder_part0.axmodel filter=lfs diff=lfs merge=lfs -text
39
  models/zipvoice_ax650/decoder_part1.axmodel filter=lfs diff=lfs merge=lfs -text
40
  models/zipvoice_ax650/decoder_part2.axmodel filter=lfs diff=lfs merge=lfs -text
 
44
  models/zipvoice_distill_ax630C/decoder_part1.axmodel filter=lfs diff=lfs merge=lfs -text
45
  models/zipvoice_distill_ax630C/decoder_part2.axmodel filter=lfs diff=lfs merge=lfs -text
46
  models/zipvoice_distill_ax630C/decoder_part3.axmodel filter=lfs diff=lfs merge=lfs -text
 
 
 
 
47
  models/zipvoice_distill_ax630C/encoder.axmodel filter=lfs diff=lfs merge=lfs -text
48
  models/zipvoice_distill_ax650/decoder_part0.axmodel filter=lfs diff=lfs merge=lfs -text
49
  models/zipvoice_distill_ax650/decoder_part1.axmodel filter=lfs diff=lfs merge=lfs -text
50
  models/zipvoice_distill_ax650/decoder_part2.axmodel filter=lfs diff=lfs merge=lfs -text
51
  models/zipvoice_distill_ax650/decoder_part3.axmodel filter=lfs diff=lfs merge=lfs -text
52
  models/zipvoice_distill_ax650/encoder.axmodel filter=lfs diff=lfs merge=lfs -text
53
+ outputs/en_long_paragraph.wav filter=lfs diff=lfs merge=lfs -text
54
  outputs/en_long_paragraph_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
55
+ outputs/en_sentence.wav filter=lfs diff=lfs merge=lfs -text
56
  outputs/en_sentence_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
57
+ outputs/zh_long_paragraph.wav filter=lfs diff=lfs merge=lfs -text
58
  outputs/zh_long_paragraph_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
59
+ outputs/zh_sentence.wav filter=lfs diff=lfs merge=lfs -text
60
  outputs/zh_sentence_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
61
+ en_long_paragraph_ax650.wav filter=lfs diff=lfs merge=lfs -text
62
+ en_long_paragraph_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
63
+ en_sentence_ax650.wav filter=lfs diff=lfs merge=lfs -text
64
+ en_sentence_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
65
+ zh_long_paragraph_ax650.wav filter=lfs diff=lfs merge=lfs -text
66
+ zh_long_paragraph_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
67
+ zh_sentence_ax650.wav filter=lfs diff=lfs merge=lfs -text
68
+ zh_sentence_distill_ax650.wav filter=lfs diff=lfs merge=lfs -text
.gitignore DELETED
@@ -1,24 +0,0 @@
1
- # HuggingFace CLI 缓存
2
- .msc/
3
- .mv/
4
- ._____temp/
5
- .cache/
6
-
7
- # 运行产物
8
- outputs/
9
- reference/
10
- __pycache__/
11
- *.pyc
12
-
13
- # 编译产物/源码(HUG 仓只放可执行文件,不放源码)
14
- cpp/src/
15
- cpp/third_party/
16
- cpp/toolchains/
17
- cpp/cmake/
18
- cpp/scripts/
19
- cpp/CMakeLists.txt
20
- cpp/build_*.sh
21
- cpp/download_bsp.sh
22
- cpp/*.cpp
23
- cpp/*.hpp
24
- build/
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
README.md CHANGED
@@ -1,39 +1,18 @@
1
  # ZipVoice.AXERA
2
 
3
- [ZipVoice]((https://github.com/k2-fsa/ZipVoice)) AXERA 板端 & AXCL 算力卡推理 demo。
4
 
5
  ## 功能
6
 
7
  - 支持中文和英文语音生成。
8
  - 支持语音克隆。
9
  - 支持 ZipVoice、ZipVoice Distill
10
- - 支持 C++ 和 Python 推理
11
- - 支持 AX650、AX630C
12
 
13
  ## 模型说明
14
 
15
- ZipVoice Distill 是 ZipVoice 的蒸馏版本,主要优势是在较小性能损失下提升推理速度。
16
-
17
- ## 性能对比
18
-
19
- ### Distill 模型 (num_step=4)
20
-
21
- | 场景 | 音频时长 | Python AX650<br>耗时 RTF | C++ AX650<br>耗时 RTF | C++ AXCL<br>耗时 RTF | Python AX630C<br>耗时 RTF | C++ AX630C<br>耗时 RTF |
22
- |------|---------|-------------------------|----------------------|---------------------|--------------------------|-----------------------|
23
- | 中文句子 | 6.41s | 1.992s/0.311 | 1.057s/0.165 | 0.918s/**0.143** | 10.296s/1.606 | 4.650s/0.725 |
24
- | 中文段落 | 44.97s | 13.457s/0.301 | 7.357s/0.164 | 6.443s/**0.143** | 71.574s/1.600 | 32.590s/0.725 |
25
- | 英文句子 | 6.41s | 2.045s/0.319 | 1.067s/0.166 | 0.919s/**0.143** | 10.686s/1.667 | 4.654s/0.726 |
26
- | 英文段落 | 59.16s | 19.715s/0.305 | 10.529s/0.178 | 9.207s/**0.156** | 106.183s/1.640 | 46.541s/0.787 |
27
-
28
- ### 普通模型 (num_step=10)
29
-
30
- | 场景 | 音频时长 | Python AX650 (耗时/RTF) | C++ AX650 (耗时/RTF) | C++ AXCL (耗时/RTF) |
31
- |------|---------|------------------------|---------------------|---------------------|
32
- | 中文句子 | 6.41s | 5.781s / 0.902 | 5.709s / 0.891 | 5.452s / **0.850** |
33
- | 中文段落 | 44.97s | 40.292s / 0.901 | 39.714s / 0.883 | 38.138s / **0.848** |
34
- | 英文句子 | 6.41s | 5.711s / 0.891 | 5.693s / 0.888 | 5.447s / **0.850** |
35
- | 英文段落 | 59.16s | 62.161s / 0.960 | 56.759s / 0.957 | 54.451s / **0.920** |
36
 
 
37
 
38
  ## 模型转换
39
 
@@ -49,8 +28,6 @@ ZipVoice Distill 是 ZipVoice 的蒸馏版本,主要优势是在较小性能
49
  - [M4N-Dock(爱芯派Pro)](https://wiki.sipeed.com/hardware/zh/maixIV/m4ndock/m4ndock.html)
50
  - [M.2 Accelerator Card](https://docs.m5stack.com/zh_CN/ai_hardware/LLM-8850_Card)
51
 
52
- - AX630C
53
-
54
 
55
  ## 目录结构
56
 
@@ -62,30 +39,20 @@ ZipVoice.AXERA/
62
  ├── models/
63
  │ ├── zipvoice_ax650/
64
  │ ├── zipvoice_distill_ax650/
65
- ── zipvoice_distill_ax630C/
66
- │ └── vocoder/ # vocos_full.axmodel(AX650/AX630C)
67
  ├── resources/
68
- │ ├── vocos-mel-24khz/ # Python 推理用 vocoder 权重
69
- │ └── zipvoice_hf/ # tokens.txt
70
- ├── bin/ # C++ 推理可执行文件(源码见 GitHub)
71
- │ ├── zipvoice_ax650
72
- │ ├── zipvoice_ax630c
73
- │ └── zipvoice_axcl
74
  ├── scripts/
75
- ├── infer_zipvoice_axera.py # Python 推理入口
76
- ├── run_ax650.sh # AX650 板端
77
- ├── run_ax630c.sh # AX630C 板端
78
- ├── run_axcl.sh # AXCL 算力卡
79
  ├── requirements.txt
80
  └── README.md
81
  ```
82
 
83
- ## Python
84
-
85
- ### 环境
86
 
87
  安装 pyaxengine:
88
- [pyaxengine Releases](https://github.com/AXERA-TECH/pyaxengine/releases/latest) 下载对应版本安装:
89
  ```bash
90
  pip3 install axengine-x.x.x-py3-none-any.whl
91
  ```
@@ -98,7 +65,7 @@ conda activate ZipVoice
98
  pip3 install -r requirements.txt
99
  ```
100
 
101
- ### 推理命令
102
 
103
  进入目录:
104
 
@@ -106,7 +73,7 @@ pip3 install -r requirements.txt
106
  cd ZipVoice.AXERA
107
  ```
108
 
109
- #### AX650 ZipVoice
110
 
111
  中文句子:
112
 
@@ -196,7 +163,7 @@ RTF: 0.9600
196
  音频:[outputs/en_long_paragraph_ax650.wav](outputs/en_long_paragraph_ax650.wav)
197
  提示音:[assets/moss_prompts/en_4_4p5s.wav](assets/moss_prompts/en_4_4p5s.wav)
198
 
199
- #### AX650 ZipVoice Distill
200
 
201
  中文句子:
202
 
@@ -286,44 +253,15 @@ RTF: 0.3045
286
  音频:[outputs/en_long_paragraph_distill_ax650.wav](outputs/en_long_paragraph_distill_ax650.wav)
287
  提示音:[assets/moss_prompts/en_4_4p5s.wav](assets/moss_prompts/en_4_4p5s.wav)
288
 
289
- #### AX630C ZipVoice Distill
290
-
291
- 普通模型推理太慢,这边仅提供 Distill 模型,参数与 AX650 一致,仅 `--model-name` 改为 `zipvoice_distill_ax630C`:
292
-
293
- ```bash
294
- python3 infer_zipvoice_axera.py \
295
- --model-name zipvoice_distill_ax630C \
296
- --text "今天午后天气很好,我打开窗户,听见远处有人聊天,水杯也轻轻晃了一下。" \
297
- --prompt-text "不管怎么样我和汤姆还是要感谢贝尔卡金的援手" \
298
- --prompt-wav assets/moss_prompts/zh_1_4p5s.wav --seed 42
299
- ```
300
 
301
- ### 参数说明
302
 
303
- - `--model-name`:`zipvoice_ax650` / `zipvoice_distill_ax650` / `zipvoice_distill_ax630C`。
304
  - `--prompt-wav`:参考音频,用于控制音色,建议 3-5s。
305
  - `--prompt-text`:参考音频对应文本,必须尽量和 `prompt-wav` 内容一致。
306
  - `--num-step`:采样步数。默认从模型目录的 `runtime_config.json` 读取。
307
  - `--max-feat-len`:decoder 固定 feature 长度,当前模型均为 1024。
308
 
309
- ## C++
310
-
311
- C++ 推理可执行文件在 `bin/` 下,**无需编译**,直接运行:
312
-
313
- ```bash
314
- # AX650 板端
315
- bash run_ax650.sh distill zh sentence # distill/standard, zh/en, sentence/paragraph
316
- # AX630C 板端
317
- bash run_ax630c.sh zh sentence
318
- # AXCL 算力卡
319
- bash run_axcl.sh distill zh paragraph
320
- ```
321
-
322
- C++ 源码见 [GitHub: AXERA-TECH/ZipVoice.AXERA/cpp](https://github.com/AXERA-TECH/ZipVoice.AXERA/tree/main/cpp)。
323
-
324
-
325
  ## 参考
326
 
327
  - [ZipVoice](https://github.com/k2-fsa/ZipVoice)
328
- - [Pulsar2 Docs](https://pulsar2-docs.readthedocs.io/en/latest/pulsar2/introduction.html)
329
- - [模型量化工程](https://github.com/AXERA-TECH/ZipVoice.AXERA)
 
1
  # ZipVoice.AXERA
2
 
3
+ [ZipVoice]((https://github.com/k2-fsa/ZipVoice)) AXERA 板端推理 demo。
4
 
5
  ## 功能
6
 
7
  - 支持中文和英文语音生成。
8
  - 支持语音克隆。
9
  - 支持 ZipVoice、ZipVoice Distill
 
 
10
 
11
  ## 模型说明
12
 
13
+ ZipVoice Distill 是 ZipVoice 的蒸馏版本,主要优势是在较小性能损失下提升推理速度。初步测试,AX650 ZipVoice Distill 在长文本场景下相比基础版模型约有 3 倍速度提升,RTF 在 0.3 左右,效果没有明显下降。
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
14
 
15
+ AX630C 版本当前推理结果差,RTF 约为 1.5 左右,需要继续调优。
16
 
17
  ## 模型转换
18
 
 
28
  - [M4N-Dock(爱芯派Pro)](https://wiki.sipeed.com/hardware/zh/maixIV/m4ndock/m4ndock.html)
29
  - [M.2 Accelerator Card](https://docs.m5stack.com/zh_CN/ai_hardware/LLM-8850_Card)
30
 
 
 
31
 
32
  ## 目录结构
33
 
 
39
  ├── models/
40
  │ ├── zipvoice_ax650/
41
  │ ├── zipvoice_distill_ax650/
42
+ ── zipvoice_distill_ax630C/
 
43
  ├── resources/
44
+ │ ├── vocos-mel-24khz/
45
+ │ └── zipvoice_hf/
 
 
 
 
46
  ├── scripts/
47
+ ├── infer_zipvoice_axera.py
 
 
 
48
  ├── requirements.txt
49
  └── README.md
50
  ```
51
 
52
+ ## 环境
 
 
53
 
54
  安装 pyaxengine:
55
+
56
  ```bash
57
  pip3 install axengine-x.x.x-py3-none-any.whl
58
  ```
 
65
  pip3 install -r requirements.txt
66
  ```
67
 
68
+ ## 推理命令
69
 
70
  进入目录:
71
 
 
73
  cd ZipVoice.AXERA
74
  ```
75
 
76
+ ### AX650 ZipVoice
77
 
78
  中文句子:
79
 
 
163
  音频:[outputs/en_long_paragraph_ax650.wav](outputs/en_long_paragraph_ax650.wav)
164
  提示音:[assets/moss_prompts/en_4_4p5s.wav](assets/moss_prompts/en_4_4p5s.wav)
165
 
166
+ ### AX650 ZipVoice Distill
167
 
168
  中文句子:
169
 
 
253
  音频:[outputs/en_long_paragraph_distill_ax650.wav](outputs/en_long_paragraph_distill_ax650.wav)
254
  提示音:[assets/moss_prompts/en_4_4p5s.wav](assets/moss_prompts/en_4_4p5s.wav)
255
 
 
 
 
 
 
 
 
 
 
 
 
256
 
257
+ ## 参数说明
258
 
259
+ - `--model-name`:选择模型目录。可选 `zipvoice_ax650``zipvoice_distill_ax650``zipvoice_distill_ax630C`。
260
  - `--prompt-wav`:参考音频,用于控制音色,建议 3-5s。
261
  - `--prompt-text`:参考音频对应文本,必须尽量和 `prompt-wav` 内容一致。
262
  - `--num-step`:采样步数。默认从模型目录的 `runtime_config.json` 读取。
263
  - `--max-feat-len`:decoder 固定 feature 长度,当前模型均为 1024。
264
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
265
  ## 参考
266
 
267
  - [ZipVoice](https://github.com/k2-fsa/ZipVoice)
 
 
bin/zipvoice_ax630c DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:adbc16b5766ca2eede9b14b405d5b534e03b66b29043d179b41aa0a5df77d721
3
- size 527832
 
 
 
 
bin/zipvoice_ax650 DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:8653e75af7dc2571df341a6172a2342c19f6c99761ae640261c798c86087fe67
3
- size 527848
 
 
 
 
bin/zipvoice_axcl DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:de730337c6c5ecf11e805f7267447ff387cf5ee80131d1ad4875129586285fc2
3
- size 581640
 
 
 
 
configuration.json DELETED
@@ -1 +0,0 @@
1
- {"framework": "pytorch", "task": "others", "allow_remote": true}
 
 
infer_zipvoice_axera.py CHANGED
@@ -58,7 +58,6 @@ def parse_args() -> argparse.Namespace:
58
  help="Allow raw features_len up to this ratio of max_feat_len before splitting",
59
  )
60
  p.add_argument("--silence-ms", type=int, default=140)
61
- p.add_argument("--vocoder-model", default=None, help="Path to vocos axmodel. If set, use axmodel vocoder instead of PyTorch.")
62
  return p.parse_args()
63
 
64
 
@@ -159,13 +158,7 @@ def main() -> None:
159
  t_shift=args.t_shift,
160
  )
161
  logging.debug("Loading vocoder...")
162
- if args.vocoder_model:
163
- from scripts.common_infer import load_axmodel_vocoder, axmodel_vocoder_decode
164
- vocoder = load_axmodel_vocoder(args.vocoder_model)
165
- vocoder_is_axmodel = True
166
- else:
167
- vocoder = load_vocoder(repo_dir)
168
- vocoder_is_axmodel = False
169
  logging.info("模型加载完成")
170
 
171
  audios: list[np.ndarray] = []
@@ -190,16 +183,13 @@ def main() -> None:
190
  seed=args.seed + index - 1,
191
  )
192
 
193
- if vocoder_is_axmodel:
194
- from scripts.common_infer import axmodel_vocoder_decode
195
- audio = axmodel_vocoder_decode(
196
- vocoder, pred_features,
197
- feat_scale=args.feat_scale, target_rms=args.target_rms, prompt_rms=prompt_rms)
198
- else:
199
- audio = vocoder_decode_loaded(
200
- vocoder, pred_features,
201
- feat_scale=args.feat_scale, target_rms=args.target_rms,
202
- prompt_rms=prompt_rms)
203
  segment_sec = time.perf_counter() - t_segment_start
204
  audio_sec = len(audio) / 24000
205
 
 
58
  help="Allow raw features_len up to this ratio of max_feat_len before splitting",
59
  )
60
  p.add_argument("--silence-ms", type=int, default=140)
 
61
  return p.parse_args()
62
 
63
 
 
158
  t_shift=args.t_shift,
159
  )
160
  logging.debug("Loading vocoder...")
161
+ vocoder = load_vocoder(repo_dir)
 
 
 
 
 
 
162
  logging.info("模型加载完成")
163
 
164
  audios: list[np.ndarray] = []
 
183
  seed=args.seed + index - 1,
184
  )
185
 
186
+ audio = vocoder_decode_loaded(
187
+ vocoder,
188
+ pred_features,
189
+ feat_scale=args.feat_scale,
190
+ target_rms=args.target_rms,
191
+ prompt_rms=prompt_rms,
192
+ )
 
 
 
193
  segment_sec = time.perf_counter() - t_segment_start
194
  audio_sec = len(audio) / 24000
195
 
models/vocoder/vocos_full.axmodel DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:70a7a897f6beb0b6e8536413a0ea6ee2132bec1257e51208e80eca506bda4c40
3
- size 15413002
 
 
 
 
models/vocoder/vocos_full_ax630c.axmodel DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:4b704aa370e7875099615ebbb2772c24e7d97529df81f11b713ca04e4261903e
3
- size 16208038
 
 
 
 
models/zipvoice_distill_ax630C/decoder4_split_manifest.json CHANGED
@@ -1,104 +1,68 @@
1
  {
2
- "version": 1,
3
- "model_type": "zipvoice_distill",
4
- "encoder": {
5
- "name": "encoder_core",
6
- "file": "encoder.axmodel",
7
- "inputs": [
8
- "cat_tokens"
9
- ],
10
- "outputs": [
11
- "encoded"
12
- ]
13
- },
14
- "decoder_parts": [
15
- {
16
- "name": "fm_decoder_part0",
17
- "file": "decoder_part0.axmodel",
18
- "inputs": [
19
- "t",
20
- "x",
21
- "text_condition",
22
- "speech_condition",
23
- "guidance_scale",
24
- "padding_mask"
25
- ],
26
- "outputs": [
27
- "decoder_hidden_p0",
28
- "time_hidden"
29
- ]
30
- },
31
- {
32
- "name": "fm_decoder_part1",
33
- "file": "decoder_part1.axmodel",
34
- "inputs": [
35
- "decoder_hidden_p0",
36
- "time_hidden",
37
- "padding_mask"
38
- ],
39
- "outputs": [
40
- "decoder_hidden_p1"
41
- ]
42
- },
43
- {
44
- "name": "fm_decoder_part2",
45
- "file": "decoder_part2.axmodel",
46
- "inputs": [
47
- "decoder_hidden_p1",
48
- "time_hidden",
49
- "padding_mask"
50
- ],
51
- "outputs": [
52
- "decoder_hidden_p2"
53
- ]
54
- },
55
- {
56
- "name": "fm_decoder_part3_0",
57
- "file": "decoder_part3_0.axmodel",
58
- "inputs": [
59
- "decoder_hidden_p2",
60
- "time_hidden",
61
- "padding_mask"
62
- ],
63
- "outputs": [
64
- "/fm_decoder/4/0/bypass/Add_output_0"
65
- ]
66
  },
67
- {
68
- "name": "fm_decoder_part3_1",
69
- "file": "decoder_part3_1.axmodel",
70
- "inputs": [
71
- "/fm_decoder/4/0/bypass/Add_output_0",
72
- "time_hidden",
73
- "padding_mask"
74
- ],
75
- "outputs": [
76
- "/fm_decoder/4/1/bypass/Add_output_0"
77
- ]
78
- },
79
- {
80
- "name": "fm_decoder_part3_2",
81
- "file": "decoder_part3_2.axmodel",
82
- "inputs": [
83
- "/fm_decoder/4/1/bypass/Add_output_0",
84
- "time_hidden",
85
- "padding_mask"
86
- ],
87
- "outputs": [
88
- "/fm_decoder/4/2/bypass/Add_output_0"
89
- ]
90
- },
91
- {
92
- "name": "fm_decoder_part3_3",
93
- "file": "decoder_part3_3.axmodel",
94
- "inputs": [
95
- "/fm_decoder/4/2/bypass/Add_output_0",
96
- "time_hidden",
97
- "padding_mask"
98
- ],
99
- "outputs": [
100
- "v"
101
- ]
102
- }
103
- ]
104
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  {
2
+ "version": 1,
3
+ "model_type": "zipvoice_distill",
4
+ "encoder": {
5
+ "name": "encoder_core",
6
+ "file": "encoder.axmodel",
7
+ "inputs": [
8
+ "cat_tokens"
9
+ ],
10
+ "outputs": [
11
+ "encoded"
12
+ ]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
13
  },
14
+ "decoder_parts": [
15
+ {
16
+ "name": "fm_decoder_part0",
17
+ "file": "decoder_part0.axmodel",
18
+ "inputs": [
19
+ "t",
20
+ "x",
21
+ "text_condition",
22
+ "speech_condition",
23
+ "guidance_scale",
24
+ "padding_mask"
25
+ ],
26
+ "outputs": [
27
+ "decoder_hidden_p0",
28
+ "time_hidden"
29
+ ]
30
+ },
31
+ {
32
+ "name": "fm_decoder_part1",
33
+ "file": "decoder_part1.axmodel",
34
+ "inputs": [
35
+ "decoder_hidden_p0",
36
+ "time_hidden",
37
+ "padding_mask"
38
+ ],
39
+ "outputs": [
40
+ "decoder_hidden_p1"
41
+ ]
42
+ },
43
+ {
44
+ "name": "fm_decoder_part2",
45
+ "file": "decoder_part2.axmodel",
46
+ "inputs": [
47
+ "decoder_hidden_p1",
48
+ "time_hidden",
49
+ "padding_mask"
50
+ ],
51
+ "outputs": [
52
+ "decoder_hidden_p2"
53
+ ]
54
+ },
55
+ {
56
+ "name": "fm_decoder_part3",
57
+ "file": "decoder_part3.axmodel",
58
+ "inputs": [
59
+ "decoder_hidden_p2",
60
+ "time_hidden",
61
+ "padding_mask"
62
+ ],
63
+ "outputs": [
64
+ "v"
65
+ ]
66
+ }
67
+ ]
68
+ }
models/zipvoice_distill_ax630C/decoder_part0.axmodel CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2113f394363a575904669017de685b4d8d73693736a285a4ce34782eba1a4531
3
- size 38600656
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1f0bd9974d94f349c0001d624d3c2c7140295abbe9c12c0db3556d5759d845a8
3
+ size 42595736
models/zipvoice_distill_ax630C/decoder_part1.axmodel CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6afd89c302f971b10677f27093bf60e7cb8f81ec191cde5c43781954976142be
3
- size 32510786
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cfe296e53b7b8f3842aea1bbc60e48f946c0b5f0ae57c639faa4d31d75ff9c5e
3
+ size 31881170
models/zipvoice_distill_ax630C/decoder_part2.axmodel CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c883008c38b1ee727b8a68989dd4861d368a1cb593f4548e44fc6200dee89e6d
3
- size 35030020
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e95d65c96ba38c7a3cee6ecc558567a4d79e1803ba427749442690afc3b99da9
3
+ size 35101628
models/zipvoice_distill_ax630C/decoder_part3_0.axmodel DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:0b4d6a68bbc3286ed14bbe1d7f783bcc8cd559fe709151a99079d6558c1ce7d9
3
- size 10038370
 
 
 
 
models/zipvoice_distill_ax630C/decoder_part3_1.axmodel DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:744d3121d74670fec1f29f9a7ce5d158cb215e9b822464ad123fe8daeb8247f1
3
- size 10038193
 
 
 
 
models/zipvoice_distill_ax630C/decoder_part3_2.axmodel DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:719c53fb3e4e0949b893c25814f705d33b12cd442fb84434d00dc3867c846ac1
3
- size 10038201
 
 
 
 
models/zipvoice_distill_ax630C/decoder_part3_3.axmodel DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:7c44654f65f1398e611b02d9da2a23e931008ffa9ee6a53ebf027fd0da04bd17
3
- size 10004459
 
 
 
 
models/zipvoice_distill_ax630C/encoder.axmodel CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7193cbc302f3063522194b8287e1adfc250b83ab67c38c064bba74bef3eacb94
3
- size 7659503
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ffbd78ded4a05cf472f78a66e34d9c4dac4c5525f6a7b1d2cc5dbb4d96e7538f
3
+ size 8394155
run_ax630c.sh DELETED
@@ -1,51 +0,0 @@
1
- #!/bin/bash
2
- # ZipVoice AX630C 板端一键运行
3
- # 用法: bash run_ax630c.sh [zh|en] [sentence|paragraph]
4
-
5
- MODEL="${1:-zh}"
6
- MODE="${2:-sentence}"
7
-
8
- BIN="bin/zipvoice_ax630c"
9
- MODEL_DIR="./models/zipvoice_distill_ax630C"
10
- TOKEN="./resources/zipvoice_hf/zipvoice/tokens.txt"
11
- VOCODER="./models/vocoder/vocos_full_ax630c.axmodel"
12
-
13
- cd "$(dirname "$0")"
14
-
15
- if [ "$MODEL" = "en" ]; then
16
- PWAV="./assets/moss_prompts/en_4_4p5s.wav"
17
- PTEXT="This is almost twice the current industry production level per train."
18
- DTEXT="This morning, a small train left the station, carrying sleepy passengers toward a bright coastal town."
19
- TFILE="./assets/paragraphs/en_scavenger.txt"
20
- REPO="--repo-dir ."
21
- else
22
- PWAV="./assets/moss_prompts/zh_1_4p5s.wav"
23
- PTEXT="不管怎么样我和汤姆还是要感谢贝尔卡金的援手"
24
- DTEXT="今天午后天气很好,我打开窗户,听见远处有人聊天,水杯也轻轻晃了一下。"
25
- TFILE="./assets/paragraphs/zh_ginkgo.txt"
26
- REPO=""
27
- fi
28
-
29
- OUT="output_ax630c_${MODEL}_${MODE}.wav"
30
-
31
- echo "=========================================="
32
- echo " ZipVoice AX630C"
33
- echo "=========================================="
34
- echo " MODEL: $MODEL_DIR"
35
- echo " LANG: $MODEL MODE: $MODE"
36
- echo " OUTPUT: $OUT"
37
- echo "=========================================="
38
-
39
- if [ "$MODE" = "paragraph" ]; then
40
- exec $BIN \
41
- --model-dir "$MODEL_DIR" --token-file "$TOKEN" \
42
- --prompt-wav "$PWAV" --prompt-text "$PTEXT" \
43
- --text-file "$TFILE" \
44
- --vocoder-model "$VOCODER" --output-wav "$OUT" --seed 42 $REPO
45
- else
46
- exec $BIN \
47
- --model-dir "$MODEL_DIR" --token-file "$TOKEN" \
48
- --prompt-wav "$PWAV" --prompt-text "$PTEXT" \
49
- --text "$DTEXT" \
50
- --vocoder-model "$VOCODER" --output-wav "$OUT" --seed 42 $REPO
51
- fi
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
run_ax650.sh DELETED
@@ -1,57 +0,0 @@
1
- #!/bin/bash
2
- # ZipVoice AXERA 板端一键运行
3
- # 用法: bash run_ax650.sh [standard|distill] [zh|en] [sentence|paragraph]
4
-
5
- MODEL="${1:-distill}"
6
- LANG="${2:-zh}"
7
- MODE="${3:-sentence}"
8
-
9
- BIN="bin/zipvoice_ax650"
10
- TOKEN="./resources/zipvoice_hf/zipvoice/tokens.txt"
11
- VOCODER="./models/vocoder/vocos_full.axmodel"
12
-
13
- cd "$(dirname "$0")"
14
-
15
- if [ "$MODEL" = "distill" ]; then
16
- MODEL_DIR="./models/zipvoice_distill_ax650"
17
- else
18
- MODEL_DIR="./models/zipvoice_ax650"
19
- fi
20
-
21
- if [ "$LANG" = "en" ]; then
22
- PWAV="./assets/moss_prompts/en_4_4p5s.wav"
23
- PTEXT="This is almost twice the current industry production level per train."
24
- DTEXT="This morning, a small train left the station, carrying sleepy passengers toward a bright coastal town."
25
- TFILE="./assets/paragraphs/en_scavenger.txt"
26
- REPO="--repo-dir ."
27
- else
28
- PWAV="./assets/moss_prompts/zh_1_4p5s.wav"
29
- PTEXT="不管怎么样我和汤姆还是要感谢贝尔卡金的援手"
30
- DTEXT="今天午后天气很好,我打开窗户,听见远处有人聊天,水杯也轻轻晃了一下。"
31
- TFILE="./assets/paragraphs/zh_ginkgo.txt"
32
- REPO=""
33
- fi
34
-
35
- OUT="output_${LANG}_${MODEL}_${MODE}.wav"
36
-
37
- echo "=========================================="
38
- echo " ZipVoice AX650"
39
- echo "=========================================="
40
- echo " MODEL: $MODEL_DIR"
41
- echo " LANG: $LANG MODE: $MODE"
42
- echo " OUTPUT: $OUT"
43
- echo "=========================================="
44
-
45
- if [ "$MODE" = "paragraph" ]; then
46
- exec $BIN \
47
- --model-dir "$MODEL_DIR" --token-file "$TOKEN" \
48
- --prompt-wav "$PWAV" --prompt-text "$PTEXT" \
49
- --text-file "$TFILE" \
50
- --vocoder-model "$VOCODER" --output-wav "$OUT" --seed 42 $REPO
51
- else
52
- exec $BIN \
53
- --model-dir "$MODEL_DIR" --token-file "$TOKEN" \
54
- --prompt-wav "$PWAV" --prompt-text "$PTEXT" \
55
- --text "$DTEXT" \
56
- --vocoder-model "$VOCODER" --output-wav "$OUT" --seed 42 $REPO
57
- fi
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
run_axcl.sh DELETED
@@ -1,57 +0,0 @@
1
- #!/bin/bash
2
- # ZipVoice AXCL 算力卡一键运行
3
- # 用法: bash run_axcl.sh [standard|distill] [zh|en] [sentence|paragraph]
4
-
5
- MODEL="${1:-distill}"
6
- LANG="${2:-zh}"
7
- MODE="${3:-sentence}"
8
-
9
- BIN="bin/zipvoice_axcl"
10
- TOKEN="./resources/zipvoice_hf/zipvoice/tokens.txt"
11
- VOCODER="./models/vocoder/vocos_full.axmodel"
12
-
13
- cd "$(dirname "$0")"
14
-
15
- if [ "$MODEL" = "distill" ]; then
16
- MODEL_DIR="./models/zipvoice_distill_ax650"
17
- else
18
- MODEL_DIR="./models/zipvoice_ax650"
19
- fi
20
-
21
- if [ "$LANG" = "en" ]; then
22
- PWAV="./assets/moss_prompts/en_4_4p5s.wav"
23
- PTEXT="This is almost twice the current industry production level per train."
24
- DTEXT="This morning, a small train left the station, carrying sleepy passengers toward a bright coastal town."
25
- TFILE="./assets/paragraphs/en_scavenger.txt"
26
- REPO="--repo-dir ."
27
- else
28
- PWAV="./assets/moss_prompts/zh_1_4p5s.wav"
29
- PTEXT="不管怎么样我和汤姆还是要感谢贝尔卡金的援手"
30
- DTEXT="今天午后天气很好,我打开窗户,听见远处有人聊天,水杯也轻轻晃了一下。"
31
- TFILE="./assets/paragraphs/zh_ginkgo.txt"
32
- REPO=""
33
- fi
34
-
35
- OUT="output_${LANG}_${MODEL}_${MODE}.wav"
36
-
37
- echo "=========================================="
38
- echo " ZipVoice AXCL"
39
- echo "=========================================="
40
- echo " MODEL: $MODEL_DIR"
41
- echo " LANG: $LANG MODE: $MODE"
42
- echo " OUTPUT: $OUT"
43
- echo "=========================================="
44
-
45
- if [ "$MODE" = "paragraph" ]; then
46
- exec $BIN \
47
- --model-dir "$MODEL_DIR" --token-file "$TOKEN" \
48
- --prompt-wav "$PWAV" --prompt-text "$PTEXT" \
49
- --text-file "$TFILE" \
50
- --vocoder-model "$VOCODER" --output-wav "$OUT" --seed 42 $REPO
51
- else
52
- exec $BIN \
53
- --model-dir "$MODEL_DIR" --token-file "$TOKEN" \
54
- --prompt-wav "$PWAV" --prompt-text "$PTEXT" \
55
- --text "$DTEXT" \
56
- --vocoder-model "$VOCODER" --output-wav "$OUT" --seed 42 $REPO
57
- fi
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
scripts/__pycache__/__init__.cpython-310.pyc DELETED
Binary file (155 Bytes)
 
scripts/__pycache__/local_tokenizer.cpython-310.pyc DELETED
Binary file (7.28 kB)
 
scripts/__pycache__/text_processing.cpython-310.pyc DELETED
Binary file (8.46 kB)
 
scripts/common_infer.py CHANGED
@@ -66,74 +66,3 @@ def vocoder_decode_loaded(
66
  if prompt_rms < target_rms:
67
  wav = wav * prompt_rms / target_rms
68
  return wav.squeeze().cpu().numpy()
69
-
70
-
71
- def load_axmodel_vocoder(model_path: str):
72
- """Load vocoder axmodel for inference."""
73
- import axengine as axe
74
-
75
- session = axe.InferenceSession(model_path)
76
- return session
77
-
78
-
79
- def axmodel_vocoder_decode(session, features: np.ndarray, feat_scale: float,
80
- target_rms: float, prompt_rms: float) -> np.ndarray:
81
- """Decode mel features to audio using axmodel vocoder + IRFFT."""
82
- import math
83
-
84
- # features shape: (1, T, 100) or (T, 100), squeeze batch dim
85
- features = np.squeeze(features)
86
- if features.ndim == 2:
87
- features = features.T # (T, 100) → (100, T)
88
- T = features.shape[1]
89
- n_mels = features.shape[0]
90
- n_fft, hop = 1024, 256
91
- T_model = 620
92
-
93
- # Undo feat_scale and pad to [1, n_mels, T_model]
94
- inv_scale = 1.0 / feat_scale if feat_scale != 0 else 1.0
95
- mel_input = np.zeros((1, n_mels, T_model), dtype=np.float32)
96
- t_actual = min(T, T_model)
97
- for t in range(t_actual):
98
- mel_input[0, :, t] = features[:, t] * inv_scale
99
-
100
- # Run axmodel: mel → (real, imag)
101
- real, imag = session.run(None, {"mel": mel_input})
102
- real, imag = real.squeeze(0), imag.squeeze(0) # [T_model, n_freqs]
103
-
104
- # IRFFT + overlap-add (matching C++ vocoder)
105
- n_freqs = n_fft // 2 + 1
106
- window = 0.5 * (1.0 - np.cos(2.0 * math.pi * np.arange(n_fft) / (n_fft - 1)))
107
- window_sq = window ** 2
108
- out_len = (T - 1) * hop + n_fft
109
- audio = np.zeros(out_len, dtype=np.float32)
110
- envelope = np.zeros(out_len, dtype=np.float32)
111
-
112
- for t_idx in range(T):
113
- spec = np.zeros(n_fft, dtype=np.complex64)
114
- spec[0] = real[t_idx, 0]
115
- for k in range(1, n_freqs - 1):
116
- spec[k] = real[t_idx, k] + 1j * imag[t_idx, k]
117
- spec[n_fft - k] = real[t_idx, k] - 1j * imag[t_idx, k]
118
- spec[n_freqs - 1] = real[t_idx, n_freqs - 1]
119
- ifft_out = np.fft.irfft(spec, n=n_fft).real
120
-
121
- pos = t_idx * hop
122
- for n in range(n_fft):
123
- p = pos + n
124
- if p < out_len:
125
- audio[p] += ifft_out[n] * window[n]
126
- envelope[p] += window_sq[n]
127
-
128
- audio /= np.maximum(envelope, 1e-10)
129
- pad = n_fft // 2
130
- audio = audio[pad:out_len - pad].astype(np.float32)
131
-
132
- # RMS normalize (numpy version)
133
- rms = np.sqrt(np.mean(audio ** 2))
134
- if rms < target_rms and rms > 1e-10:
135
- audio = audio * (target_rms / rms)
136
- if prompt_rms < target_rms:
137
- audio = audio * (prompt_rms / target_rms)
138
- return audio
139
-
 
66
  if prompt_rms < target_rms:
67
  wav = wav * prompt_rms / target_rms
68
  return wav.squeeze().cpu().numpy()
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
scripts/zipvoice_decoder4_runtime.py CHANGED
@@ -276,7 +276,6 @@ class Decoder4ZipVoiceBoardRuntime:
276
  encoded = self.run_encoder(cat_tokens)
277
  t_enc = time.perf_counter() - t_start
278
  logging.debug(" encoder: %.3f s (output shape=%s)", t_enc, encoded.shape)
279
- # DEBUG: dump intermediates
280
 
281
  t_start = time.perf_counter()
282
  text_condition, features_len = self.duration_expand(
@@ -288,7 +287,6 @@ class Decoder4ZipVoiceBoardRuntime:
288
  )
289
  t_dur = time.perf_counter() - t_start
290
  logging.debug(" duration_expand: %.3f s (features_len=%d)", t_dur, features_len)
291
- # DEBUG: dump
292
 
293
  seq_len = self.decoder_seq_len or self.max_feat_len
294
  if features_len > seq_len:
 
276
  encoded = self.run_encoder(cat_tokens)
277
  t_enc = time.perf_counter() - t_start
278
  logging.debug(" encoder: %.3f s (output shape=%s)", t_enc, encoded.shape)
 
279
 
280
  t_start = time.perf_counter()
281
  text_condition, features_len = self.duration_expand(
 
287
  )
288
  t_dur = time.perf_counter() - t_start
289
  logging.debug(" duration_expand: %.3f s (features_len=%d)", t_dur, features_len)
 
290
 
291
  seq_len = self.decoder_seq_len or self.max_feat_len
292
  if features_len > seq_len:
scripts/zipvoice_decoder4_runtime_encoder_onnx.py DELETED
@@ -1,87 +0,0 @@
1
- #!/usr/bin/env python3
2
-
3
- from __future__ import annotations
4
-
5
- import logging
6
- from pathlib import Path
7
- from typing import Dict, List
8
-
9
- import numpy as np
10
- import onnxruntime as ort
11
-
12
- from scripts.zipvoice_decoder4_runtime import Decoder4ZipVoiceBoardRuntime
13
- from scripts.zipvoice_runtime import AxeSession
14
-
15
-
16
- class _OrtSession:
17
- """ONNX Runtime wrapper with the AxeSession interface used by split runtime."""
18
-
19
- _TYPE_TO_DTYPE = {
20
- "tensor(float)": np.float32,
21
- "tensor(float32)": np.float32,
22
- "tensor(double)": np.float64,
23
- "tensor(int32)": np.int32,
24
- "tensor(int64)": np.int64,
25
- "tensor(uint8)": np.uint8,
26
- "tensor(bool)": np.bool_,
27
- }
28
-
29
- def __init__(self, model_path: str | Path):
30
- self.path = Path(model_path)
31
- if not self.path.exists():
32
- raise FileNotFoundError(f"ONNX model not found: {self.path}")
33
- self._session = ort.InferenceSession(
34
- str(self.path),
35
- providers=["CPUExecutionProvider"],
36
- )
37
- self._inputs = self._session.get_inputs()
38
- self._outputs = self._session.get_outputs()
39
-
40
- @property
41
- def input_names(self) -> List[str]:
42
- return [item.name for item in self._inputs]
43
-
44
- @property
45
- def output_names(self) -> List[str]:
46
- return [item.name for item in self._outputs]
47
-
48
- def _coerce_one(self, name: str, value: np.ndarray) -> np.ndarray:
49
- info = next((item for item in self._inputs if item.name == name), None)
50
- array = np.asarray(value)
51
- if info is None:
52
- return np.ascontiguousarray(array)
53
-
54
- dtype = self._TYPE_TO_DTYPE.get(str(info.type).lower())
55
- if dtype is not None:
56
- array = array.astype(dtype, copy=False)
57
-
58
- if list(info.shape) == [] and array.shape == (1,):
59
- array = array.reshape(())
60
-
61
- return np.ascontiguousarray(array)
62
-
63
- def run(self, feed_dict: Dict[str, np.ndarray]) -> Dict[str, np.ndarray]:
64
- feed = {name: self._coerce_one(name, value) for name, value in feed_dict.items()}
65
- outputs = self._session.run(None, feed)
66
- return {name: value for name, value in zip(self.output_names, outputs)}
67
-
68
-
69
- class Decoder4ZipVoiceBoardRuntimeEncoderOnnx(Decoder4ZipVoiceBoardRuntime):
70
- """Runs encoder with ONNX Runtime, all decoder parts with axmodel."""
71
-
72
- def _load_models(self) -> None:
73
- self.sessions = {}
74
-
75
- encoder_name = self.encoder_info["name"]
76
- encoder_path = self.models_dir / "encoder_core.onnx"
77
- logging.info("encoder 使用 ONNX Runtime: %s", encoder_path)
78
- self.sessions[encoder_name] = _OrtSession(encoder_path)
79
-
80
- for info in self.decoder_parts:
81
- name = info["name"]
82
- path = self.models_dir / info["file"]
83
- logging.debug("Loading %s from %s", name, path)
84
- self.sessions[name] = AxeSession(path)
85
-
86
- self.decoder_label = "decoder4(encoder_onnx+part0-3_axmodel)"
87
- logging.debug("Loaded encoder ONNX + %d decoder axmodels", len(self.decoder_parts))
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
scripts/zipvoice_decoder4_runtime_part0_onnx.py DELETED
@@ -1,94 +0,0 @@
1
- #!/usr/bin/env python3
2
-
3
- from __future__ import annotations
4
-
5
- import logging
6
- from pathlib import Path
7
- from typing import Dict, List
8
-
9
- import numpy as np
10
- import onnxruntime as ort
11
-
12
- from scripts.zipvoice_decoder4_runtime import Decoder4ZipVoiceBoardRuntime
13
- from scripts.zipvoice_runtime import AxeSession
14
-
15
-
16
- class _OrtSession:
17
- """ONNX Runtime wrapper with the small AxeSession interface used by the board runtime."""
18
-
19
- _TYPE_TO_DTYPE = {
20
- "tensor(float)": np.float32,
21
- "tensor(float32)": np.float32,
22
- "tensor(double)": np.float64,
23
- "tensor(int32)": np.int32,
24
- "tensor(int64)": np.int64,
25
- "tensor(uint8)": np.uint8,
26
- "tensor(bool)": np.bool_,
27
- }
28
-
29
- def __init__(self, model_path: str | Path):
30
- self.path = Path(model_path)
31
- if not self.path.exists():
32
- raise FileNotFoundError(f"ONNX model not found: {self.path}")
33
- self._session = ort.InferenceSession(
34
- str(self.path),
35
- providers=["CPUExecutionProvider"],
36
- )
37
- self._inputs = self._session.get_inputs()
38
- self._outputs = self._session.get_outputs()
39
-
40
- @property
41
- def input_names(self) -> List[str]:
42
- return [item.name for item in self._inputs]
43
-
44
- @property
45
- def output_names(self) -> List[str]:
46
- return [item.name for item in self._outputs]
47
-
48
- def _coerce_one(self, name: str, value: np.ndarray) -> np.ndarray:
49
- info = next((item for item in self._inputs if item.name == name), None)
50
- array = np.asarray(value)
51
- if info is None:
52
- return np.ascontiguousarray(array)
53
-
54
- dtype = self._TYPE_TO_DTYPE.get(str(info.type).lower())
55
- if dtype is not None:
56
- array = array.astype(dtype, copy=False)
57
-
58
- # The exported part0 ONNX keeps t/guidance_scale as scalar inputs ([]),
59
- # while the axmodel path feeds them as shape [1]. Normalize only for ONNX.
60
- if list(info.shape) == [] and array.shape == (1,):
61
- array = array.reshape(())
62
-
63
- return np.ascontiguousarray(array)
64
-
65
- def run(self, feed_dict: Dict[str, np.ndarray]) -> Dict[str, np.ndarray]:
66
- feed = {name: self._coerce_one(name, value) for name, value in feed_dict.items()}
67
- outputs = self._session.run(None, feed)
68
- return {name: value for name, value in zip(self.output_names, outputs)}
69
-
70
-
71
- class Decoder4ZipVoiceBoardRuntimePart0Onnx(Decoder4ZipVoiceBoardRuntime):
72
- """Runs decoder part0 with ONNX Runtime, all other split models with axmodel."""
73
-
74
- def _load_models(self) -> None:
75
- self.sessions = {}
76
-
77
- encoder_name = self.encoder_info["name"]
78
- encoder_path = self.models_dir / self.encoder_info["file"]
79
- logging.debug("Loading %s from %s", encoder_name, encoder_path)
80
- self.sessions[encoder_name] = AxeSession(encoder_path)
81
-
82
- for index, info in enumerate(self.decoder_parts):
83
- name = info["name"]
84
- if index == 0:
85
- path = self.models_dir / "fm_decoder_part0.onnx"
86
- logging.info("part0 使用 ONNX Runtime: %s", path)
87
- self.sessions[name] = _OrtSession(path)
88
- else:
89
- path = self.models_dir / info["file"]
90
- logging.debug("Loading %s from %s", name, path)
91
- self.sessions[name] = AxeSession(path)
92
-
93
- self.decoder_label = "decoder4(part0_onnx+part1-3_axmodel)"
94
- logging.debug("Loaded encoder axmodel + part0 ONNX + %d decoder axmodels", len(self.decoder_parts) - 1)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
scripts/zipvoice_decoder4_runtime_part3_onnx.py DELETED
@@ -1,93 +0,0 @@
1
- #!/usr/bin/env python3
2
-
3
- from __future__ import annotations
4
-
5
- import logging
6
- from pathlib import Path
7
- from typing import Dict, List
8
-
9
- import numpy as np
10
- import onnxruntime as ort
11
-
12
- from scripts.zipvoice_decoder4_runtime import Decoder4ZipVoiceBoardRuntime
13
- from scripts.zipvoice_runtime import AxeSession
14
-
15
-
16
- class _OrtSession:
17
- """ONNX Runtime wrapper with the AxeSession interface used by split runtime."""
18
-
19
- _TYPE_TO_DTYPE = {
20
- "tensor(float)": np.float32,
21
- "tensor(float32)": np.float32,
22
- "tensor(double)": np.float64,
23
- "tensor(int32)": np.int32,
24
- "tensor(int64)": np.int64,
25
- "tensor(uint8)": np.uint8,
26
- "tensor(bool)": np.bool_,
27
- }
28
-
29
- def __init__(self, model_path: str | Path):
30
- self.path = Path(model_path)
31
- if not self.path.exists():
32
- raise FileNotFoundError(f"ONNX model not found: {self.path}")
33
- self._session = ort.InferenceSession(
34
- str(self.path),
35
- providers=["CPUExecutionProvider"],
36
- )
37
- self._inputs = self._session.get_inputs()
38
- self._outputs = self._session.get_outputs()
39
-
40
- @property
41
- def input_names(self) -> List[str]:
42
- return [item.name for item in self._inputs]
43
-
44
- @property
45
- def output_names(self) -> List[str]:
46
- return [item.name for item in self._outputs]
47
-
48
- def _coerce_one(self, name: str, value: np.ndarray) -> np.ndarray:
49
- info = next((item for item in self._inputs if item.name == name), None)
50
- array = np.asarray(value)
51
- if info is None:
52
- return np.ascontiguousarray(array)
53
-
54
- dtype = self._TYPE_TO_DTYPE.get(str(info.type).lower())
55
- if dtype is not None:
56
- array = array.astype(dtype, copy=False)
57
-
58
- if list(info.shape) == [] and array.shape == (1,):
59
- array = array.reshape(())
60
-
61
- return np.ascontiguousarray(array)
62
-
63
- def run(self, feed_dict: Dict[str, np.ndarray]) -> Dict[str, np.ndarray]:
64
- feed = {name: self._coerce_one(name, value) for name, value in feed_dict.items()}
65
- outputs = self._session.run(None, feed)
66
- return {name: value for name, value in zip(self.output_names, outputs)}
67
-
68
-
69
- class Decoder4ZipVoiceBoardRuntimePart3Onnx(Decoder4ZipVoiceBoardRuntime):
70
- """Runs decoder part3 with ONNX Runtime, encoder and part0-2 with axmodel."""
71
-
72
- def _load_models(self) -> None:
73
- self.sessions = {}
74
-
75
- encoder_name = self.encoder_info["name"]
76
- encoder_path = self.models_dir / self.encoder_info["file"]
77
- logging.debug("Loading %s from %s", encoder_name, encoder_path)
78
- self.sessions[encoder_name] = AxeSession(encoder_path)
79
-
80
- last_index = len(self.decoder_parts) - 1
81
- for index, info in enumerate(self.decoder_parts):
82
- name = info["name"]
83
- if index == last_index:
84
- path = self.models_dir / "fm_decoder_part3.onnx"
85
- logging.info("part3 使用 ONNX Runtime: %s", path)
86
- self.sessions[name] = _OrtSession(path)
87
- else:
88
- path = self.models_dir / info["file"]
89
- logging.debug("Loading %s from %s", name, path)
90
- self.sessions[name] = AxeSession(path)
91
-
92
- self.decoder_label = "decoder4(part0-2_axmodel+part3_onnx)"
93
- logging.debug("Loaded encoder axmodel + %d decoder axmodels + part3 ONNX", last_index)