conda activate torch280_py310_ali
# 关键:transformers 必须使用 4.51.3,高版本会导致音频失真
pip install transformers==4.51.3 --no-deps
pip install "tokenizers>=0.21,<0.22" --no-deps
pip install "huggingface-hub>=0.30.0,<1.0" --no-depsunset https_proxy http_proxy ALL_PROXY
python -c "from modelscope import snapshot_download; snapshot_download('FunAudioLLM/Fun-CosyVoice3-0.5B-2512')"export ASCEND_RT_VISIBLE_DEVICES=6,7
cd 01_CosyVoice
python run_npu.py输出音频保存在 output_npu/ 目录下。
- zero-shot TTS(中文 + 英文)
- cross-lingual TTS
- instruct TTS
- fp16 autocast 推理
- transformers 版本: 必须使用 4.51.3,5.x 版本会导致 Qwen2 LLM 生成的 speech tokens 失真
- iSTFT: 用
torch.fft.irfft+scatter_add替代torch.istft(NPU 不支持 fold 算子),全程在 NPU 上执行 - f0_predictor: 从原始 fp64 改为 fp32,全程 NPU 执行(实测精度无损)
- 推理性能: fp16 模式下 warmup 后 RTF ≈ 1.7-1.9
- empty_cache: NPU 的 empty_cache 默认关闭以避免同步开销,需要时设置
export COSYVOICE_NPU_EMPTY_CACHE=1