1git clone --recursive https://github.com/FunAudioLLM/CosyVoice.git
2# If you failed to clone submodule due to network failures, please run following command until success
3cd CosyVoice
4git submodule update --init --recursive
1conda create -n cosyvoice -y python=3.10
2conda activate cosyvoice
3# pynini is required by WeTextProcessing, use conda to install it as it can be executed on all platform.
4conda install -y -c conda-forge pynini==2.1.5
5pip install -r requirements.txt -i https://mirrors.aliyun.com/pypi/simple/ --trusted-host=mirrors.aliyun.com
6
7# If you encounter sox compatibility issues
8# ubuntu
9sudo apt-get install sox libsox-dev
10# centos
11sudo yum install sox sox-devel
1# SDK模型下载
2from modelscope import snapshot_download
3snapshot_download('iic/CosyVoice2-0.5B', local_dir='pretrained_models/CosyVoice2-0.5B')
4snapshot_download('iic/CosyVoice-300M', local_dir='pretrained_models/CosyVoice-300M')
5snapshot_download('iic/CosyVoice-300M-25Hz', local_dir='pretrained_models/CosyVoice-300M-25Hz')
6snapshot_download('iic/CosyVoice-300M-SFT', local_dir='pretrained_models/CosyVoice-300M-SFT')
7snapshot_download('iic/CosyVoice-300M-Instruct', local_dir='pretrained_models/CosyVoice-300M-Instruct')
8snapshot_download('iic/CosyVoice-ttsfrd', local_dir='pretrained_models/CosyVoice-ttsfrd')
1# git模型下载,请确保已安装git lfs
2mkdir -p pretrained_models
3git clone https://www.modelscope.cn/iic/CosyVoice2-0.5B.git pretrained_models/CosyVoice2-0.5B
4git clone https://www.modelscope.cn/iic/CosyVoice-300M.git pretrained_models/CosyVoice-300M
5git clone https://www.modelscope.cn/iic/CosyVoice-300M-25Hz.git pretrained_models/CosyVoice-300M-25Hz
6git clone https://www.modelscope.cn/iic/CosyVoice-300M-SFT.git pretrained_models/CosyVoice-300M-SFT
7git clone https://www.modelscope.cn/iic/CosyVoice-300M-Instruct.git pretrained_models/CosyVoice-300M-Instruct
8git clone https://www.modelscope.cn/iic/CosyVoice-ttsfrd.git pretrained_models/CosyVoice-ttsfrd
Notice that this step is not necessary. If you do not install ttsfrd package, we will use WeTextProcessing by default.
1cd pretrained_models/CosyVoice-ttsfrd/
2unzip resource.zip -d .
3pip install ttsfrd_dependency-0.1-py3-none-any.whl
4pip install ttsfrd-0.4.2-cp310-cp310-linux_x86_64.whl
1import sys
2sys.path.append('third_party/Matcha-TTS')
3from cosyvoice.cli.cosyvoice import CosyVoice, CosyVoice2
4from cosyvoice.utils.file_utils import load_wav
5import torchaudio
1cosyvoice = CosyVoice2('pretrained_models/CosyVoice2-0.5B', load_jit=False, load_trt=False, fp16=False)
2
3# NOTE if you want to reproduce the results on https://funaudiollm.github.io/cosyvoice2, please add text_frontend=False during inference
4# zero_shot usage
5prompt_speech_16k = load_wav('./asset/zero_shot_prompt.wav', 16000)
6for i, j in enumerate(cosyvoice.inference_zero_shot('收到好友从远方寄来的生日礼物,那份意外的惊喜与深深的祝福让我心中充满了甜蜜的快乐,笑容如花儿般绽放。', '希望你以后能够做的比我还好呦。', prompt_speech_16k, stream=False)):
7 torchaudio.save('zero_shot_{}.wav'.format(i), j['tts_speech'], cosyvoice.sample_rate)
8
9# fine grained control, for supported control, check cosyvoice/tokenizer/tokenizer.py#L248
10for i, j in enumerate(cosyvoice.inference_cross_lingual('在他讲述那个荒诞故事的过程中,他突然[laughter]停下来,因为他自己也被逗笑了[laughter]。', prompt_speech_16k, stream=False)):
11 torchaudio.save('fine_grained_control_{}.wav'.format(i), j['tts_speech'], cosyvoice.sample_rate)
12
13# instruct usage
14for i, j in enumerate(cosyvoice.inference_instruct2('收到好友从远方寄来的生日礼物,那份意外的惊喜与深深的祝福让我心中充满了甜蜜的快乐,笑容如花儿般绽放。', '用四川话说这句话', prompt_speech_16k, stream=False)):
15 torchaudio.save('instruct_{}.wav'.format(i), j['tts_speech'], cosyvoice.sample_rate)
16
17# bistream usage, you can use generator as input, this is useful when using text llm model as input
18# NOTE you should still have some basic sentence split logic because llm can not handle arbitrary sentence length
19def text_generator():
20 yield '收到好友从远方寄来的生日礼物,'
21 yield '那份意外的惊喜与深深的祝福'
22 yield '让我心中充满了甜蜜的快乐,'
23 yield '笑容如花儿般绽放。'
24for i, j in enumerate(cosyvoice.inference_zero_shot(text_generator(), '希望你以后能够做的比我还好呦。', prompt_speech_16k, stream=False)):
25 torchaudio.save('zero_shot_{}.wav'.format(i), j['tts_speech'], cosyvoice.sample_rate)
1cosyvoice = CosyVoice('pretrained_models/CosyVoice-300M-SFT', load_jit=False, load_trt=False, fp16=False)
2# sft usage
3print(cosyvoice.list_available_spks())
4# change stream=True for chunk stream inference
5for i, j in enumerate(cosyvoice.inference_sft('你好,我是通义生成式语音大模型,请问有什么可以帮您的吗?', '中文女', stream=False)):
6 torchaudio.save('sft_{}.wav'.format(i), j['tts_speech'], cosyvoice.sample_rate)
7
8cosyvoice = CosyVoice('pretrained_models/CosyVoice-300M') # or change to pretrained_models/CosyVoice-300M-25Hz for 25Hz inference
9# zero_shot usage, <|zh|><|en|><|jp|><|yue|><|ko|> for Chinese/English/Japanese/Cantonese/Korean
10prompt_speech_16k = load_wav('./asset/zero_shot_prompt.wav', 16000)
11for i, j in enumerate(cosyvoice.inference_zero_shot('收到好友从远方寄来的生日礼物,那份意外的惊喜与深深的祝福让我心中充满了甜蜜的快乐,笑容如花儿般绽放。', '希望你以后能够做的比我还好呦。', prompt_speech_16k, stream=False)):
12 torchaudio.save('zero_shot_{}.wav'.format(i), j['tts_speech'], cosyvoice.sample_rate)
13# cross_lingual usage
14prompt_speech_16k = load_wav('./asset/cross_lingual_prompt.wav', 16000)
15for i, j in enumerate(cosyvoice.inference_cross_lingual('<|en|>And then later on, fully acquiring that company. So keeping management in line, interest in line with the asset that\'s coming into the family is a reason why sometimes we don\'t buy the whole thing.', prompt_speech_16k, stream=False)):
16 torchaudio.save('cross_lingual_{}.wav'.format(i), j['tts_speech'], cosyvoice.sample_rate)
17# vc usage
18prompt_speech_16k = load_wav('./asset/zero_shot_prompt.wav', 16000)
19source_speech_16k = load_wav('./asset/cross_lingual_prompt.wav', 16000)
20for i, j in enumerate(cosyvoice.inference_vc(source_speech_16k, prompt_speech_16k, stream=False)):
21 torchaudio.save('vc_{}.wav'.format(i), j['tts_speech'], cosyvoice.sample_rate)
22
23cosyvoice = CosyVoice('pretrained_models/CosyVoice-300M-Instruct')
24# instruct usage, support <laughter></laughter><strong></strong>[laughter][breath]
25for i, j in enumerate(cosyvoice.inference_instruct('在面对挑战时,他展现了非凡的<strong>勇气</strong>与<strong>智慧</strong>。', '中文男', 'Theo \'Crimson\', is a fiery, passionate rebel leader. Fights with fervor for justice, but struggles with impulsiveness.', stream=False)):
26 torchaudio.save('instruct_{}.wav'.format(i), j['tts_speech'], cosyvoice.sample_rate)
You can use our web demo page to get familiar with CosyVoice quickly.
Please see the demo website for details.
1# change iic/CosyVoice-300M-SFT for sft inference, or iic/CosyVoice-300M-Instruct for instruct inference
2python3 webui.py --port 50000 --model_dir pretrained_models/CosyVoice-300M
Optionally, if you want service deployment,
you can run following steps.
1cd runtime/python
2docker build -t cosyvoice:v1.0 .
3# change iic/CosyVoice-300M to iic/CosyVoice-300M-Instruct if you want to use instruct inference
4# for grpc usage
5docker run -d --runtime=nvidia -p 50000:50000 cosyvoice:v1.0 /bin/bash -c "cd /opt/CosyVoice/CosyVoice/runtime/python/grpc && python3 server.py --port 50000 --max_conc 4 --model_dir iic/CosyVoice-300M && sleep infinity"
6cd grpc && python3 client.py --port 50000 --mode <sft|zero_shot|cross_lingual|instruct>
7# for fastapi usage
8docker run -d --runtime=nvidia -p 50000:50000 cosyvoice:v1.0 /bin/bash -c "cd /opt/CosyVoice/CosyVoice/runtime/python/fastapi && python3 server.py --port 50000 --model_dir iic/CosyVoice-300M && sleep infinity"
9cd fastapi && python3 client.py --port 50000 --mode <sft|zero_shot|cross_lingual|instruct>
You can directly discuss on
Github Issues.
You can also scan the QR code to join our official Dingding chat group.
The content provided above is for academic purposes only and is intended to demonstrate technical capabilities. Some examples are sourced from the internet. If any content infringes on your rights, please contact us to request its removal.