pip3 install -U funasr1git clone https://github.com/alibaba/FunASR.git && cd FunASR
2pip3 install -e ./pip3 install -U modelscope| Model Name | Task Details | Training Data | Parameters |
|---|---|---|---|
| paraformer-zh (⭐ 🤗 ) | speech recognition, with timestamps, non-streaming | 60000 hours, Mandarin | 220M |
( ⭐ 🤗 ) | speech recognition, streaming | 60000 hours, Mandarin | 220M |
| paraformer-en ( ⭐ 🤗 ) | speech recognition, with timestamps, non-streaming | 50000 hours, English | 220M |
| conformer-en ( ⭐ 🤗 ) | speech recognition, non-streaming | 50000 hours, English | 220M |
| ct-punc ( ⭐ 🤗 ) | punctuation restoration | 100M, Mandarin and English | 1.1G |
| fsmn-vad ( ⭐ 🤗 ) | voice activity detection | 5000 hours, Mandarin and English | 0.4M |
| fa-zh ( ⭐ 🤗 ) | timestamp prediction | 5000 hours, Mandarin | 38M |
| cam++ ( ⭐ 🤗 ) | speaker verification/diarization | 5000 hours | 7.2M |
funasr +model=paraformer-zh +vad_model="fsmn-vad" +punc_model="ct-punc" +input=asr_example_zh.wavwav_id wav_pat1from funasr import AutoModel
2# paraformer-zh is a multi-functional asr model
3# use vad, punc, spk or not as you need
4model = AutoModel(model="paraformer-zh", model_revision="v2.0.4",
5 vad_model="fsmn-vad", vad_model_revision="v2.0.4",
6 punc_model="ct-punc-c", punc_model_revision="v2.0.4",
7 # spk_model="cam++", spk_model_revision="v2.0.2",
8 )
9res = model.generate(input=f"{model.model_path}/example/asr_example.wav",
10 batch_size_s=300,
11 hotword='魔搭')
12print(res)model_hub: represents the model repository, ms stands for selecting ModelScope download, hf stands for selecting Huggingface download.1from funasr import AutoModel
2
3chunk_size = [0, 10, 5] #[0, 10, 5] 600ms, [0, 8, 4] 480ms
4encoder_chunk_look_back = 4 #number of chunks to lookback for encoder self-attention
5decoder_chunk_look_back = 1 #number of encoder chunks to lookback for decoder cross-attention
6
7model = AutoModel(model="paraformer-zh-streaming", model_revision="v2.0.4")
8
9import soundfile
10import os
11
12wav_file = os.path.join(model.model_path, "example/asr_example.wav")
13speech, sample_rate = soundfile.read(wav_file)
14chunk_stride = chunk_size[1] * 960 # 600ms
15
16cache = {}
17total_chunk_num = int(len((speech)-1)/chunk_stride+1)
18for i in range(total_chunk_num):
19 speech_chunk = speech[i*chunk_stride:(i+1)*chunk_stride]
20 is_final = i == total_chunk_num - 1
21 res = model.generate(input=speech_chunk, cache=cache, is_final=is_final, chunk_size=chunk_size, encoder_chunk_look_back=encoder_chunk_look_back, decoder_chunk_look_back=decoder_chunk_look_back)
22 print(res)chunk_size is the configuration for streaming latency. [0,10,5] indicates that the real-time display granularity is 10*60=600ms, and the lookahead information is 5*60=300ms. Each inference input is 600ms (sample points are 16000*0.6=960), and the output is the corresponding text. For the last speech segment input, is_final=True needs to be set to force the output of the last word.1from funasr import AutoModel
2
3model = AutoModel(model="fsmn-vad", model_revision="v2.0.4")
4wav_file = f"{model.model_path}/example/asr_example.wav"
5res = model.generate(input=wav_file)
6print(res)1from funasr import AutoModel
2
3chunk_size = 200 # ms
4model = AutoModel(model="fsmn-vad", model_revision="v2.0.4")
5
6import soundfile
7
8wav_file = f"{model.model_path}/example/vad_example.wav"
9speech, sample_rate = soundfile.read(wav_file)
10chunk_stride = int(chunk_size * sample_rate / 1000)
11
12cache = {}
13total_chunk_num = int(len((speech)-1)/chunk_stride+1)
14for i in range(total_chunk_num):
15 speech_chunk = speech[i*chunk_stride:(i+1)*chunk_stride]
16 is_final = i == total_chunk_num - 1
17 res = model.generate(input=speech_chunk, cache=cache, is_final=is_final, chunk_size=chunk_size)
18 if len(res[0]["value"]):
19 print(res)1from funasr import AutoModel
2
3model = AutoModel(model="ct-punc", model_revision="v2.0.4")
4res = model.generate(input="那今天的会就到这里吧 happy new year 明年见")
5print(res)1from funasr import AutoModel
2
3model = AutoModel(model="fa-zh", model_revision="v2.0.4")
4wav_file = f"{model.model_path}/example/asr_example.wav"
5text_file = f"{model.model_path}/example/text.txt"
6res = model.generate(input=(wav_file, text_file), data_type=("sound", "text"))
7print(res)