Fine-tuned
facebook/wav2vec2-xls-r-1b on French using the train and validation splits of
Common Voice 8.0,
MediaSpeech,
Multilingual TEDx,
Multilingual LibriSpeech, and
Voxpopuli.
When using this model, make sure that your speech input is sampled at 16kHz.
This model has been fine-tuned by the
HuggingSound tool, and thanks to the GPU credits generously given by the
OVHcloud :)
1from huggingsound import SpeechRecognitionModel
2
3model = SpeechRecognitionModel("jonatasgrosman/wav2vec2-xls-r-1b-french")
4audio_paths = ["/path/to/file.mp3", "/path/to/another_file.wav"]
5
6transcriptions = model.transcribe(audio_paths)
1import torch
2import librosa
3from datasets import load_dataset
4from transformers import Wav2Vec2ForCTC, Wav2Vec2Processor
5
6LANG_ID = "fr"
7MODEL_ID = "jonatasgrosman/wav2vec2-xls-r-1b-french"
8SAMPLES = 10
9
10test_dataset = load_dataset("common_voice", LANG_ID, split=f"test[:{SAMPLES}]")
11
12processor = Wav2Vec2Processor.from_pretrained(MODEL_ID)
13model = Wav2Vec2ForCTC.from_pretrained(MODEL_ID)
14
15# Preprocessing the datasets.
16# We need to read the audio files as arrays
17def speech_file_to_array_fn(batch):
18 speech_array, sampling_rate = librosa.load(batch["path"], sr=16_000)
19 batch["speech"] = speech_array
20 batch["sentence"] = batch["sentence"].upper()
21 return batch
22
23test_dataset = test_dataset.map(speech_file_to_array_fn)
24inputs = processor(test_dataset["speech"], sampling_rate=16_000, return_tensors="pt", padding=True)
25
26with torch.no_grad():
27 logits = model(inputs.input_values, attention_mask=inputs.attention_mask).logits
28
29predicted_ids = torch.argmax(logits, dim=-1)
30predicted_sentences = processor.batch_decode(predicted_ids)
1@misc{grosman2021xlsr-1b-french,
2 title={Fine-tuned {XLS-R} 1{B} model for speech recognition in {F}rench},
3 author={Grosman, Jonatas},
4 howpublished={\url{https://huggingface.co/jonatasgrosman/wav2vec2-xls-r-1b-french}},
5 year={2022}
6}