Views
No views yet
| Training Loss | Epoch | Step | Validation Loss | Wer |
|---|---|---|---|---|
| 0.6739 | 2.8329 | 2000 | 0.5742 | 0.4900 |
| 0.4767 | 5.6657 | 4000 | 0.4601 | 0.4101 |
| 0.3889 | 8.4986 | 6000 | 0.3921 | 0.3329 |
| 0.3381 | 11.3314 | 8000 | 0.3323 | 0.3081 |
| 0.2842 | 14.1643 | 10000 | 0.3467 | 0.3081 |
| 0.2505 | 16.9972 | 12000 | 0.3186 | 0.2833 |
| 0.2158 | 19.8300 | 14000 | 0.3003 | 0.2522 |
| 0.1885 | 22.6629 | 16000 | 0.2877 | 0.2405 |
| 0.1695 | 25.4958 | 18000 | 0.3089 | 0.2405 |
| 0.1494 | 28.3286 | 20000 | 0.2924 | 0.2254 |
| 0.1331 | 31.1615 | 22000 | 0.2796 | 0.2068 |
| 0.1293 | 33.9943 | 24000 | 0.2734 | 0.1895 |
| 0.1083 | 36.8272 | 26000 | 0.2844 | 0.1826 |
| 0.0955 | 39.6601 | 28000 | 0.2665 | 0.1744 |
| 0.085 | 42.4929 | 30000 | 0.2772 | 0.1695 |
| 0.0799 | 45.3258 | 32000 | 0.2747 | 0.1654 |
| 0.072 | 48.1586 | 34000 | 0.2746 | 0.1558 |
| 0.0934 | 50.9915 | 36000 | 0.2979 | 0.1764 |
| 0.0912 | 53.8244 | 38000 | 0.2914 | 0.1778 |
| 0.0812 | 56.6572 | 40000 | 0.2762 | 0.1785 |
| 0.0779 | 59.4901 | 42000 | 0.2752 | 0.1688 |
| 0.0718 | 62.3229 | 44000 | 0.2623 | 0.1633 |
| 0.0656 | 65.1558 | 46000 | 0.2704 | 0.1647 |
| 0.0606 | 67.9887 | 48000 | 0.2632 | 0.1571 |
| 0.0564 | 70.8215 | 50000 | 0.2711 | 0.1551 |
| 0.0562 | 73.6544 | 52000 | 0.2727 | 0.1523 |
1import torch
2import torchaudio
3import gradio as gr
4import numpy as np
5from transformers import pipeline, AutoProcessor, AutoModelForCTC
6
7device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
8
9# Load the model and processor
10MODEL_NAME = "<fill this to your model>"
11processor = AutoProcessor.from_pretrained(MODEL_NAME)
12model = AutoModelForCTC.from_pretrained(MODEL_NAME)
13
14# Move model to GPU
15model.to(device)
16
17# Create the pipeline with the model and processor
18transcriber = pipeline("automatic-speech-recognition", model=model, tokenizer=processor.tokenizer, feature_extractor=processor.feature_extractor, device=device)
19
20def transcribe(audio):
21 sr, y = audio
22 y = y.astype(np.float32)
23 y /= np.max(np.abs(y))
24
25 return transcriber({"sampling_rate": sr, "raw": y})["text"]
26
27demo = gr.Interface(
28 transcribe,
29 gr.Audio(sources=["upload"]),
30 "text",
31)
32
33demo.launch(share=True)1import torch
2import torchaudio
3import gradio as gr
4import numpy as np
5from transformers import pipeline, AutoProcessor, AutoModelForCTC
6
7device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
8
9# Load the model and processor
10MODEL_NAME = "<fill this to actual model>"
11processor = AutoProcessor.from_pretrained(MODEL_NAME)
12model = AutoModelForCTC.from_pretrained(MODEL_NAME)
13
14# Move model to GPU
15model.to(device)
16
17# Load audio file
18AUDIO_PATH = "<replace 'path_to_audio_file.wav' with the actual path to your audio file>"
19audio_input, sample_rate = torchaudio.load(AUDIO_PATH)
20
21# Ensure the audio is mono (1 channel)
22if audio_input.shape[0] > 1:
23 audio_input = torch.mean(audio_input, dim=0, keepdim=True)
24
25# Resample audio if necessary
26if sample_rate != 16000:
27 resampler = torchaudio.transforms.Resample(orig_freq=sample_rate, new_freq=16000)
28 audio_input = resampler(audio_input)
29
30# Process the audio input
31input_values = processor(audio_input.squeeze(), sampling_rate=16000, return_tensors="pt").input_values
32
33# Move input values to GPU
34input_values = input_values.to(device)
35
36# Perform inference
37with torch.no_grad():
38 logits = model(input_values).logits
39
40# Decode the logits to text
41predicted_ids = torch.argmax(logits, dim=-1)
42transcription = processor.batch_decode(predicted_ids)[0]
43
44print("Transcription:", transcription)