Views
No views yet
1import torch
2import torchaudio
3import gradio as gr
4import numpy as np
5from transformers import pipeline, AutoProcessor, AutoModelForCTC
6
7device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
8
9# Load the model and processor
10MODEL_NAME = "<fill this to your model>"
11processor = AutoProcessor.from_pretrained(MODEL_NAME)
12model = AutoModelForCTC.from_pretrained(MODEL_NAME)
13
14# Move model to GPU
15model.to(device)
16
17# Create the pipeline with the model and processor
18transcriber = pipeline("automatic-speech-recognition", model=model, tokenizer=processor.tokenizer, feature_extractor=processor.feature_extractor, device=device)
19
20def transcribe(audio):
21 sr, y = audio
22 y = y.astype(np.float32)
23 y /= np.max(np.abs(y))
24
25 return transcriber({"sampling_rate": sr, "raw": y})["text"]
26
27demo = gr.Interface(
28 transcribe,
29 gr.Audio(sources=["upload"]),
30 "text",
31)
32
33demo.launch(share=True)1import torch
2import torchaudio
3import gradio as gr
4import numpy as np
5from transformers import pipeline, AutoProcessor, AutoModelForCTC
6
7device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
8
9# Load the model and processor
10MODEL_NAME = "<fill this to actual model>"
11processor = AutoProcessor.from_pretrained(MODEL_NAME)
12model = AutoModelForCTC.from_pretrained(MODEL_NAME)
13
14# Move model to GPU
15model.to(device)
16
17# Load audio file
18AUDIO_PATH = "<replace 'path_to_audio_file.wav' with the actual path to your audio file>"
19audio_input, sample_rate = torchaudio.load(AUDIO_PATH)
20
21# Ensure the audio is mono (1 channel)
22if audio_input.shape[0] > 1:
23 audio_input = torch.mean(audio_input, dim=0, keepdim=True)
24
25# Resample audio if necessary
26if sample_rate != 16000:
27 resampler = torchaudio.transforms.Resample(orig_freq=sample_rate, new_freq=16000)
28 audio_input = resampler(audio_input)
29
30# Process the audio input
31input_values = processor(audio_input.squeeze(), sampling_rate=16000, return_tensors="pt").input_values
32
33# Move input values to GPU
34input_values = input_values.to(device)
35
36# Perform inference
37with torch.no_grad():
38 logits = model(input_values).logits
39
40# Decode the logits to text
41predicted_ids = torch.argmax(logits, dim=-1)
42transcription = processor.batch_decode(predicted_ids)[0]
43
44print("Transcription:", transcription)| Training Loss | Epoch | Step | Validation Loss | Wer |
|---|---|---|---|---|
| 0.5826 | 2.8329 | 2000 | 0.4733 | 0.4445 |
| 0.3478 | 5.6657 | 4000 | 0.3538 | 0.3191 |
| 0.2532 | 8.4986 | 6000 | 0.3085 | 0.2646 |
| 0.2028 | 11.3314 | 8000 | 0.2799 | 0.2467 |
| 0.1628 | 14.1643 | 10000 | 0.2623 | 0.2095 |
| 0.1407 | 16.9972 | 12000 | 0.2510 | 0.2068 |
| 0.1154 | 19.8300 | 14000 | 0.2922 | 0.1937 |
| 0.1044 | 22.6629 | 16000 | 0.2660 | 0.1730 |
| 0.0929 | 25.4958 | 18000 | 0.2818 | 0.1868 |
| 0.0798 | 28.3286 | 20000 | 0.2573 | 0.1633 |
| 0.074 | 31.1615 | 22000 | 0.2398 | 0.1647 |
| 0.0678 | 33.9943 | 24000 | 0.2601 | 0.1606 |
| 0.0628 | 36.8272 | 26000 | 0.2627 | 0.1613 |
| 0.057 | 39.6601 | 28000 | 0.2393 | 0.1468 |
| 0.0547 | 42.4929 | 30000 | 0.2662 | 0.1585 |
| 0.0512 | 45.3258 | 32000 | 0.2544 | 0.1502 |
| 0.0446 | 48.1586 | 34000 | 0.2542 | 0.1502 |
| 0.045 | 50.9915 | 36000 | 0.2624 | 0.1516 |
| 0.0403 | 53.8244 | 38000 | 0.2487 | 0.1420 |
| 0.0378 | 56.6572 | 40000 | 0.2498 | 0.1330 |
| 0.0353 | 59.4901 | 42000 | 0.2495 | 0.1309 |
| 0.0337 | 62.3229 | 44000 | 0.2505 | 0.1316 |
| 0.029 | 65.1558 | 46000 | 0.2373 | 0.1247 |
| 0.0277 | 67.9887 | 48000 | 0.2543 | 0.1282 |
| 0.0283 | 70.8215 | 50000 | 0.2547 | 0.1234 |
| 0.0275 | 73.6544 | 52000 | 0.2470 | 0.1227 |