Views
No views yet
1import torch
2import numpy as np
3import torchaudio
4from transformers import AutoModel, AutoFeatureExtractor
5from huggingface_hub import snapshot_download
6
7# Download all expert files
8repo_path = snapshot_download(repo_id="laion/Majestrino-1.00-voice-experts")
9
10# Load Majestrino encoder
11encoder = AutoModel.from_pretrained("laion/Majestrino-1.00", trust_remote_code=True)
12feature_extractor = AutoFeatureExtractor.from_pretrained("laion/Majestrino-1.00")
13encoder.eval()
14
15# Load audio file (will be resampled to 16kHz)
16audio_path = "your_audio.wav"
17waveform, sr = torchaudio.load(audio_path)
18if waveform.shape[0] > 1:
19 waveform = waveform.mean(dim=0, keepdim=True)
20if sr != 16000:
21 waveform = torchaudio.functional.resample(waveform, sr, 16000)
22
23# Extract Majestrino embedding
24with torch.no_grad():
25 inputs = feature_extractor(waveform.squeeze().numpy(), sampling_rate=16000, return_tensors="pt")
26 embedding = encoder(**inputs).last_hidden_state.mean(dim=1) # [1, 768]
27 embedding = torch.nn.functional.normalize(embedding, p=2, dim=-1)
28
29# Load speech detector
30speech_std = np.load(f"{repo_path}/speech_standardization.npz")
31speech_mean, speech_std = torch.tensor(speech_std['mean']), torch.tensor(speech_std['std'])
32speech_detector = torch.load(f"{repo_path}/speech_detector_best.pt", map_location='cpu')
33speech_detector.eval()
34
35# Check if audio contains speech
36emb_std = (embedding - speech_mean) / speech_std
37with torch.no_grad():
38 is_speech = torch.sigmoid(speech_detector(emb_std)).item() > 0.5
39
40print(f"Contains speech: {is_speech}")
41
42if is_speech:
43 # Load expert standardization
44 expert_std = np.load(f"{repo_path}/expert_standardization.npz")
45 expert_mean = torch.tensor(expert_std['mean'])
46 expert_std_val = torch.tensor(expert_std['std'])
47
48 # Load a few example experts
49 experts = {
50 'GEND': torch.load(f"{repo_path}/experts/GEND_ce.pt", map_location='cpu'),
51 'AGEV': torch.load(f"{repo_path}/experts/AGEV_huber.pt", map_location='cpu'),
52 'VALN': torch.load(f"{repo_path}/experts/VALN_huber.pt", map_location='cpu'),
53 'TEMP': torch.load(f"{repo_path}/experts/TEMP_huber.pt", map_location='cpu'),
54 }
55
56 # Standardize embedding for experts
57 emb_std = (embedding - expert_mean) / expert_std_val
58
59 # Run inference
60 results = {}
61 with torch.no_grad():
62 for dim, model in experts.items():
63 model.eval()
64 output = model(emb_std)
65
66 if dim == 'GEND': # CE expert
67 pred = torch.argmax(output, dim=-1).item()
68 else: # Huber expert
69 pred = torch.clamp(torch.round(output), 0, 6).item()
70
71 results[dim] = int(pred)
72
73 print("\nVoice attributes:")
74 print(f" Gender (0=M, 6=F): {results['GEND']}")
75 print(f" Age (0=child, 6=elderly): {results['AGEV']}")
76 print(f" Valence (0=negative, 6=positive): {results['VALN']}")
77 print(f" Tempo (0=very slow, 6=very fast): {results['TEMP']}")768 → 256 → ReLU → Dropout(0.3) → 128 → ReLU → Dropout(0.3) → 64 → ReLU → output768 → 13 → ReLU → Dropout(0.5) → 1| Dimension | Name | Type | Balanced Holdout | Gemini Pro | Samples |
|---|---|---|---|---|---|
| AGEV | Perceived Age | Huber | 90.8% | 92.2% | 1,560 |
| AROU | Arousal | Huber | 90.1% | 89.4% | 2,256 |
| ARSH | Arousal Shift | CE | 88.7% | 91.8% | 150 |
| ATCK | Attack | Huber | 87.8% | 93.4% | 1,344 |
| BKGN | Background Noise | Huber | 86.0% | 72.9% | 270 |
| BRGT | Brightness | CE | 68.2% | 64.3% | 108 |
| CHNK | Chunking | Huber | 81.0% | 89.1% | 456 |
| CLRT | Articulation Clarity | CE | 83.2% | 91.1% | 765 |
| COGL | Cognitive Load | Huber | 57.7% | 71.7% | 912 |
| DARC | Dynamic Arc | Huber | 62.4% | 73.7% | 144 |
| DFLU | Disfluency | Huber | 69.7% | 75.9% | 870 |
| EMPH | Emphasis | Huber | 85.8% | 90.5% | 672 |
| ESTH | Esthetics | Huber | 87.3% | 93.7% | 7,455 |
| EXPL | Explicitness | Huber | 88.7% | 90.8% | 372 |
| FOCS | Focus | CE | 83.3% | 80.3% | 1,446 |
| FULL | Fullness | CE | 73.5% | 69.0% | 138 |
| GEND | Perceived Gender | CE | 70.9% | 82.2% | 1,080 |
| HARM | Harmonicity | Huber | 82.9% | 89.3% | 1,190 |
| METL | Metallic Character | Huber | 88.6% | 93.9% | 565 |
| RANG | Pitch Range | CE | 75.5% | 82.8% | 312 |
| RCQL | Recording Quality | Huber | 91.2% | 88.2% | 3,212 |
| REGS | Register | Huber | 78.7% | 81.5% | 570 |
| RESP | Respiration | Huber | 85.1% | 87.4% | 1,164 |
| ROUG | Roughness | Huber | 81.3% | 84.6% | 678 |
| R_CHST | Chest Resonance | Huber | 84.4% | 88.1% | 2,616 |
| R_HEAD | Head Resonance | Huber | 85.0% | 90.2% | 1,848 |
| R_MASK | Mask Resonance | Huber | 82.6% | 91.9% | 1,236 |
| R_MIXD | Mixed Resonance | CE | 77.9% | 87.0% | 390 |
| R_NASL | Nasal Resonance | Huber | 86.3% | 84.9% | 168 |
| R_ORAL | Oral Resonance | CE | 77.8% | 87.1% | 126 |
| R_THRT | Throat Resonance | Huber | 82.6% | 85.7% | 798 |
| SMTH | Smoothness | Huber | 80.8% | 90.4% | 5,125 |
| STNC | Stance | Huber | 78.6% | 85.8% | 3,300 |
| STRU | Structure | CE | 84.5% | 91.0% | 1,098 |
| S_ASMR | ASMR Style | Huber | 86.8% | 93.6% | 2,130 |
| S_AUTH | Authoritative Style | Huber | 81.8% | 89.1% | 2,712 |
| S_CART | Cartoonish Style | Huber | 82.2% | 86.6% | 1,674 |
| S_CASU | Casual Style | Huber | 78.5% | 85.6% | 768 |
| S_CONV | Conversational Style | Huber | 83.3% | 82.9% | 1,110 |
| S_DRAM | Dramatic Style | Huber | 77.6% | 88.0% | 2,502 |
| S_FORM | Formal Style | Huber | 92.1% | 91.1% | 1,505 |
| S_MONO | Monologue Style | Huber | 68.0% | 70.0% | 18,990 |
| S_NARR | Narrator Style | Huber | 88.1% | 80.2% | 4,695 |
| S_NEWS | Newsreader Style | Huber | 90.9% | 68.6% | 8,270 |
| S_PLAY | Playful Style | Huber | 83.6% | 84.0% | 5,700 |
| S_RANT | Ranting/Angry Style | CE | 82.9% | 86.5% | 7,230 |
| S_STRY | Storytelling Style | Huber | 86.8% | 84.0% | 3,725 |
| S_TECH | Teacher/Didactic Style | Huber | 91.5% | 71.5% | 4,785 |
| S_WHIS | Whisper Style | Huber | 90.0% | 89.2% | 744 |
| TEMP | Tempo | Huber | 78.3% | 75.7% | 246 |
| TENS | Tension | Huber | 86.8% | 87.2% | 1,644 |
| VALN | Valence | Huber | 86.3% | 90.5% | 9,648 |
| VALS | Valence Shift | Huber | 49.7% | 32.6% | 80 |
| VFLX | Velocity Flux | CE | 92.9% | 94.3% | 30 |
| VOLT | Volatility | Huber | 71.0% | 87.6% | 348 |
| VULN | Vulnerability | CE | 78.3% | 82.9% | 1,734 |
| WARM | Warmth | Huber | 82.2% | 88.5% | 1,092 |
experts/
├── AGEV_huber.pt # Perceived Age expert
├── AROU_huber.pt # Arousal expert
├── ARSH_ce.pt # Arousal Shift expert
├── ... (54 more experts)
├── WARM_huber.pt # Warmth expert
speech_detector_best.pt # Binary speech classifier
expert_standardization.npz # Mean/std for expert inputs
speech_standardization.npz # Mean/std for speech detector
inference.py # Complete inference script1@misc{majestrino-voice-experts,
2 title={Majestrino-1.00 Voice Experts: Interpretable Voice Attribute Prediction},
3 author={LAION},
4 year={2026},
5 publisher={HuggingFace},
6 howpublished={\url{https://huggingface.co/laion/Majestrino-1.00-voice-experts}}
7}