Views
No views yet

| # | Degraded input | Restored — 44.1 kHz | DNSMOS OVRL |
|---|---|---|---|
| 1 | 2.16 → 3.50 (+1.34) | ||
| 2 | 2.83 → 3.59 (+0.76) | ||
| 3 | 2.56 → 3.65 (+1.10) | ||
| 4 | 3.32 → 3.89 (+0.57) |
sig_bak_ovr.onnx) for perceptual quality, CER against the ground-truth transcript for content preservation.| Model | DNSMOS OVRL ↑ | SIG ↑ | BAK ↑ | CER (median) ↓ |
|---|---|---|---|---|
| Sidon | 3.923 | 4.129 | 4.416 | 0.016 |
| Diamond (ours) | 3.829 | 4.056 | 4.348 | 0.028 |
| RE-USE | 3.789 | 4.003 | 4.344 | 0.014 |
| Resemble Enhance | 3.764 | 3.994 | 4.301 | 0.027 |
| UniSE | 3.752 | 4.014 | 4.261 | 0.027 |
| VoiceFixer | 3.566 | 3.790 | 4.226 | 0.042 |
| input (degraded) | 3.575 | 3.890 | 4.082 | 0.014 |
Use it responsibly. Restored speech is synthesized speech. Don't present it as an unaltered recording, and don't use it to fabricate or misattribute what someone said.
1@software{diamond_2026,
2 author = {Almaz Zholdoshbek uulu, Ulanbek Abdurazakov, Denis Pavlov and Nursultan Bakashov},
3 title = {Diamond: A Sequence-to-Sequence Model for Speech Restoration via an Autoregressive RQ-Transformer over Neural Audio Codec Tokens},
4 year = {2026},
5 publisher = {Hugging Face},
6 howpublished = {\url{https://huggingface.co/nineninesix/diamond-1.0}},
7 note = {Trained from scratch; no pretrained backbone}
8}1@inproceedings{kumar2023dac,
2 title={High-Fidelity Audio Compression with Improved RVQGAN},
3 author={Kumar, Rithesh and Seetharaman, Prem and Luebs, Alejandro and Kumar, Ishaan and Kumar, Kundan},
4 booktitle={NeurIPS},
5 year={2023},
6 note={arXiv:2306.06546}
7}
8
9@inproceedings{lee2022rqtransformer,
10 title={Autoregressive Image Generation using Residual Quantization},
11 author={Lee, Doyup and Kim, Chiheon and Kim, Saehoon and Cho, Minsu and Han, Wook-Shin},
12 booktitle={CVPR},
13 year={2022},
14 note={arXiv:2203.01941}
15}
16
17@article{defossez2024moshi,
18 title={Moshi: a speech-text foundation model for real-time dialogue},
19 author={D{\'e}fossez, Alexandre and Mazar{\'e}, Laurent and Orsini, Manu and Royer, Am{\'e}lie and P{\'e}rez, Patrick and J{\'e}gou, Herv{\'e} and Grave, Edouard and Zeghidour, Neil},
20 journal={arXiv preprint arXiv:2410.00037},
21 year={2024}
22}
23
24@inproceedings{copet2023musicgen,
25 title={Simple and Controllable Music Generation},
26 author={Copet, Jade and Kreuk, Felix and Gat, Itai and Remez, Tal and Kant, David and Synnaeve, Gabriel and Adi, Yossi and D{\'e}fossez, Alexandre},
27 booktitle={NeurIPS},
28 year={2023},
29 note={arXiv:2306.05284}
30}
31
32@inproceedings{koizumi2023librittsr,
33 title={LibriTTS-R: A Restored Multi-Speaker Text-to-Speech Corpus},
34 author={Koizumi, Yuma and Zen, Heiga and Karita, Shigeki and Ding, Yifan and Yatabe, Kohei and Morioka, Nobuyuki and Bacchiani, Michiel and Zhang, Yu and Han, Wei and Bapna, Ankur},
35 booktitle={Interspeech},
36 year={2023},
37 note={arXiv:2305.18802}
38}
39
40@inproceedings{reddy2022dnsmos,
41 title={DNSMOS P.835: A Non-Intrusive Perceptual Objective Speech Quality Metric to Evaluate Noise Suppressors},
42 author={Reddy, Chandan K. A. and Gopal, Vishak and Cutler, Ross},
43 booktitle={ICASSP},
44 year={2022},
45 note={arXiv:2110.01763}
46}