Views
No views yet

| GPU Model | VRAM | Cost ($/hr) | RTF |
|---|---|---|---|
| RTX 5090 | 32GB | $0.423 | 0.190 |
| RTX 4080 | 16GB | $0.220 | 0.200 |
| RTX 5060 Ti | 16GB | $0.138 | 0.529 |
| RTX 4060 Ti | 16GB | $0.122 | 0.537 |
| RTX 3060 | 12GB | $0.093 | 0.600 |
1pip install kani-tts
2pip install -U "transformers==4.57.1" # for LFM2 !!!1from kani_tts import KaniTTS
2
3model = KaniTTS('nineninesix/kani-tts-400m-ar')
4
5# Generate audio from text
6audio, text = model("Your text here")
7
8# Save to file (requires soundfile)
9model.save_audio(audio, "output.wav")1from kani_tts import KaniTTS
2
3model = KaniTTS(
4 'nineninesix/kani-tts-400m-ar',
5 temperature=0.7, # Control randomness (default: 1.0)
6 top_p=0.9, # Nucleus sampling (default: 0.95)
7 max_new_tokens=2000, # Max audio length (default: 1200)
8 repetition_penalty=1.2, # Prevent repetition (default: 1.1)
9 suppress_logs=True, # Suppress library logs (default: True)
10 show_info=True, # Show model info on init (default: True)
11)
12
13audio, text = model("Your text here")1from kani_tts import KaniTTS
2from IPython.display import Audio as aplay
3
4model = KaniTTS('nineninesix/kani-tts-400m-ar')
5audio, text = model("Your text here")
6
7# Play audio in notebook
8aplay(audio, rate=model.sample_rate)أما أخوه غير الشقيق توماس، فكان يعيش حياةً طيبة في لندن، وكان مستعداً لمساعدته بما يلزم.أَمَّا أَخوهُ غَيْرُ الشَّقِيقِ تُوماسُ، فَكانَ يَعيشُ حَياةً طَيِّبَةً في لَنْدَنَ، وَكانَ مُسْتَعِدًّا لِمُساعَدَتِهِ بِما يَلْزَمُ.pyarabic or camel-tools| Text | Audio |
|---|---|
| مرحباً، اسمي تارا، وأنا نموذج لتوليد الصوت يمكنه أن يتحدث كالبشر تماماً. | |
| زقزقت عصافيرٌ مرِحة هذا الصباح على شجرة البلوط العتيقة خارج نافذتي. | |
| يا عزيزي، ما زلت لا أشعر بتحسن... سأذهب للنوم. | |
| أما أخوه غير الشقيق توماس، فكان يعيش حياةً طيبة في لندن، وكان مستعداً لمساعدته بما يلزم. |
@inproceedings{emilialarge,
author={He, Haorui and Shang, Zengqiang and Wang, Chaoren and Li, Xuyuan and Gu, Yicheng and Hua, Hua and Liu, Liwei and Yang, Chen and Li, Jiaqi and Shi, Peiyang and Wang, Yuancheng and Chen, Kai and Zhang, Pengyuan and Wu, Zhizheng},
title={Emilia: A Large-Scale, Extensive, Multilingual, and Diverse Dataset for Speech Generation},
booktitle={arXiv:2501.15907},
year={2025}
}@article{emonet_voice_2025,
author={Schuhmann, Christoph and Kaczmarczyk, Robert and Rabby, Gollam and Friedrich, Felix and Kraus, Maurice and Nadi, Kourosh and Nguyen, Huu and Kersting, Kristian and Auer, Sören},
title={EmoNet-Voice: A Fine-Grained, Expert-Verified Benchmark for Speech Emotion Detection},
journal={arXiv preprint arXiv:2506.09827},
year={2025}
}@dataset{masrispeech_full,
author = {Yahya Muhammad Alnwsany},
title = {MasriSpeech-Full: Large-Scale Egyptian Arabic Speech Corpus},
year = {2025},
publisher = {Hugging Face},
url = {https://huggingface.co/collections/NightPrince/masrispeech-dataset-68594e59e46fd12c723f1544}
}@misc{linagora2024Linto-tn,
title = {LinTO Audio and Textual Datasets to Train and Evaluate Automatic Speech Recognition in Tunisian Arabic Dialect},
author = {Hedi Naouara and Jérôme Louradour and Jean-Pierre Lorré},
year = {2025},
month = {March},
eprint={2504.02604},
archivePrefix={arXiv},
primaryClass={cs.CL},
note={Good Data Workshop, AAAI 2025},
url={arxiv.org/abs/2504.02604},
}@misc{abdallah2023leveraging,
title={Leveraging Data Collection and Unsupervised Learning for Code-switched Tunisian Arabic Automatic Speech Recognition},
author={Ahmed Amine Ben Abdallah and Ata Kabboudi and Amir Kanoun and Salah Zaiem},
year={2023},
eprint={2309.11327},
archivePrefix={arXiv},
primaryClass={eess.AS}
}
@data{e1qb-jv46-21,
doi = {10.21227/e1qb-jv46},
url = {https://dx.doi.org/10.21227/e1qb-jv46},
author = {Al-Fetyani, Mohammad and Al-Barham, Muhammad and Abandah, Gheith and Alsharkawi, Adham and Dawas, Maha},
publisher = {IEEE Dataport},
title = {MASC: Massive Arabic Speech Corpus},
year = {2021}
}
@misc{toyin2025arvoicemultispeakerdatasetarabic,
title={ArVoice: A Multi-Speaker Dataset for Arabic Speech Synthesis},
author={Hawau Olamide Toyin and Rufael Marew and Humaid Alblooshi and Samar M. Magdy and Hanan Aldarmaki},
year={2025},
eprint={2505.20506},
archivePrefix={arXiv},
primaryClass={cs.CL},
url={https://arxiv.org/abs/2505.20506},
}