1@misc{qwen3.5,
2 title = {{Qwen3.5}: Towards Native Multimodal Agents},
3 author = {{Qwen Team}},
4 month = {February},
5 year = {2026},
6 url = {https://qwen.ai/blog?id=qwen3.5}
7}
8
9
10@inproceedings{kwon2023vllm,
11 title={Efficient Memory Management for Large Language Model Serving with PagedAttention},
12 author={Kwon, Woosuk and Li, Zhuohan and Zhuang, Siyuan and Sheng, Ying and Zheng, Lianmin and Yu, Cody Hao and Gonzalez, Joseph E and Zhang, Hao and Stoica, Ion},
13 booktitle={Proceedings of the 29th Symposium on Operating Systems Principles (SOSP)},
14 pages={611--626},
15 year={2023},
16 eprint={2309.06180},
17 archivePrefix={arXiv}
18}
19
20@article{dao2023flashattention2,
21 title={FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning},
22 author={Dao, Tri},
23 journal={arXiv preprint arXiv:2307.08691},
24 year={2023}
25}
26
27@article{nvidia2025nanocodec,
28 title={NanoCodec: Towards High-Quality Ultra Fast Speech LLM Inference},
29 author={Casanova, Edresson and Neekhara, Paarth and Langman, Ryan and Hussain, Shehzeen and Ghosh, Subhankar and Yang, Xuesong and Juki{\'c}, Ante and Li, Jason and Ginsburg, Boris},
30 journal={arXiv preprint arXiv:2508.05835},
31 year={2025}
32}
33
34@article{mentzer2023fsq,
35 title={Finite Scalar Quantization: VQ-VAE Made Simple},
36 author={Mentzer, Fabian and Agustsson, Eirikur and Tschannen, Michael and Malireddy, Srikanth and Alshina, Elena},
37 journal={arXiv preprint arXiv:2309.15505},
38 year={2023}
39}
40
41@article{nvidia2024magpie,
42 title={Improving Robustness of LLM-based Speech Synthesis by Learning Monotonic Alignment},
43 author={Neekhara, Paarth and Hussain, Shehzeen and Ghosh, Subhankar and Li, Jason and Valle, Rafael and Badlani, Rohan and Ginsburg, Boris},
44 journal={arXiv preprint arXiv:2406.17957},
45 year={2024}
46}
47
48@article{ho2022cfg,
49 title={Classifier-Free Diffusion Guidance},
50 author={Ho, Jonathan and Salimans, Tim},
51 journal={arXiv preprint arXiv:2207.12598},
52 year={2022}
53}
54
55@article{rafailov2023dpo,
56 title={Direct Preference Optimization: Your Language Model is Secretly a Reward Model},
57 author={Rafailov, Rafael and Sharma, Archit and Mitchell, Eric and Ermon, Stefano and Manning, Christopher D and Finn, Chelsea},
58 journal={arXiv preprint arXiv:2305.18290},
59 year={2023}
60}
61
62@article{meng2024simpo,
63 title={SimPO: Simple Preference Optimization with a Reference-Free Reward},
64 author={Meng, Yu and Xia, Mengzhou and Chen, Danqi},
65 journal={arXiv preprint arXiv:2405.14734},
66 year={2024}
67}
68
69@article{li2023blip2,
70 title={BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models},
71 author={Li, Junnan and Li, Dongxu and Savarese, Silvio and Hoi, Steven},
72 journal={arXiv preprint arXiv:2301.12597},
73 year={2023}
74}
75
76@inproceedings{khosla2020supcon,
77 title={Supervised Contrastive Learning},
78 author={Khosla, Prannay and Teterwak, Piotr and Wang, Chen and Sarna, Aaron and Tian, Yonglong and Isola, Phillip and Maschinot, Aaron and Liu, Ce and Krishnan, Dilip},
79 booktitle={Advances in Neural Information Processing Systems (NeurIPS)},
80 volume={33},
81 pages={18661--18673},
82 year={2020},
83 eprint={2004.11362},
84 archivePrefix={arXiv}
85}
86
87@article{voicestar2025,
88 title={VoiceStar: Robust Zero-Shot Autoregressive TTS with Duration Control and Extrapolation},
89 author={Peng, Puyuan and Li, Shang-Wen and Mohamed, Abdelrahman and Harwath, David},
90 journal={arXiv preprint arXiv:2505.19462},
91 year={2025}
92}
93
94@article{radford2022whisper,
95 title={Robust Speech Recognition via Large-Scale Weak Supervision},
96 author={Radford, Alec and Kim, Jong Wook and Xu, Tao and Brockman, Greg and McLeavey, Christine and Sutskever, Ilya},
97 journal={arXiv preprint arXiv:2212.04356},
98 year={2022}
99}
100
101@inproceedings{emilialarge,
102 author={He, Haorui and Shang, Zengqiang and Wang, Chaoren and Li, Xuyuan and Gu, Yicheng and Hua, Hua and Liu, Liwei and Yang, Chen and Li, Jiaqi and Shi, Peiyang and Wang, Yuancheng and Chen, Kai and Zhang, Pengyuan and Wu, Zhizheng},
103 title={Emilia: A Large-Scale, Extensive, Multilingual, and Diverse Dataset for Speech Generation},
104 booktitle={arXiv:2501.15907},
105 year={2025}
106}
107
108@article{emonet_voice_2025,
109 author={Schuhmann, Christoph and Kaczmarczyk, Robert and Rabby, Gollam and Friedrich, Felix and Kraus, Maurice and Nadi, Kourosh and Nguyen, Huu and Kersting, Kristian and Auer, Sören},
110 title={EmoNet-Voice: A Fine-Grained, Expert-Verified Benchmark for Speech Emotion Detection},
111 journal={arXiv preprint arXiv:2506.09827},
112 year={2025}
113}
114
115@article{chen2021wavlm,
116 title={WavLM: Large-Scale Self-Supervised Pre-Training for Full Stack Speech Processing},
117 author={Chen, Sanyuan and Wang, Chengyi and Chen, Zhengyang and Wu, Yu and Liu, Shujie and Chen, Zhuoyuan and Li, Jinyu and others},
118 journal={arXiv preprint arXiv:2110.13900},
119 year={2021}
120}