Views
No views yet
| Binaural Encoder | mAP (↑) | ER20° (↓) | MAE (↓) | DER (↓) |
|---|---|---|---|---|
| SpatialAST | 49.90 | 24.43 | 17.87 | 32.50 |
| DSpAST (stage 1) | 53.05 | 98.56 | 95.57 | 97.58 |
| DSpAST (stage 2) | 52.64 | 20.31 | 14.44 | 28.35 |
| DSpAST (stage 3) | 54.53 | 20.28 | 14.44 | 28.03 |
1@article{wilkinghoff2025dspast,
2 author = {Wilkinghoff, Kevin and
3 Tan, Zheng-Hua},
4 title = {{DSpAST:} Disentangled Representations for Spatial Audio Reasoning with Large Language Models},
5 journal = {arXiv:2509.13927},
6 year = {2025}
7}1@inproceedings{zheng2024bat,
2 author = {Zheng, Zhisheng and
3 Peng, Puyuan and
4 Ma, Ziyang and
5 Chen, Xie and
6 Choi, Eunsol and
7 Harwath, David},
8 title = {{BAT:} Learning to Reason about Spatial Sounds with Large Language Models},
9 booktitle = {Proc. ICML},
10 year = {2024}
11}