Views
No views yet
[!NOTE] This checkpoint was trained from a mixture of stage-2 and stage-3 data, yielding much better general chatting capabilities but slightly sub-optimal grounding performance. It shall be considered as the default setting for this model.
@inproceedings{liu2024etbench,
title={E.T. Bench: Towards Open-Ended Event-Level Video-Language Understanding},
author={Liu, Ye and Ma, Zongyang and Qi, Zhongang and Wu, Yang and Chen, Chang Wen and Shan, Ying},
booktitle={Neural Information Processing Systems (NeurIPS)},
year={2024}
}