RLHF-V is trained on
RLHF-V-Dataset, which contains
fine-grained segment-level human corrections on diverse instructions. The base model is trained on
UniMM-Chat, which is a high-quality knowledge-intensive SFT dataset. We introduce a new method
Dense Direct Preference Optimization (DDPO) that can make better use of the fine-grained annotations.
For more details, please refer to our
paper.
1@article{yu2023rlhf,
2 title={Rlhf-v: Towards trustworthy mllms via behavior alignment from fine-grained correctional human feedback},
3 author={Yu, Tianyu and Yao, Yuan and Zhang, Haoye and He, Taiwen and Han, Yifeng and Cui, Ganqu and Hu, Jinyi and Liu, Zhiyuan and Zheng, Hai-Tao and Sun, Maosong and others},
4 journal={arXiv preprint arXiv:2312.00849},
5 year={2023}
6}
7
8@article{yu2024rlaifv,
9 title={RLAIF-V: Aligning MLLMs through Open-Source AI Feedback for Super GPT-4V Trustworthiness},
10 author={Yu, Tianyu and Zhang, Haoye and Yao, Yuan and Dang, Yunkai and Chen, Da and Lu, Xiaoman and Cui, Ganqu and He, Taiwen and Liu, Zhiyuan and Chua, Tat-Seng and Sun, Maosong},
11 journal={arXiv preprint arXiv:2405.17220},
12 year={2024},
13}