@misc{dai202415mmultimodalfacialimagetext,
title={15M Multimodal Facial Image-Text Dataset},
author={Dawei Dai and YuTang Li and YingGe Liu and Mingming Jia and Zhang YuanHui and Guoyin Wang},
year={2024},
eprint={2407.08515},
archivePrefix={arXiv},
primaryClass={cs.CV},
url={
https://arxiv.org/abs/2407.08515},
}
@article{dai2026facecaption15m,
title={FaceCaption-15M: Benchmarking and Enhancing Facial Vision-Language… See the full description on the dataset page:
https://huggingface.co/datasets/OpenFace-CQUPT/FaceCaptionMask-1M.