1
2@article{xiong2024efficient,
3 title={Efficient Deformable ConvNets: Rethinking Dynamic and Sparse Operator for Vision Applications},
4 author={Yuwen Xiong and Zhiqi Li and Yuntao Chen and Feng Wang and Xizhou Zhu and Jiapeng Luo and Wenhai Wang and Tong Lu and Hongsheng Li and Yu Qiao and Lewei Lu and Jie Zhou and Jifeng Dai},
5 journal={arXiv preprint arXiv:2401.06197},
6 year={2024}
7}
8
9@article{wang2022internimage,
10 title={InternImage: Exploring Large-Scale Vision Foundation Models with Deformable Convolutions},
11 author={Wang, Wenhai and Dai, Jifeng and Chen, Zhe and Huang, Zhenhang and Li, Zhiqi and Zhu, Xizhou and Hu, Xiaowei and Lu, Tong and Lu, Lewei and Li, Hongsheng and others},
12 journal={arXiv preprint arXiv:2211.05778},
13 year={2022}
14}
15
16@inproceedings{zhu2022uni,
17 title={Uni-perceiver: Pre-training unified architecture for generic perception for zero-shot and few-shot tasks},
18 author={Zhu, Xizhou and Zhu, Jinguo and Li, Hao and Wu, Xiaoshi and Li, Hongsheng and Wang, Xiaohua and Dai, Jifeng},
19 booktitle={CVPR},
20 pages={16804--16815},
21 year={2022}
22}
23
24@article{zhu2022uni,
25 title={Uni-perceiver-moe: Learning sparse generalist models with conditional moes},
26 author={Zhu, Jinguo and Zhu, Xizhou and Wang, Wenhai and Wang, Xiaohua and Li, Hongsheng and Wang, Xiaogang and Dai, Jifeng},
27 journal={arXiv preprint arXiv:2206.04674},
28 year={2022}
29}
30
31@article{li2022uni,
32 title={Uni-Perceiver v2: A Generalist Model for Large-Scale Vision and Vision-Language Tasks},
33 author={Li, Hao and Zhu, Jinguo and Jiang, Xiaohu and Zhu, Xizhou and Li, Hongsheng and Yuan, Chun and Wang, Xiaohua and Qiao, Yu and Wang, Xiaogang and Wang, Wenhai and others},
34 journal={arXiv preprint arXiv:2211.09808},
35 year={2022}
36}
37
38@article{yang2022bevformer,
39 title={BEVFormer v2: Adapting Modern Image Backbones to Bird's-Eye-View Recognition via Perspective Supervision},
40 author={Yang, Chenyu and Chen, Yuntao and Tian, Hao and Tao, Chenxin and Zhu, Xizhou and Zhang, Zhaoxiang and Huang, Gao and Li, Hongyang and Qiao, Yu and Lu, Lewei and others},
41 journal={arXiv preprint arXiv:2211.10439},
42 year={2022}
43}
44
45@article{su2022towards,
46 title={Towards All-in-one Pre-training via Maximizing Multi-modal Mutual Information},
47 author={Su, Weijie and Zhu, Xizhou and Tao, Chenxin and Lu, Lewei and Li, Bin and Huang, Gao and Qiao, Yu and Wang, Xiaogang and Zhou, Jie and Dai, Jifeng},
48 journal={arXiv preprint arXiv:2211.09807},
49 year={2022}
50}
51
52@inproceedings{li2022bevformer,
53 title={Bevformer: Learning bird’s-eye-view representation from multi-camera images via spatiotemporal transformers},
54 author={Li, Zhiqi and Wang, Wenhai and Li, Hongyang and Xie, Enze and Sima, Chonghao and Lu, Tong and Qiao, Yu and Dai, Jifeng},
55 booktitle={ECCV},
56 pages={1--18},
57 year={2022},
58}