If this work is helpful for your research, please consider citing InternVideo.
@article{wang2024internvideo2, title={Internvideo2: Scaling video foundation models for multimodal video understanding}, author={Wang, Yi and Li, Kunchang and Li, Xinhao and Yu, Jiashuo and He, Yinan and Wang, Chenting and Chen, Guo and Pei, Baoqi and Zheng, Rongkun and Xu, Jilan and Wang, Zun and others}, journal={arXiv preprint arXiv:2403.15377}, year={2024} }