@inproceedings{luo2024monointernvl,
title = {Mono-InternVL: Pushing the Boundaries of Monolithic Multimodal Large Language Models with Endogenous Visual Pre-training},
author = {G. Luo and Xue Yang and Wenhan Dou and Zhaokai Wang and Jifeng Dai and Yu Qiao and Xizhou Zhu},
year = {2024},
booktitle = {Computer Vision and Pattern Recognition},
eprint = {2410.08202},
archivePrefix = {arXiv},
doi = {10.1109/CVPR52734.2025.02324},
url = {https://arxiv.org/abs/2410.08202},
}