Unleash the Power of Vision-Language Models by Visual Attention Prompt and Multimodal Interaction
- IEEE transactions on multimedia
- 19
@article{zhang2025unleash,
title = {Unleash the Power of Vision-Language Models by Visual Attention Prompt and Multimodal Interaction},
author = {Wenyao Zhang and Le Wu and Zequn Zhang and Tao Yu and Chao Ma and Xin Jin and Xiaokang Yang and Wenjun Zeng},
year = {2025},
journal = {IEEE transactions on multimedia},
doi = {10.1109/TMM.2024.3521785},
url = {https://doi.org/10.1109/TMM.2024.3521785},
}