Visual Representation Matters: Exploiting Temporal Differences in Video-to-Audio Generation
- 2026-08-05
- 0
@misc{chen2026visual,
title = {Visual Representation Matters: Exploiting Temporal Differences in Video-to-Audio Generation},
author = {Zehua Chen and Junyou Wang and Yuxuan Jiang and Zhenying Fang and Yusheng Dai and Jianfei Chen and Ziwei Liu and Jun Zhu},
year = {2026},
eprint = {2608.04902},
archivePrefix = {arXiv},
url = {https://arxiv.org/abs/2608.04902},
}