@inproceedings{peng2024voicetextblender,
title = {VoiceTextBlender: Augmenting Large Language Models with Speech Capabilities via Single-Stage Joint Speech-Text Supervised Fine-Tuning},
author = {Yifan Peng and Krishna C. Puvvada and Zhehuai Chen and Piotr Żelasko and He Huang and Kunal Dhawan and Ke Hu and Shinji Watanabe and Jagadeesh Balam and Boris Ginsburg},
year = {2024},
booktitle = {North American Chapter of the Association for Computational Linguistics},
eprint = {2410.17485},
archivePrefix = {arXiv},
url = {https://arxiv.org/abs/2410.17485},
}