Reinforcement Learning with Semantic Rewards Enables Low-Resource Language Expansion without Alignment Tax
Minzu University of China · Shanghai Jiao Tong University · Peking University · Chinese Academy of Sciences · University of Macau · Hainan University
- Findings of the Association for Computational Linguistics: ACL 2026
- 2026-01-01
- 1
@inproceedings{su2026reinforcement,
title = {Reinforcement Learning with Semantic Rewards Enables Low-Resource Language Expansion without Alignment Tax},
author = {Zeli Su and Ziyin Zhang and Zhou Liu and Xuexian Song and Zhankai Xu and Longfei Zheng and Xiaolu Zhang and Rong Fu and Guixian Xu and Wentao Zhang},
year = {2026},
booktitle = {Findings of the Association for Computational Linguistics: ACL 2026},
eprint = {2605.14366},
archivePrefix = {arXiv},
doi = {10.18653/v1/2026.findings-acl.880},
url = {https://arxiv.org/abs/2605.14366},
}