Mitigating Forgetting Between Supervised and Reinforcement Learning Yields Stronger Reasoners
- 2025-10-06
- 18
@misc{yuan2025mitigating,
title = {Mitigating Forgetting Between Supervised and Reinforcement Learning Yields Stronger Reasoners},
author = {Xiangchi Yuan and Xiang Chen and Tong Yu and Dachuan Shi and Can Jin and Wenke Lee and Saayan Mitra},
year = {2025},
eprint = {2510.04454},
archivePrefix = {arXiv},
url = {https://arxiv.org/abs/2510.04454},
}