COPF: Continual Learning Human Preference through Optimal Policy Fitting
- 14
@misc{zhang2023copf,
title = {COPF: Continual Learning Human Preference through Optimal Policy Fitting},
author = {Han Zhang and Lin Gui and Yuan-Zhao Zhai and Hui Wang and Yu Lei and Ruifeng Xu},
year = {2023},
url = {https://www.semanticscholar.org/paper/b6a2f5f0c41ae933bebb6cd09bf5d065d7f675b0},
}