TD3R: Stabilizing the Offline-to-Online Reinforcement Learning by Restricting Policy Updates
- 2025-07-28
- 1
@inproceedings{huang2025tdr,
title = {TD3R: Stabilizing the Offline-to-Online Reinforcement Learning by Restricting Policy Updates},
author = {Biao Huang and Xianghui Wang and Xinming Zhang and Rongfei Chen and Shanze Wang and Mingao Tan and Wei Zhang},
year = {2025},
booktitle = {Cybersecurity and Cyberforensics Conference},
doi = {10.23919/CCC64809.2025.11179713},
url = {https://doi.org/10.23919/CCC64809.2025.11179713},
}