Deep Policy Gradient Methods Without Batch Updates, Target Networks, or Replay Buffers
- Neural Information Processing Systems
- 2024-11-22
- 25
@article{vasan2024deep,
title = {Deep Policy Gradient Methods Without Batch Updates, Target Networks, or Replay Buffers},
author = {Gautham Vasan and Mohamed Elsayed and Alireza Azimi and Junjie He and Fahim Shariar and Colin Bellinger and Martha White and A. Rupam Mahmood},
year = {2024},
journal = {Neural Information Processing Systems},
eprint = {2411.15370},
archivePrefix = {arXiv},
url = {https://arxiv.org/abs/2411.15370},
}