Simple and Scalable Strategies to Continually Pre-train Large Language Models
- 2024-03-13
- 130
@article{ibrahim2024simple,
title = {Simple and Scalable Strategies to Continually Pre-train Large Language Models},
author = {Adam Ibrahim and Benjamin Therien and Kshitij Gupta and Mats L. Richter and Quentin Anthony and Timothée Lesort and Eugene Belilovsky and Irina Rish},
year = {2024},
journal = {TMLR},
eprint = {2403.08763},
archivePrefix = {arXiv},
url = {https://arxiv.org/abs/2403.08763},
}