@article{dontdropdropoutoptimizinglayersparsityfo, title = {Don't Drop Dropout: Optimizing Layer Sparsity for Efficient LLM Training and Inference}, author = {Mostafa Elhoushi and Alex Pretko and Nolan Dey and Bin Claire Zhang and Gavia Gray and Gurpreet Gosal and Abdulrahman Mahmoud and Shane Bergsma and Joel Hestness}, year = {2026}, eprint = {2609.05275}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2609.05275}, }