@misc{indiciaefb6cc217183d, title = {Demystifying Reinforcement Learning Post-Training of Language Models}, author = {Donovan Clay and Saket Gollapudi and Sankar Harilal and Min Jang and Jacob Morrison and Sewoong Oh and Natasha Jaques}, year = {2026}, url = {https://arxiv.org/abs/2608.24949}, note = {Source identifier: 2608.24949} }