@misc{indiciae8b60ff95bd2f, title = {Gains and Collapse in On-Policy Distillation:A Reinforcement Learning Perspective}, author = {Han Cui and Jianhao Yan and Yun Luo and Hongbo Zhang and Zhizhang Fu and Yue Zhang}, year = {2026}, url = {https://arxiv.org/abs/2610.03185}, note = {Source identifier: 2610.03185} }