@misc{indiciaed73e7dd693fd, title = {Explore More, Drift Less: Outcome-Only Reinforcement Learning Can Suffice for Long-Horizon Interactive Agents}, author = {Liming Pu and Xiaoxia Li and Yifu Liu and Teng Cao and Bin Yang}, year = {2026}, url = {https://arxiv.org/abs/2609.01245}, note = {Source identifier: 2609.01245} }