@misc{indiciae7d5538d7353f, title = {Overconfident Errors Need Stronger Correction: Asymmetric Confidence Penalties for Reinforcement Learning}, author = {Yuanda Xu and Hejian Sang and Zhengze Zhou and Ran He and Zhipeng Wang}, year = {2026}, url = {https://arxiv.org/abs/2602.21420}, note = {Source identifier: 2602.21420} }