@misc{indiciae9d68d30e9fff, title = {Posterior sampling for reinforcement learning: worst-case regret bounds}, author = {Shipra Agrawal and Randy Jia}, year = {2020}, url = {https://arxiv.org/abs/1705.07041}, note = {Source identifier: 1705.07041} }