@misc{indiciae2caf0fc942e9, title = {Learning Online Alignments with Continuous Rewards Policy Gradient}, author = {Yuping Luo and Chung-Cheng Chiu and Navdeep Jaitly and Ilya Sutskever}, year = {2016}, url = {https://arxiv.org/abs/1608.01281}, note = {Source identifier: 1608.01281} }