In this paper, we investigate the use of discourse-aware rewards with reinforcement learning to guide a model to generate long, coherent text. In particular, we propose to learn neural rewards to model cross-sentence ordering as a means to approximate desired discourse structure. Empirical results demonstrate that a generator trained with the learned reward produces more coherent and less repetitive text than models trained with cross-entropy or with reinforcement learning with commonly used scores as rewards.
@article{arxiv.1805.03766,
title = {Discourse-Aware Neural Rewards for Coherent Text Generation},
author = {Antoine Bosselut and Asli Celikyilmaz and Xiaodong He and Jianfeng Gao and Po-Sen Huang and Yejin Choi},
journal= {arXiv preprint arXiv:1805.03766},
year = {2018}
}