In this paper, we investigate the driving factors behind concatenation, a simple but effective data augmentation method for low-resource neural machine translation. Our experiments suggest that discourse context is unlikely the cause for the improvement of about +1 BLEU across four language pairs. Instead, we demonstrate that the improvement comes from three other factors unrelated to discourse: context diversity, length diversity, and (to a lesser extent) position shifting.
@article{arxiv.2105.01691,
title = {Data Augmentation by Concatenation for Low-Resource Translation: A Mystery and a Solution},
author = {Toan Q. Nguyen and Kenton Murray and David Chiang},
journal= {arXiv preprint arXiv:2105.01691},
year = {2021}
}