This paper proposes a tool for efficiently constructing high-quality parallel corpora with minimizing human labor and making this tool publicly available. Our proposed construction process is based on neural machine translation (NMT) to allow for it to not only coexist with human translation, but also improve its efficiency by combining data quality control with human translation in a data-centric approach.
@article{arxiv.2111.00191,
title = {How should human translation coexist with NMT? Efficient tool for building high quality parallel corpus},
author = {Chanjun Park and Seolhwa Lee and Hyeonseok Moon and Sugyeong Eo and Jaehyung Seo and Heuiseok Lim},
journal= {arXiv preprint arXiv:2111.00191},
year = {2021}
}
Comments
Accepted for Data-centric AI workshop at NeurIPS 2021