In this paper, we propose efficient and less resource-intensive strategies for parsing of code-mixed data. These strategies are not constrained by in-domain annotations, rather they leverage pre-existing monolingual annotated resources for training. We show that these methods can produce significantly better results as compared to an informed baseline. Besides, we also present a data set of 450 Hindi and English code-mixed tweets of Hindi multilingual speakers for evaluation. The data set is manually annotated with Universal Dependencies.
@article{arxiv.1703.10772,
title = {Joining Hands: Exploiting Monolingual Treebanks for Parsing of Code-mixing Data},
author = {Irshad Ahmad Bhat and Riyaz Ahmad Bhat and Manish Shrivastava and Dipti Misra Sharma},
journal= {arXiv preprint arXiv:1703.10772},
year = {2017}
}