This paper investigates the effects of data size and frequency range on distributional semantic models. We compare the performance of a number of representative models for several test settings over data of varying sizes, and over test items of various frequency. Our results show that neural network-based models underperform when the data is small, and that the most reliable model over data of varying sizes and frequency ranges is the inverted factorized model.
@article{arxiv.1609.08293,
title = {The Effects of Data Size and Frequency Range on Distributional Semantic Models},
author = {Magnus Sahlgren and Alessandro Lenci},
journal= {arXiv preprint arXiv:1609.08293},
year = {2016}
}