This paper proposes a new method to generate synthetic data sets based on copula models. Our goal is to produce surrogate data resembling real data in terms of marginal and joint distributions. We present a complete and reliable algorithm for generating a synthetic data set comprising numeric or categorical variables. Applying our methodology to two datasets shows better performance compared to other methods such as SMOTE and autoencoders.
@article{arxiv.2203.17250,
title = {Generation and Simulation of Synthetic Datasets with Copulas},
author = {Regis Houssou and Mihai-Cezar Augustin and Efstratios Rappos and Vivien Bonvin and Stephan Robert-Nicoud},
journal= {arXiv preprint arXiv:2203.17250},
year = {2022}
}