Due to the limitation of strong-labeled sound event detection data set, using synthetic data to improve the sound event detection system performance has been a new research focus. In this paper, we try to exploit the usage of synthetic data to improve the feature representation. Based on metric learning, we proposed inter-frame distance loss function for domain adaptation, and prove the effectiveness of it on sound event detection. We also applied multi-task learning with synthetic data. We find the the best performance can be achieved when the two methods being used together. The experiment on DCASE 2018 task 4 test set and DCASE 2019 task 4 synthetic set both show competitive results.
@article{arxiv.2011.00695,
title = {Learning generic feature representation with synthetic data for weakly-supervised sound event detection by inter-frame distance loss},
author = {Yuxin Huang and Liwei Lin and Xiangdong Wang and Hong Liu and Yueliang Qian and Min Liu and Kazushige Ouchi},
journal= {arXiv preprint arXiv:2011.00695},
year = {2020}
}