We study what provable privacy attacks can be shown on trained, 2-layer ReLU neural networks. We explore two types of attacks; data reconstruction attacks, and membership inference attacks. We prove that theoretical results on the implicit bias of 2-layer neural networks can be used to provably reconstruct a set of which at least a constant fraction are training points in a univariate setting, and can also be used to identify with high probability whether a given point was used in the training set in a high dimensional setting. To the best of our knowledge, our work is the first to show provable vulnerabilities in this implicit-bias-driven setting.
@article{arxiv.2410.07632,
title = {Provable Privacy Attacks on Trained Shallow Neural Networks},
author = {Guy Smorodinsky and Gal Vardi and Itay Safran},
journal= {arXiv preprint arXiv:2410.07632},
year = {2025}
}