It is shown that for deep neural networks, a single wide layer of width N+1 (N being the number of training samples) suffices to prove the connectivity of sublevel sets of the training loss function. In the two-layer setting, the same property may not hold even if one has just one neuron less (i.e. width N can lead to disconnected sublevel sets).
@article{arxiv.2101.08576,
title = {A Note on Connectivity of Sublevel Sets in Deep Learning},
author = {Quynh Nguyen},
journal= {arXiv preprint arXiv:2101.08576},
year = {2021}
}