We study the adversarial robustness of information bottleneck models for classification. Previous works showed that the robustness of models trained with information bottlenecks can improve upon adversarial training. Our evaluation under a diverse range of white-box l∞ attacks suggests that information bottlenecks alone are not a strong defense strategy, and that previous results were likely influenced by gradient obfuscation.
@article{arxiv.2107.05712,
title = {A Closer Look at the Adversarial Robustness of Information Bottleneck Models},
author = {Iryna Korshunova and David Stutz and Alexander A. Alemi and Olivia Wiles and Sven Gowal},
journal= {arXiv preprint arXiv:2107.05712},
year = {2021}
}