@online{Walter2204.12393,
TITLE = {On Fragile Features and Batch Normalization in Adversarial Training},
AUTHOR = {Walter, Nils Philipp and Stutz, David and Schiele, Bernt},
LANGUAGE = {eng},
URL = {https://arxiv.org/abs/2204.12393},
EPRINT = {2204.12393},
EPRINTTYPE = {arXiv},
YEAR = {2022},
ABSTRACT = {Modern deep learning architecture utilize batch normalization (BN) to<br>stabilize training and improve accuracy. It has been shown that the BN layers<br>alone are surprisingly expressive. In the context of robustness against<br>adversarial examples, however, BN is argued to increase vulnerability. That is,<br>BN helps to learn fragile features. Nevertheless, BN is still used in<br>adversarial training, which is the de-facto standard to learn robust features.<br>In order to shed light on the role of BN in adversarial training, we<br>investigate to what extent the expressiveness of BN can be used to robustify<br>fragile features in comparison to random features. On CIFAR10, we find that<br>adversarially fine-tuning just the BN layers can result in non-trivial<br>adversarial robustness. Adversarially training only the BN layers from scratch,<br>in contrast, is not able to convey meaningful adversarial robustness. Our<br>results indicate that fragile features can be used to learn models with<br>moderate adversarial robustness, while random features cannot<br>},
}
