@online{Fan_arXiv2002.01827,
TITLE = {Analyzing the Dependency of {ConvNets} on Spatial Information},
AUTHOR = {Fan, Yue and Xian, Yongqin and Losch, Max Maria and Schiele, Bernt},
LANGUAGE = {eng},
URL = {https://arxiv.org/abs/2002.01827},
EPRINT = {2002.01827},
EPRINTTYPE = {arXiv},
YEAR = {2020},
ABSTRACT = {Intuitively, image classification should profit from using spatial<br>information. Recent work, however, suggests that this might be overrated in<br>standard CNNs. In this paper, we are pushing the envelope and aim to further<br>investigate the reliance on spatial information. We propose spatial shuffling<br>and GAP+FC to destroy spatial information during both training and testing<br>phases. Interestingly, we observe that spatial information can be deleted from<br>later layers with small performance drops, which indicates spatial information<br>at later layers is not necessary for good performance. For example, test<br>accuracy of VGG-16 only drops by 0.03% and 2.66% with spatial information<br>completely removed from the last 30% and 53% layers on CIFAR100, respectively.<br>Evaluation on several object recognition datasets (CIFAR100, Small-ImageNet,<br>ImageNet) with a wide range of CNN architectures (VGG16, ResNet50, ResNet152)<br>shows an overall consistent pattern.<br>},
}
