<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2022.890016</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Understanding Robustness and Generalization of Artificial Neural Networks Through Fourier Masks</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Karantzas</surname> <given-names>Nikos</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x02020;</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Besier</surname> <given-names>Emma</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1886882/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ortega Caro</surname> <given-names>Josue</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Pitkow</surname> <given-names>Xaq</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="author-notes" rid="fn003"><sup>&#x02021;</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Tolias</surname> <given-names>Andreas S.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="author-notes" rid="fn003"><sup>&#x02021;</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Patel</surname> <given-names>Ankit B.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="author-notes" rid="fn003"><sup>&#x02021;</sup></xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Anselmi</surname> <given-names>Fabio</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<xref ref-type="author-notes" rid="fn003"><sup>&#x02021;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1028595/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Neuroscience, Baylor College of Medicine</institution>, <addr-line>Houston, TX</addr-line>, <country>United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>Center for Neuroscience and Artificial Intelligence, Baylor College of Medicine</institution>, <addr-line>Houston, TX</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Electrical and Computer Engineering, Rice University</institution>, <addr-line>Houston, TX</addr-line>, <country>United States</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Boulbaba Ben Amor, Inception Institute of Artificial Intelligence (IIAI), United Arab Emirates</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Gabriel Nathan Perdue, Fermi National Accelerator Laboratory (DOE), United States; Cristian Rusu, University of Bucharest, Romania</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Fabio Anselmi <email>Fabio.Anselmi&#x00040;bcm.edu</email></corresp>
<fn fn-type="other" id="fn001"><p>This article was submitted to Machine Learning and Artificial Intelligence, a section of the journal Frontiers in Artificial Intelligence</p></fn>
<fn fn-type="equal" id="fn002"><p>&#x02020;These author share first authorship</p></fn>
<fn fn-type="equal" id="fn003"><p>&#x02021;These author share senior authorship</p></fn></author-notes>
<pub-date pub-type="epub">
<day>12</day>
<month>07</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>5</volume>
<elocation-id>890016</elocation-id>
<history>
<date date-type="received">
<day>05</day>
<month>03</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>20</day>
<month>06</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2022 Karantzas, Besier, Ortega Caro, Pitkow, Tolias, Patel and Anselmi.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Karantzas, Besier, Ortega Caro, Pitkow, Tolias, Patel and Anselmi</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Despite the enormous success of artificial neural networks (ANNs) in many disciplines, the characterization of their computations and the origin of key properties such as generalization and robustness remain open questions. Recent literature suggests that robust networks with good generalization properties tend to be biased toward processing low frequencies in images. To explore the frequency bias hypothesis further, we develop an algorithm that allows us to learn <italic>modulatory masks</italic> highlighting the <italic>essential input frequencies</italic> needed for preserving a trained network&#x00027;s performance. We achieve this by imposing <italic>invariance</italic> in the loss with respect to such modulations in the input frequencies. We first use our method to test the low-frequency preference hypothesis of adversarially trained or data-augmented networks. Our results suggest that adversarially robust networks indeed exhibit a low-frequency bias but we find this bias is also dependent on directions in frequency space. However, this is not necessarily true for other types of data augmentation. Our results also indicate that the essential frequencies in question are effectively the ones used to achieve generalization in the first place. Surprisingly, images seen through these modulatory masks are not recognizable and resemble texture-like patterns.</p></abstract>
<kwd-group>
<kwd>Fourier analysis</kwd>
<kwd>symmetry</kwd>
<kwd>robustness</kwd>
<kwd>generalization</kwd>
<kwd>neural networks</kwd>
<kwd>data augmentation</kwd>
</kwd-group>
<counts>
<fig-count count="8"/>
<table-count count="3"/>
<equation-count count="8"/>
<ref-count count="21"/>
<page-count count="11"/>
<word-count count="5513"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1. Introduction</title>
<p>Artificial neural networks (ANNs) have achieved impressive performance in a variety of tasks, e.g., object recognition, function approximation, natural language processing, etc. (LeCun et al., <xref ref-type="bibr" rid="B11">2015</xref>). However, their computational capacity remains rather opaque. In particular, the operations performed by ANNs are profoundly constrained by the choice of architecture, initialization, optimization techniques, etc., and such constraints have a significant impact on key properties such as generalization power and robustness. Studying adversarial robustness has been a very active area of research, since it is closely related to how trustworthy and reliable neural networks can be (Goodfellow et al., <xref ref-type="bibr" rid="B7">2014</xref>). One of the most explored directions has been the analysis of adversarial perturbations from a frequency standpoint. For example, the work of Yin et al. (<xref ref-type="bibr" rid="B21">2019</xref>) establishes a relationship between the frequency domain of different noises (e.g Adversarial examples and Common corruptions) and model performance. In particular, they show that deep neural networks are more sensitive to high frequency adversarial attacks or common corruptions such as random noise, contrast change, and blurring. Additionally, adversarial perturbations of commonly trained models tend to be higher frequency than their adversarially trained counterparts. Furthermore, (Wang et al., <xref ref-type="bibr" rid="B20">2020</xref>) found that high frequency features are necessary for good generalization performance while the work of Sharma et al. (<xref ref-type="bibr" rid="B16">2019</xref>) shows that performance improvements in white-box and black-box transfer settings can be achieved only when low frequency components are preserved.</p>
<p>These results have led to various methodologies that help us understand artificial neural networks through a frequency lens. One such method is Neural Anisotropic Directions (NADs) (Ortiz-Jimenez et al., <xref ref-type="bibr" rid="B14">2020a</xref>,<xref ref-type="bibr" rid="B15">b</xref>). NADs are input directions for which a network is able to linearly classify data. Furthermore, Tsuzuku and Sato (<xref ref-type="bibr" rid="B19">2019</xref>) introduced a method to compute a neural network&#x00027;s sensitivity to input directions in the Fourier domain. Moreover, Li et al. (<xref ref-type="bibr" rid="B12">2022</xref>) show that robust deep learning object recognition models rely on low frequency information in natural images. Finally, Abello et al. (<xref ref-type="bibr" rid="B1">2021</xref>) divides the image frequency spectrum into disjoint disks and provides evidence that mid or high-level frequencies are important for ANN classification.</p>
<p>In this work we introduce a simple and easy-to-use method to <italic>learn</italic> the input frequency features that a network deems essential in order to achieve its classification performance. We visualize the relevant frequencies by learning a <italic>modulatory mask</italic> on the Fourier transform of the input data that defines a modulation-invariant loss function obtained <italic>via</italic> a simple optimization algorithm (Section 2.1). We compare such masks with their adversarially trained or data augmented counterparts (Section 3). In the case of adversarial training, the comparison is done at two levels of analysis. At a global level, we learn a mask for the entire test set. Our goal is to find the frequencies that allow for <italic>robust generalization</italic>. At a single image level, we explore the frequencies responsible for adversarial success/failure. Those comparisons allow us to test the hypothesis that adversarially trained models have a bias toward low frequency features and assess if the same holds for other types of data augmentation.</p>
<p>In the case of adversarial augmentation, our results confirm the low frequency bias hypothesis. However, they also highlight that the important frequency redistribution due to the augmentation is highly anisotropic. In the case of common data augmentations instead, our results show how the frequency reorganization depends on the type of augmentation, e.g., rotation- or scale-augmented models exhibit mid-high and low frequency biases, respectively.</p>
<p>The single-image mask analysis reveals that only a few, class-specific frequencies are crucial to determine a network&#x00027;s decision. Moreover, <italic>those frequencies are effectively the ones used to achieve its performance</italic>. In fact, mask-filtered images do not alter performance at all. However, surprisingly, they are not recognizable. They are characterized by texture-like patterns. This is in line with previous work by Geirhos et al. (<xref ref-type="bibr" rid="B6">2019</xref>), which provided evidence that Convolutional Neural Networks (CNNs) are biased toward textures rather than shapes in object recognition. Our method differs from all previous ones in that we explicitly learn the frequencies defining the features a model is sensitive to.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>2. Methods</title>
<sec>
<title>2.1. Approach</title>
<p>Artificial neural networks and their associated task-dependent losses define highly non-linear functions of their input. In terms of the frequency content found in a signal, the effect of the application of a non-linear function can be understood by considering the following simple one-dimensional example. Suppose <italic>f</italic>(<italic>t</italic>) &#x0003D; cos(<italic>w</italic><sub>1</sub><italic>t</italic>)&#x0002B;cos(<italic>w</italic><sub>2</sub><italic>t</italic>) is a sound wave and let &#x003C3;(<italic>t</italic>) &#x0003D; <italic>t</italic><sup>2</sup>. Then</p>
<disp-formula id="E1"><mml:math id="M3"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mo>&#x002218;</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mn>2</mml:mn></mml:mfrac><mml:mo stretchy='false'>[</mml:mo><mml:mn>2</mml:mn><mml:mo>+</mml:mo><mml:mi>cos</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mn>2</mml:mn><mml:msub><mml:mi>w</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>+</mml:mo><mml:mi>cos</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mn>2</mml:mn><mml:msub><mml:mi>w</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>+</mml:mo><mml:mn>2</mml:mn><mml:mi>cos</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>+</mml:mo><mml:mn>2</mml:mn><mml:mi>cos</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>]</mml:mo><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>We see that one of the effects of &#x003C3; on <italic>f</italic> is to generate the new frequency components <italic>w</italic><sub>1</sub>&#x02212;<italic>w</italic><sub>2</sub>, <italic>w</italic><sub>1</sub>&#x0002B;<italic>w</italic><sub>2</sub>, 2<italic>w</italic><sub>1</sub>, 2<italic>w</italic><sub>2</sub>. The first two are due to a phenomenon called <italic>intermodulation</italic>, the last are due to what is called <italic>harmonic distortion</italic>. Harmonic distortion has been studied in the context of neural networks with different activation functions by Christian et al. (<xref ref-type="bibr" rid="B4">2021</xref>), where an empirical demonstration and theoretical arguments are given to support the claim that the presence of non-linear elements mainly causes a spread in the frequency content of the loss function. Their reasoning is the following: let <italic>&#x003D5;</italic>: &#x0211D; &#x02192; &#x0211D; be a non-linear function and <italic>T&#x003D5;</italic> denote its Taylor expansion around the origin. For <italic>x</italic> &#x02208; &#x0211D;<sup><italic>d</italic></sup>, using the convolution theorem yields</p>
<disp-formula id="E2"><label>(1)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:mi>F</mml:mi><mml:mi>T</mml:mi><mml:mi>&#x003D5;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mi>F</mml:mi><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mi>n</mml:mi></mml:munder><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:munder><mml:munder><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x02299;</mml:mo><mml:mo>&#x022EF;</mml:mo><mml:mo>&#x02299;</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy='true'>&#x0FE38;</mml:mo></mml:munder><mml:mrow><mml:mtext>n-times</mml:mtext></mml:mrow></mml:munder></mml:mrow></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mi>n</mml:mi></mml:munder><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:munder><mml:munder><mml:mrow><mml:mover accent='true'><mml:mi>x</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mo>&#x0002A;</mml:mo><mml:mo>&#x022EF;</mml:mo><mml:mo>&#x0002A;</mml:mo><mml:mover accent='true'><mml:mi>x</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover></mml:mrow><mml:mo stretchy='true'>&#x0FE38;</mml:mo></mml:munder><mml:mrow><mml:mtext>n-times</mml:mtext></mml:mrow></mml:munder><mml:mo>,</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>&#x003D5;</italic> is acting pointwise on the components of <italic>x</italic>, <inline-formula><mml:math id="M5"><mml:mi>F</mml:mi><mml:mi>x</mml:mi><mml:mo>=</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula>, and the RHS is a weighted sum of self-convolutions. Christian et al. (<xref ref-type="bibr" rid="B4">2021</xref>) show that repeated convolutions broaden the frequency spectrum by adding higher frequency components corresponding to large coefficients <italic>a</italic><sub><italic>n</italic></sub>, an effect they call &#x0201C;blue shift&#x0201D;. A visual illustration of the blue-shift effect is shown in <xref ref-type="fig" rid="F1">Figure 1</xref> where we considered a one dimensional sinusoidal stimulus <italic>s</italic> filtered by softplus, tanh, ReLU, and hardtanh non-linearities. Additional to the blue shift effect (harmonic distortion), we also see the impact of intermodulation.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Non-linear distortions in the frequency domain due to the application of <bold>(B)</bold> softplus, <bold>(C)</bold> tanh, <bold>(D)</bold> ReLU, and <bold>(E)</bold> hardtanh non-linear activations on <italic>s</italic>(&#x000B7;) &#x0003D; sin(&#x000B7;) <bold>(A)</bold>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-05-890016-g0001.tif"/>
</fig>
<p>Let us now consider a more complex non-linear function such as a trained neural network. In this case, the non-linear distortion induced by the network will be manifested in its representation space and therefore in its decision making.</p>
<p>As mentioned above, one of the purposes of this work is to propose an algorithm to identify the <italic>essential input frequencies in a trained ANN&#x00027;s decisions</italic>. To this end, let us consider an image dataset <inline-formula><mml:math id="M10"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>, where <inline-formula><mml:math id="M11"><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> denotes the <italic>i</italic>-th input image and <italic>y</italic><sub><italic>i</italic></sub>&#x02208; &#x02124; <sub><italic>C</italic></sub> its associated label (<italic>C</italic> denotes the number of classes). We split <inline-formula><mml:math id="M12"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:math></inline-formula> into a training set <inline-formula><mml:math id="M13"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and a validation set <inline-formula><mml:math id="M14"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. We obtain the masks <italic>via</italic> the following optimization algorithm: we first pre-train a network &#x003A6; on <inline-formula><mml:math id="M15"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> with the objective of solving a classification task. We subsequently freeze the weights of &#x003A6; and attach a pre-processing layer whose weights are the entries <italic>m</italic><sub><italic>ij</italic></sub> of a mask matrix <inline-formula><mml:math id="M16"><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003A6;</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>. This layer acts as follows: for every <inline-formula><mml:math id="M17"><mml:mi>x</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> we modulate its Fourier transform <inline-formula><mml:math id="M18"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">F</mml:mi></mml:mrow><mml:mi>x</mml:mi></mml:math></inline-formula> by computing the product <inline-formula><mml:math id="M19"><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003A6;</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02299;</mml:mo><mml:mrow><mml:mi mathvariant="-tex-caligraphic">F</mml:mi></mml:mrow><mml:mi>x</mml:mi></mml:math></inline-formula>, where &#x02299; indicates the Hadamard product. We next compute the inverse Fourier transform <inline-formula><mml:math id="M20"><mml:mover accent="true"><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">F</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003A6;</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02299;</mml:mo><mml:mrow><mml:mi mathvariant="-tex-caligraphic">F</mml:mi></mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, which is then fed into the network (see <xref ref-type="fig" rid="F2">Figure 2</xref>). Finally, we learn the mask <italic>M</italic><sub>&#x003A6;</sub> by solving the optimization problem</p>
<disp-formula id="E3"><label>(2)</label><mml:math id="M21"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:msub><mml:mi>M</mml:mi><mml:mtext>&#x003A6;</mml:mtext></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:mo>,</mml:mo><mml:mi>p</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:msub><mml:mtext>argmin</mml:mtext><mml:mrow><mml:msub><mml:mi>M</mml:mi><mml:mtext>&#x003A6;</mml:mtext></mml:msub></mml:mrow></mml:msub><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mi mathvariant='-tex-caligraphic'>X</mml:mi><mml:mi>V</mml:mi></mml:msub></mml:mrow></mml:munder><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo>[</mml:mo> <mml:mrow><mml:mi>&#x02112;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mtext>&#x003A6;</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:mover accent='true'><mml:mi>x</mml:mi><mml:mo>&#x000AF;</mml:mo></mml:mover><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>&#x02112;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mtext>&#x003A6;</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow> <mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:msup></mml:mrow></mml:mstyle><mml:mo>+</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:mo>&#x02016;</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mtext>&#x003A6;</mml:mtext></mml:msub><mml:msub><mml:mo>&#x02016;</mml:mo><mml:mi>p</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x000A0;&#x000A0;</mml:mtext></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x003BB;</mml:mtext><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mi>&#x0211D;</mml:mi><mml:mo>+</mml:mo></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003A6; denotes the pre-trained network, &#x003BB;||<italic>M</italic><sub>&#x003A6;</sub>||<sub><italic>p</italic></sub> is a regularization term penalizing the <italic>p</italic>-norm of the learned mask, and <inline-formula><mml:math id="M23"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">L</mml:mi></mml:mrow></mml:math></inline-formula> is the loss function associated with the classification task. The first term in Equation (2) enforces an <italic>invariance</italic> in the loss with respect to the transformation <inline-formula><mml:math id="M24"><mml:mi>x</mml:mi><mml:mo>&#x021A6;</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:math></inline-formula> induced by the mask. The latter is key because we are expecting the desired frequencies to be revealed when there is no change in the loss <inline-formula><mml:math id="M25"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">L</mml:mi></mml:mrow></mml:math></inline-formula> and maximal change in the <italic>p</italic>-norm of the mask <italic>M</italic><sub>&#x003A6;</sub>. In other words, the mask is determined by a <italic>symmetry operation in the Fourier space of the input with minimal</italic> <italic>p</italic><italic>-norm</italic>. A solution to Equation (2) is a mask <italic>M</italic><sub>&#x003A6;</sub> addressing the question: which frequencies are essential in this trained ANN&#x00027;s decision making? Such masks, obtained for various data augmentation choices reveal the frequencies associated with each particular choice.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>A schematic of the preprocessing layer defined by the mask: the input <italic>x</italic> is transformed into the Fourier domain where it is filtered with a learnable mask <italic>M</italic><sub>&#x003A6;</sub>. <inline-formula><mml:math id="M1"><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003A6;</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02299;</mml:mo><mml:mrow><mml:mi mathvariant="-tex-caligraphic">F</mml:mi></mml:mrow><mml:mi>x</mml:mi></mml:math></inline-formula> is then remapped into the pixel domain through the inverse Fourier transform <inline-formula><mml:math id="M2"><mml:mover accent="true"><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">F</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003A6;</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02299;</mml:mo><mml:mrow><mml:mi mathvariant="-tex-caligraphic">F</mml:mi></mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-05-890016-g0002.tif"/>
</fig>
<p>At this point, we note that the mask is learned on the validation set <inline-formula><mml:math id="M26"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and not on the training set <inline-formula><mml:math id="M27"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. This is because we are interested in exploring the minimal set of frequencies preserving the <italic>generalization</italic> power of &#x003A6;. Moreover, we tested the stability of our mask generation algorithm across different runs. This is crucial since it attests to the reliability of our qualitative and quantitative analyses. We also note that masks can be obtained for single images, simply considering a single <inline-formula><mml:math id="M28"><mml:mi>x</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> in Equation (2) instead of the full validation set or a subset of it (e.g., class-specific masks, see <bold>Figure 5</bold>).</p>
</sec>
<sec>
<title>2.2. Dataset and Simulations</title>
<p>Our data consisted of 6, 644 image/label pairs from 5 classes of ImageNet (Deng et al., <xref ref-type="bibr" rid="B5">2009</xref>). Four thousand seven hundred and ten of those pairs belong to our training set <inline-formula><mml:math id="M29"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and the remaining 1, 934 pairs belong to our validation set <inline-formula><mml:math id="M30"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. For simplicity, we choose grayscale versions of our dataset images, though our method can be applied for any number of input channels. Our images were centered with respect to the mean and standard deviation of <inline-formula><mml:math id="M31"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>.</p>
<p>We initially trained VGG11 (Simonyan and Zisserman, <xref ref-type="bibr" rid="B17">2015</xref>) and ResNet18 (He et al., <xref ref-type="bibr" rid="B8">2016</xref>) baseline models on <inline-formula><mml:math id="M32"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> using the Pytorch framework. The performance of the models on the task was comparable and the results produced qualitatively similar. We therefore opted to present only the results obtained for VGG11. However, the interested reader can implement both models <italic>via</italic> the GitHub repository provided. For each subsequent training run we varied the type of data augmentation used for pre-processing (adversarial examples, random scales (for scaling factors in the interval [0.5, 1.5]), random translations (max absolute fraction for horizontal and vertical translations in [0.4, 0.4]), random rotations (for angles in [0, &#x003C0;])).</p>
<p>Each of the 5 networks in total was trained using the Adam optimizer (Kingma and Ba, <xref ref-type="bibr" rid="B10">2015</xref>) and a maximum learning rate of 10<sup>&#x02212;3</sup>. The learning rate of each learnable parameter group was scheduled according to the one-cycle learning rate policy with a minimum value of 0 (Smith, <xref ref-type="bibr" rid="B18">2017</xref>). We found that this set of hyperparameter choices allowed us to achieve stable training for all our models. We trained each model for a maximum of 50 epochs and eventually evaluated our models on the validation set <inline-formula><mml:math id="M41"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. We finally saved the weight-state of each model that achieved the minimum Cross Entropy loss within the chosen interval of epochs. For each of our pre-trained networks, we learn its corresponding Fourier mask according to the algorithmic process presented in Section 2.1. We use &#x02113;<sub>1</sub>-regularization on the norm of the mask to enforce sparsity. We train masks on both the whole of <inline-formula><mml:math id="M42"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> but also for single images. Each scheme required its own hyperparameter tuning, which by simple grid search revealed the choices of &#x003BB; &#x0003D; 0.2, 0.07 for masks on <inline-formula><mml:math id="M43"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and masks for single images, respectively. In the next section, we present masks for every data augmentation scheme we chose as well as their respective differences. For a given set of masks, we center the mask differences around the origin. This helps with the interpretation of the masks without altering the geometry of the particular set.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3. Results</title>
<p>Adversarial training can be seen as a type of data augmentation where the inputs are augmented with adversarial examples (Goodfellow et al., <xref ref-type="bibr" rid="B7">2014</xref>) to increase robustness to adversarial attacks. Here, we test the commonly accepted hypothesis that adversarially trained models need low frequency features for robustness. We do so by comparing the Fourier mask learned for a vanilla network &#x003A6;<sub><italic>N</italic></sub> with that of an adversarially trained network &#x003A6;<sub><italic>A</italic></sub> when the learning occurs over the whole validation set. Specifically, we compare a naturally trained VGG11 with an adversarially trained one using the <italic>torchattacks</italic> library (Kim, <xref ref-type="bibr" rid="B9">2020</xref>) and a Projected Gradient Descent attack (PGD). Caro et al. (<xref ref-type="bibr" rid="B2">2020</xref>) has shown the frequency structures of adversarial attacks are similar across different adversarial attacks. Therefore, although the set of potential choices one can explore is vast, in this work we focus on PGD for simplicity. Besides the mask difference we also compute the radial and angular energy of each mask by considering radial and angular partitions of the frequency domain (<bold>Figure 4</bold>). We then test if the same low-frequency preference hypothesis holds true in the case of common data augmentations. To gain some intuition, let us consider a simple one-layer network whose representation is given by &#x003A6;(<italic>x</italic>) &#x0003D; &#x003C3;&#x02329;<italic>w, x</italic>&#x0232A;, where &#x003C3;:&#x0211D; &#x02192; &#x0211D; is a non-linear function, <italic>x, w</italic> &#x02208; &#x0211D; <sup><italic>d</italic></sup>, and &#x02113;:&#x0211D; &#x02192; &#x0211D;<sub>&#x0002B;</sub> is a cost function. We consider data augmentations generated by a group of transformations <inline-formula><mml:math id="M44"><mml:mi>G</mml:mi><mml:mo>:</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mo>:</mml:mo><mml:mi>&#x003B8;</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>&#x02282;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>. The augmented loss can now be expressed as</p>
<disp-formula id="E4"><mml:math id="M45"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mi>&#x02112;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>w</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:mrow><mml:mstyle displaystyle='true'><mml:mrow><mml:mo>&#x0222B;</mml:mo><mml:mi>&#x02113;</mml:mi></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x003C3;</mml:mi><mml:mo>&#x02329;</mml:mo><mml:mi>w</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mi>&#x003B8;</mml:mi></mml:msub><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x0232A;</mml:mo><mml:mo>;</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mi>&#x003B8;</mml:mi><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:mrow><mml:mstyle displaystyle='true'><mml:mrow><mml:mo>&#x0222B;</mml:mo><mml:mi>&#x02113;</mml:mi></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x003C3;</mml:mi><mml:mo>&#x02329;</mml:mo><mml:msubsup><mml:mi>g</mml:mi><mml:mi>&#x003B8;</mml:mi><mml:mo>&#x002A;</mml:mo></mml:msubsup><mml:mi>w</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x0232A;</mml:mo><mml:mo>;</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mi>&#x003B8;</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02208;</mml:mo><mml:mi mathvariant='-tex-caligraphic'>X</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the second equality holds because <inline-formula><mml:math id="M46"><mml:mrow><mml:mo>&#x02329;</mml:mo><mml:mrow><mml:mi>w</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x0232A;</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>&#x02329;</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msubsup><mml:mi>w</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x0232A;</mml:mo></mml:mrow></mml:math></inline-formula> and <italic>g</italic><sup>&#x0002A;</sup> denotes the adjoint. We note that in this context the loss function is <italic>invariant to</italic> <italic>G</italic> <italic>transformations of the weights</italic>, i.e., <inline-formula><mml:math id="M47"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">L</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mi>w</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="-tex-caligraphic">L</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> for any <italic>g</italic><sub>&#x003B8;</sub>&#x02208; <italic>G</italic> (the proof of this statement relies on simple properties of group transformations, see Chen et al., <xref ref-type="bibr" rid="B3">2020</xref>). Here, we explore the impact such an invariance of the loss function has on the learned Fourier masks. The reasoning is as follows: updating the weights of an ANN is achieved through gradient descent, i.e., <inline-formula><mml:math id="M48"><mml:mi>&#x00394;</mml:mi><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:msub><mml:mrow><mml:mo>&#x02207;</mml:mo></mml:mrow><mml:mrow><mml:mi>w</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mi mathvariant="-tex-caligraphic">L</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, where <italic>w</italic><sub><italic>t</italic></sub> denotes the weights of the network at iteration <italic>t</italic> and &#x003B1; &#x02208; &#x0211D;<sup>&#x0002B;</sup> is the learning rate. The frequency content of the gradient of the loss at iteration <italic>t</italic> affects the frequency content of the weights. In turn, the latter determine the input frequencies the network is analyzing and thus will determine the mask. In other words, the frequency content of the loss, as well as how it is modified by different data augmentations, will impact the frequency content observed in the mask.</p>
<p>Let us consider a simple one dimensional example (<italic>d</italic> &#x0003D; 1) and the translation operator. In this case the loss <inline-formula><mml:math id="M51"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">L</mml:mi></mml:mrow></mml:math></inline-formula> is <italic>invariant to translations of the weights</italic>, i.e.,</p>
<disp-formula id="E5"><mml:math id="M52"><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">L</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="-tex-caligraphic">L</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mo>&#x02200;</mml:mo><mml:mi>t</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>&#x0211D;</mml:mi><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>where <italic>T</italic><sub><italic>t</italic></sub>:&#x0211D; &#x02192; &#x0211D; is the translation operator defined as <italic>T</italic><sub><italic>t</italic></sub>(&#x000B7;) &#x0003D; &#x000B7;&#x02212;<italic>t</italic>. For <inline-formula><mml:math id="M53"><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:math></inline-formula> and <italic>t</italic> &#x02208; &#x0211D;, let <italic>q</italic><sub><italic>i</italic></sub>(&#x000B7;): &#x0003D; &#x02113;(&#x003C3;(<italic>T</italic><sub><italic>t</italic></sub>(&#x000B7;)<italic>x</italic><sub><italic>i</italic></sub>);<italic>y</italic><sub><italic>i</italic></sub>). Then the Fourier transform of <inline-formula><mml:math id="M54"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">L</mml:mi></mml:mrow></mml:math></inline-formula> yields</p>
<disp-formula id="E6"><mml:math id="M55"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mi>&#x02131;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x02112;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x003B3;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:mrow><mml:mstyle displaystyle='true'><mml:mrow><mml:msubsup><mml:mo>&#x0222B;</mml:mo><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mi>&#x0221E;</mml:mi></mml:mrow><mml:mrow><mml:mo>+</mml:mo><mml:mi>&#x0221E;</mml:mi></mml:mrow></mml:msubsup><mml:mi>&#x02131;</mml:mi></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>q</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x003B3;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mn>2</mml:mn><mml:mi>&#x003C0;</mml:mi><mml:mi>i</mml:mi><mml:mi>&#x003B3;</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msup><mml:mi>d</mml:mi><mml:mi>t</mml:mi><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:mi>&#x003B4;</mml:mi></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x003B3;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mi>&#x02131;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>q</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x003B3;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:mi>&#x02131;</mml:mi></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>q</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where we used the translation property of the Fourier transform and &#x003B4; denotes the Dirac delta. This simple example illustrates the effect of the translation operator on the loss <inline-formula><mml:math id="M56"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">L</mml:mi></mml:mrow></mml:math></inline-formula>, i.e., a shift toward low frequencies (in this case a full shift of all frequencies to the DC component, the only non-zero component in the above equation). Note that an augmentation with all possible translations is not realistic. However, even a finite range of translations in the interval <italic>t</italic> &#x02208; [&#x02212;<italic>a, a</italic>], for a sufficiently large <italic>a</italic>, will produce a similar effect. Indeed, we have</p>
<disp-formula id="E7"><mml:math id="M57"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mi>&#x02131;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x02112;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x003B3;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:mrow><mml:mstyle displaystyle='true'><mml:mrow><mml:msubsup><mml:mo>&#x0222B;</mml:mo><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mi>&#x0221E;</mml:mi></mml:mrow><mml:mi>&#x0221E;</mml:mi></mml:msubsup><mml:mi>&#x02131;</mml:mi></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>q</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:msub><mml:mi>&#x003C7;</mml:mi><mml:mrow><mml:mo stretchy='false'>[</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:mi>a</mml:mi><mml:mo stretchy='false'>]</mml:mo></mml:mrow></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mn>2</mml:mn><mml:mi>&#x003C0;</mml:mi><mml:mi>i</mml:mi><mml:mi>&#x003B3;</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msup><mml:mi>d</mml:mi><mml:mi>t</mml:mi><mml:mtext>&#x000A0;&#x000A0;</mml:mtext></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mi>a</mml:mi></mml:mrow><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:mrow><mml:mtext>sinc</mml:mtext></mml:mrow></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:mn>2</mml:mn><mml:mi>&#x003C0;</mml:mi><mml:mi>&#x003B3;</mml:mi><mml:mi>a</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mi>&#x02131;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>q</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x003B3;</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003C7; denotes the characteristic function. Thus, the impact of averaging over an interval of translations on <inline-formula><mml:math id="M58"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">L</mml:mi></mml:mrow></mml:math></inline-formula> is to dampen its frequencies with a sinc function profile, i.e., a frequency re-weighting with a <italic>bias for low frequencies</italic>. However, we stress that the above argument is developed with a 1-layer network in mind. The effect of data-augmentation with respect to random translations viewed through a deep network is expected to be more intricate.</p>
<sec>
<title>3.1. Masks Generated for the Whole Dataset</title>
<p>We generated masks over <inline-formula><mml:math id="M59"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> for networks trained to be robust to adversarial examples, random scales, translations, and rotations. The masks in <xref ref-type="fig" rid="F3">Figure 3</xref> and their differences reveal how distinct frequency biases depend on the type of data augmentation. We also note how model performance is minimally altered by the introduction of the mask layer <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p><bold>(A)</bold> Learned masks for the vanilla network <italic>M</italic><sub><italic>N</italic></sub>, <bold>(B)</bold> Adversarially trained network <italic>M</italic><sub><italic>A</italic></sub>, <bold>(C)</bold> scale-invariant network <italic>M</italic><sub><italic>S</italic></sub>, <bold>(D)</bold> translation-invariant network <italic>M</italic><sub><italic>T</italic></sub> , <bold>(E)</bold> rotation-invariant network <italic>M</italic><sub><italic>R</italic></sub> and their differences.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-05-890016-g0003.tif"/>
</fig>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Model performance (%) with and without the mask layer.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th/>
<th valign="top" align="center"><italic><bold>M</bold></italic><sub><bold><italic>N</italic></bold></sub></th>
<th valign="top" align="center"><italic><bold>M</bold></italic><sub><bold><italic>A</italic></bold></sub></th>
<th valign="top" align="center"><italic><bold>M</bold></italic><sub><bold><italic>S</italic></bold></sub></th>
<th valign="top" align="center"><italic><bold>M</bold></italic><sub><bold><italic>T</italic></bold></sub></th>
<th valign="top" align="center"><italic><bold>M</bold></italic><sub><bold><italic>R</italic></bold></sub></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Standard</td>
<td valign="top" align="center">89.56</td>
<td valign="top" align="center">79.62</td>
<td valign="top" align="center">86.62</td>
<td valign="top" align="center">85.86</td>
<td valign="top" align="center">68.47</td>
</tr>
<tr>
<td valign="top" align="left">Masked</td>
<td valign="top" align="center">89.20</td>
<td valign="top" align="center">78.97</td>
<td valign="top" align="center">86.72</td>
<td valign="top" align="center">85.35</td>
<td valign="top" align="center">68.17</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>All accuracies are reported on the non-augmented validation set</italic>.</p>
</table-wrap-foot>
</table-wrap>
<p>In the case of adversarial augmentation there exists a net bias toward low frequencies as shown by the difference between the masks generated by the vanilla and adversarial trained network in <xref ref-type="fig" rid="F3">Figure 3B-A</xref>. This is further confirmed by the radial energy difference in <xref ref-type="fig" rid="F4">Figure 4(B-A</xref>)-radial, while the angular energy difference in <xref ref-type="fig" rid="F4">Figure 4(B-A</xref>)-angular shows that the redistribution of the frequencies occurs anisotropically.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>(First row) Radial and angular partitions of the Fourier domain; Energy differences <bold>(B-A, C-A, D-A, E-A)</bold> in radial and angular directions for the augmentations in <xref ref-type="fig" rid="F3">Figures 3(B&#x02013;E</xref>).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-05-890016-g0004.tif"/>
</fig>
<p>In the case of common augmentations our results exhibit contrasting effects in the Fourier masks. While the redistribution of the mask frequencies seems to be directionally-dependent (<xref ref-type="fig" rid="F4">Figure 4B(C-A),(D-A),(E-A)</xref>-angular), only robustness to scales endows the net with a bias toward low frequencies (<xref ref-type="fig" rid="F4">Figure 4(C-A</xref>)-radial). For translations the mask implies a less clear effect (<xref ref-type="fig" rid="F4">Figure 4(D-A</xref>)-radial), where a mixed behavior is present for mid and low frequencies. Interestingly, in the case of rotational robustness, <xref ref-type="fig" rid="F4">Figure 4(E-A</xref>)-radial shows a high frequency bias.</p>
</sec>
<sec>
<title>3.2. Masks Generated for Single Images</title>
<p>To further investigate the nature of adversarial robustness and how it is related to a network&#x00027;s generalization properties in the frequency domain we generated Fourier masks <italic>M</italic><sub><italic>N,x</italic></sub> for each <italic>correctly-classified</italic> image <italic>x</italic> in the validation set <italic>X</italic><sub><italic>V</italic></sub>. Moreover, for each such image <italic>x</italic> we consider its adversarial counterpart so that all adversarial examples are miss-classified. <xref ref-type="fig" rid="F5">Figure 5</xref> (top) shows such masks randomly sampled for images in all 5 data classes trained with respect to the vanilla network &#x003A6;<sub><italic>N</italic></sub>.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p><bold>(Top)</bold> Randomly sampled single image masks divided by class. The first column corresponds to masks trained on all the images belonging to each class separately. The colormap is the same as that of <xref ref-type="fig" rid="F3">Figure 3</xref>. <bold>(Bottom) (A)</bold> Randomly sampled images; <bold>(B)</bold> filtered by their complementary masks <inline-formula><mml:math id="M6"><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>; <bold>(C)</bold> the same images filtered by their associated Fourier masks <italic>M</italic><sub><italic>N,x</italic></sub>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-05-890016-g0005.tif"/>
</fig>
<p>It is worth noting that the masks are very sparse, i.e., very few frequencies are essential for preserving the prediction of the pretrained network. Additionally, for every mask <italic>M</italic><sub><italic>N,x</italic></sub>, we also consider its complementary mask <inline-formula><mml:math id="M60"><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> defined as</p>
<disp-formula id="E8"><mml:math id="M61"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mi>c</mml:mi></mml:msubsup><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo> <mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x0003C;</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup><mml:mtext>&#x000A0;&#x000A0;</mml:mtext></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x000A0;&#x000A0;&#x000A0;otherwise.</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Filtering an image with its complementary mask <inline-formula><mml:math id="M62"><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> does not compromise our ability to recognize the filtered image (<xref ref-type="fig" rid="F5">Figure 5B</xref>, Bottom). On the contrary, filtering with the mask <italic>M</italic><sub><italic>N,x</italic></sub> renders the image unrecognizable (<xref ref-type="fig" rid="F5">Figure 5C</xref>, Bottom). Filtered images resemble texture-like patterns. Interestingly, recent work by Geirhos et al. (<xref ref-type="bibr" rid="B6">2019</xref>) shows how ImageNet-trained CNNs are strongly biased toward recognizing textures rather than shapes. <xref ref-type="fig" rid="F6">Figure 6</xref> further confirms these results extending them to the case of adversarial images showing the masks learned from the vanilla and adversarially trained networks and their corresponding filtered images. Surprisingly, performance drops drastically (&#x0007E; 45% decrease) for images filtered by complementary masks <inline-formula><mml:math id="M63"><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>. Additionally, filtering adversarial examples using masks generated from original images reverses the effect of the attack in approximately 60% of validation samples. We unpack this information in <xref ref-type="table" rid="T2">Table 2</xref> below.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p><bold>(A)</bold> Correctly classified image <inline-formula><mml:math id="M7"><mml:mi>x</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>; <bold>(B)</bold> Adversarial Image <italic>x</italic><sub><italic>A</italic></sub>; <bold>(C)</bold> <italic>M</italic><sub><italic>N,x</italic></sub> - learned mask from vanilla network; <bold>(D)</bold> <italic>M</italic><sub><italic>N,x</italic></sub> - filtered image <italic>x</italic>; <bold>(E)</bold> <italic>M</italic><sub><italic>N,x</italic></sub> - filtered image <italic>x</italic><sub><italic>A</italic></sub>; <bold>(F)</bold> <inline-formula><mml:math id="M8"><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> - binary complement of <italic>M</italic><sub><italic>N,x</italic></sub>; <bold>(G)</bold> <inline-formula><mml:math id="M9"><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> - filtered image <italic>x</italic>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-05-890016-g0006.tif"/>
</fig>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p><italic>M</italic><sub><italic>N,x</italic></sub>(<italic>X</italic>) and <italic>M</italic><sub><italic>N,x</italic></sub>(<italic>X</italic><sub><italic>A</italic></sub>) denote original images/adversarial images filtered by masks trained for original images from the vanilla network.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th/>
<th valign="top" align="center"><bold>Data x</bold></th>
<th valign="top" align="center"><bold><italic>Adv. Data x</italic><sub><italic>A</italic></sub></bold></th>
<th valign="top" align="center"><bold><italic>M</italic></bold><sub><bold><italic>N,x</italic></bold></sub><bold>(<italic>x</italic>)</bold></th>
<th valign="top" align="center"><bold><italic>M</italic></bold><bold><sub><italic>N,x</italic></sub>(<italic>x</italic><sub><italic>A</italic></sub>)</bold></th>
<th valign="top" align="center"><bold><inline-formula><mml:math id="M33"><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Model accuracy</td>
<td valign="top" align="center">100%</td>
<td valign="top" align="center">0%</td>
<td valign="top" align="center">100%</td>
<td valign="top" align="center">58.83%</td>
<td valign="top" align="center">54.4%</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic><inline-formula><mml:math id="M34"><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> denotes the set of original images filtered by the complementary masks of M<sub>N,x</sub></italic>.</p>
</table-wrap-foot>
</table-wrap>
<p>We think this is an interesting result since</p>
<list list-type="bullet">
<list-item><p>The increase in performance when testing on <italic>M</italic><sub><italic>N,x</italic></sub>(<italic>x</italic><sub><italic>A</italic></sub>) provides strong evidence that the attack mostly relies on frequencies not present in the mask <italic>M</italic><sub><italic>N,x</italic></sub>.</p></list-item>
<list-item><p>The drop in performance when testing on <inline-formula><mml:math id="M64"><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> implies that the frequencies learned by each individual mask are not only sufficient but also necessary for the task.</p></list-item>
</list>
<p>Further confirming a low frequency bias in adversarially trained networks, <xref ref-type="fig" rid="F7">Figure 7</xref> shows the percentage of perturbed images for which the per-band energy (radial or angular) of their corresponding masks <inline-formula><mml:math id="M65"><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula> exceeds that of the masks <inline-formula><mml:math id="M66"><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:math></inline-formula> generated from the non-perturbed examples. <xref ref-type="fig" rid="F7">Figure 7</xref> confirms that lower frequencies are preferred for a robust representation.</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Comparison of <inline-formula><mml:math id="M35"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:math></inline-formula>-trained masks (<inline-formula><mml:math id="M36"><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:math></inline-formula>) obtained for the vanilla network and <inline-formula><mml:math id="M37"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>-trained masks (<inline-formula><mml:math id="M38"><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula>) obtained for an adversarially trained network: the bars illustrate the percentage of masks for which the per-band energy [radial <bold>(A)</bold> and angular <bold>(B)</bold>] in in <inline-formula><mml:math id="M39"><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula> exceeds that of <inline-formula><mml:math id="M40"><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:math></inline-formula>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-05-890016-g0007.tif"/>
</fig>
<p>Finally, upon visual inspection of the learned single-image masks we also suspected that such masks exhibit class-specificity. We tested this hypothesis by learning a linear classifier on the a uniformly balanced set of single image masks <inline-formula><mml:math id="M67"><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mi>x</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>. We considered 85% of the masks to be our training set for this task and later tested the linear classifier on the remaining 15% of single-image masks. <xref ref-type="table" rid="T3">Table 3</xref> below confirms that the essential frequencies for this network&#x00027;s generalization performance are class-specific. We also tested the robustness of this experiment by randomly shuffling the labels of the learned masks and testing if a linear classifier is still able to separate the masks based on their new label assignment. <xref ref-type="table" rid="T3">Table 3</xref> shows this is not the case and the results suggest that linear separability of the masks is due to their geometry and not the representation power of the linear classifier.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Training a linear classifier to separate single-image masks trained on the test images of <italic>X</italic>.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th/>
<th valign="top" align="center"><bold><italic>M</italic><sub><italic>N,X</italic></sub></bold></th>
<th valign="top" align="center"><bold><italic>M</italic><sub><italic>N,X</italic></sub></bold></th>
</tr>
<tr>
<th/>
<th valign="top" align="center"><bold>True labels</bold></th>
<th valign="top" align="center"><bold>Shuffled labels</bold></th>
</tr>
<tr>
<th/>
<th valign="top" align="center"><bold>(%)</bold></th>
<th valign="top" align="center"><bold>(%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Training accuracy</td>
<td valign="top" align="center">93.22</td>
<td valign="top" align="center">19.50</td>
</tr>
<tr>
<td valign="top" align="left">Test accuracy</td>
<td valign="top" align="center">83.87</td>
<td valign="top" align="center">16.42</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We visually illustrate these results by performing a manifold analysis of the learned masks using UMAP (McInnes et al., <xref ref-type="bibr" rid="B13">2018</xref>) for dimension reduction and visualization. Interestingly, we found that the masks are linearly separable and that the linear network responses cluster (<xref ref-type="fig" rid="F8">Figure 8</xref>).</p>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>Clustering of the output of a linear classifier learned to separate single image Fourier masks. <inline-formula><mml:math id="M49"><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:math></inline-formula> indicates the set of single image masks computed for the correctly classified test images in <inline-formula><mml:math id="M50"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">X</mml:mi></mml:mrow></mml:math></inline-formula>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-05-890016-g0008.tif"/>
</fig>
</sec>
</sec>
<sec id="s4">
<title>4. Discussion and Conclusions</title>
<p>In this work, we proposed a simple yet powerful approach to visualize the essential frequencies a trained network is using to solve a task. Our strategy consists of learning a frequency modulatory mask characterized by two critical properties:</p>
<list list-type="bullet">
<list-item><p>It defines a symmetry in the Cross Entropy loss, i.e., it does not alter the pretrained model&#x00027;s predictions.</p></list-item>
<list-item><p>It has minimal &#x02113;<sub><italic>p</italic></sub>-norm, which for <italic>p</italic> &#x0003D; 1 guarantees the preservation of performance while promoting sparsity in the mask.</p></list-item>
</list>
<p>Using our method we tested the common hypothesis that adversarially trained networks prefer low frequency features to achieve robustness. We also tested if this hypothesis holds true for common data augmentations such as translations, scales, and rotations.</p>
<p>In the case of adversarial augmentation, our results confirm the low frequency bias hypothesis. However, they also highlight that the frequency redistribution due to the augmentation is highly anisotropic. In the case of common data augmentations instead, our results show how the frequency reorganization depends on the type of augmentation.</p>
<p>In the case of adversarial training we also run a single image analysis to detect the frequencies useful for adversarial robustness and those responsible for adversarial weakness. Here too, masks learned on adversarially trained networks concentrate more toward lower frequencies compared to those learned on vanilla networks. Furthermore, the analysis showed that only a sparse, class-specific set of frequencies is needed to classify an image. Surprisingly, mask-filtered images in this case are not recognizable and resemble texture-like patterns, supporting the idea that ANNs use fundamentally different classification strategies from humans to achieve robust generalization (Geirhos et al., <xref ref-type="bibr" rid="B6">2019</xref>).</p>
<p>To our knowledge the use of a learned mask to characterize a network&#x00027;s crucial property such as robust generalization has not been proposed before. The interpretation of the masks provides us with a detailed geometrical description of directional and radial biases in the frequency domain as well as with quantifiable differences between various training schemes.</p>
<p>Our analysis can be extended to other architectural or optimization specifics, e.g., explicit regularizations, different optimizers/initializations, etc. The same mask approach can be employed to modulate the phase and modulus in the Fourier transform of the data. Our method effectively opens up many directions in the investigation of a network&#x00027;s implicit frequency bias. Future research directions will also include a natural generalization of our approach where the image features are learned, rather then fixed to be of the Fourier type.</p>
</sec>
<sec sec-type="data-availability" id="s5">
<title>Data Availability Statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://github.com/fastai/imagenette">https://github.com/fastai/imagenette</ext-link>. Code is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/nkarantzas/FourierMasks">https://github.com/nkarantzas/FourierMasks</ext-link>.</p>
</sec>
<sec id="s6">
<title>Author Contributions</title>
<p>NK and FA conceived the conceptualized framework and wrote the first draft. NK and EB trained and analyzed models. AP, JO, AT, and XP provided the feedback along the way. AT, AP, and XP provided the funding. All authors revised, edited and provided comments on the final manuscript and contributed to the article and approved the submitted version.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>This research was supported by the Intelligence Advanced Research Projects Activity (IARPA) via Department of Interior/Interior Business Center (DoI/IBC) contract no. D16PC00003. The US Government is authorized to reproduce and distribute reprints for governmental purposes notwithstanding any copyright annotation thereon. This work is also supported by the Lifelong Learning Machines (L2M) Program of the Defense Advanced Research Projects Agency (DARPA) via contract number HR0011-18-2-0025 and R01 EY026927 to AT and by NSF NeuroNex grant 1707400.</p>
</sec>
<sec id="s8">
<title>Author Disclaimer</title>
<p>The views and conclusions contained herein are those of the authors and should not be interpreted as necessarily representing the official policies or endorsements, either expressed or implied, of IARPA, DoI/IBC or the US Government.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x00027;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<ack><p>We also thank Shell Xu Hu, Kandan Ramakrishnan, and Zhe Li for helpful discussions.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abello</surname> <given-names>A. A.</given-names></name> <name><surname>Hirata</surname> <given-names>R.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Dissecting the high-frequency bias in convolutional neural networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Workshops</source>, Nashville, <volume>TN</volume>, <fpage>863</fpage>&#x02013;<lpage>871</lpage>. <pub-id pub-id-type="doi">10.1109/CVPRW53098.2021.00096</pub-id></citation>
</ref>
<ref id="B2">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Caro</surname> <given-names>J. O.</given-names></name> <name><surname>Ju</surname> <given-names>Y.</given-names></name> <name><surname>Pyle</surname> <given-names>R.</given-names></name> <name><surname>Dey</surname> <given-names>S.</given-names></name> <name><surname>Brendel</surname> <given-names>W.</given-names></name> <name><surname>Anselmi</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Local convolutions cause an implicit bias towards high frequency adversarial examples</article-title>. <source>arXiv preprint arXiv:2006.11440</source>.</citation>
</ref>
<ref id="B3">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>S.</given-names></name> <name><surname>Dobriban</surname> <given-names>E.</given-names></name> <name><surname>Lee</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;A group-theoretic framework for data augmentation,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, Vol. 33, eds H. Larochelle, M. Ranzato, R. Hadsell, M. F. Balcan, and H. Lin (Curran Associates, Inc.), <fpage>21321</fpage>&#x02013;<lpage>21333</lpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Christian</surname> <given-names>M.-G.</given-names></name> <name><surname>David</surname> <given-names>H.</given-names></name> <name><surname>Michael</surname> <given-names>W.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Ringing relus: harmonic distortion analysis of nonlinear feedforward networks,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source>, Vienna.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Deng</surname> <given-names>J.</given-names></name> <name><surname>Dong</surname> <given-names>W.</given-names></name> <name><surname>Socher</surname> <given-names>R.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Kai</surname> <given-names>L.</given-names></name> <name><surname>Li</surname> <given-names>F. F.</given-names></name></person-group> (<year>2009</year>). <article-title>&#x0201C;ImageNet: a large-scale hierarchical image database,&#x0201D;</article-title> in <source>2009 IEEE Conference on Computer Vision and Pattern Recognition</source>, Miami, <volume>FL</volume>, <fpage>248</fpage>&#x02013;<lpage>255</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2009.5206848</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Geirhos</surname> <given-names>R.</given-names></name> <name><surname>Rubisch</surname> <given-names>P.</given-names></name> <name><surname>Michaelis</surname> <given-names>C.</given-names></name> <name><surname>Bethge</surname> <given-names>M.</given-names></name> <name><surname>Wichmann</surname> <given-names>F. A.</given-names></name> <name><surname>Brendel</surname> <given-names>W.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;ImageNet-trained CNNs are biased towards texture; increasing shape bias improves accuracy and robustness,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source>, New Orleans.</citation>
</ref>
<ref id="B7">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Goodfellow</surname> <given-names>I. J.</given-names></name> <name><surname>Shlens</surname> <given-names>J.</given-names></name> <name><surname>Szegedy</surname> <given-names>C.</given-names></name></person-group> (<year>2014</year>). <article-title>Explaining and harnessing adversarial examples</article-title>. <source>arXiv preprint arXiv:1412.6572</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1412.6572</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Deep residual learning for image recognition,&#x0201D;</article-title> in <source>2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</source>. <pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id><pub-id pub-id-type="pmid">32166560</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>H..</given-names></name></person-group> (<year>2020</year>). <article-title>Torchattacks: a Pytorch repository for adversarial attacks</article-title>. <source>arXiv preprint arXiv:2010.01950</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2010.01950</pub-id></citation>
</ref>
<ref id="B10">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kingma</surname> <given-names>D. P.</given-names></name> <name><surname>Ba</surname> <given-names>J.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Adam: a method for stochastic optimization,&#x0201D;</article-title> in <source>3rd International Conference on Learning Representations, ICLR 2015</source>, eds Y. Bengio and Y. LeCun (San Diego, CA).</citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>LeCun</surname> <given-names>Y.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2015</year>). <article-title>Deep learning</article-title>. <source>Nature</source> <volume>521</volume>, <fpage>436</fpage>&#x02013;<lpage>444</lpage>. [preprint]. <pub-id pub-id-type="doi">10.1038/nature14539</pub-id><pub-id pub-id-type="pmid">26017442</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Caro</surname> <given-names>J. O.</given-names></name> <name><surname>Rusak</surname> <given-names>E.</given-names></name> <name><surname>Brendel</surname> <given-names>W.</given-names></name> <name><surname>Bethge</surname> <given-names>M.</given-names></name> <name><surname>Anselmi</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Robust deep learning object recognition models rely on low frequency information in natural images</article-title>. <source>bioRxiv</source>. <pub-id pub-id-type="doi">10.1101/2022.01.31.478509</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>McInnes</surname> <given-names>L.</given-names></name> <name><surname>Healy</surname> <given-names>J.</given-names></name> <name><surname>Melville</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>UMAP: uniform manifold approximation and projection for dimension reduction</article-title>. <source>arxiv preprint arxiv:1802.03426</source>. <pub-id pub-id-type="doi">10.21105/joss.00861</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ortiz-Jimenez</surname> <given-names>G.</given-names></name> <name><surname>Modas</surname> <given-names>A.</given-names></name> <name><surname>Moosavi-Dezfooli</surname> <given-names>S.-M.</given-names></name> <name><surname>Frossard</surname> <given-names>P.</given-names></name></person-group> (<year>2020a</year>). <article-title>Hold me tight! influence of discriminative features on deep network boundaries</article-title>. <source>arXiv preprint arXiv:2002.06349</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2002.06349</pub-id></citation>
</ref>
<ref id="B15">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ortiz-Jimenez</surname> <given-names>G.</given-names></name> <name><surname>Modas</surname> <given-names>A.</given-names></name> <name><surname>Moosavi-Dezfooli</surname> <given-names>S.-M.</given-names></name> <name><surname>Frossard</surname> <given-names>P.</given-names></name></person-group> (<year>2020b</year>). <article-title>Neural anisotropy directions</article-title>. <source>arXiv preprint arXiv:2006.09717</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2006.09717</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sharma</surname> <given-names>Y.</given-names></name> <name><surname>Ding</surname> <given-names>G. W.</given-names></name> <name><surname>Brubaker</surname> <given-names>M. A.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;On the effectiveness of low frequency perturbations,&#x0201D;</article-title> in <source>IJCAI</source>. Macao. <pub-id pub-id-type="doi">10.24963/ijcai.2019/470</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Simonyan</surname> <given-names>K.</given-names></name> <name><surname>Zisserman</surname> <given-names>A.</given-names></name></person-group> (<year>2015</year>). <article-title>Very deep convolutional networks for large-scale image recognition</article-title>. <source>arXiv preprint arXiv:1409.1556</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1409.1556</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>L. N..</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Cyclical learning rates for training neural networks,&#x0201D;</article-title> in <source>2017 IEEE Winter Conference on Applications of Computer Vision (WACV)</source>, Santa Rosa, <volume>CA</volume>, <fpage>464</fpage>&#x02013;<lpage>472</lpage>. <pub-id pub-id-type="doi">10.1109/WACV.2017.58</pub-id><pub-id pub-id-type="pmid">34874998</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tsuzuku</surname> <given-names>Y.</given-names></name> <name><surname>Sato</surname> <given-names>I.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;On the structural sensitivity of deep convolutional networks to the directions of fourier basis functions,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, Long Beach, <volume>CA</volume>, <fpage>51</fpage>&#x02013;<lpage>60</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2019.00014</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Yang</surname> <given-names>Y.</given-names></name> <name><surname>Shrivastava</surname> <given-names>A.</given-names></name> <name><surname>Rawal</surname> <given-names>V.</given-names></name> <name><surname>Ding</surname> <given-names>Z.</given-names></name></person-group> (<year>2020</year>). <article-title>Towards frequency-based explanation for robust CNN</article-title>. <source>arXiv preprint arXiv:2005.03141</source>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yin</surname> <given-names>D.</given-names></name> <name><surname>Gontijo Lopes</surname> <given-names>R.</given-names></name> <name><surname>Shlens</surname> <given-names>J.</given-names></name> <name><surname>Cubuk</surname> <given-names>E. D.</given-names></name> <name><surname>Gilmer</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;A Fourier perspective on model robustness in computer vision,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 32</source>, eds H. Wallach, H. Larochelle, A. Beygelzimer, F. d Alch&#x000E9;-Buc, E. Fox and R. Garnett (Curran Associates, Inc.), <volume>32</volume>, <fpage>13276</fpage>&#x02013;<lpage>13286</lpage>.</citation>
</ref>
</ref-list> 
</back>
</article>