<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurorobot.</journal-id>
<journal-title>Frontiers in Neurorobotics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurorobot.</abbrev-journal-title>
<issn pub-type="epub">1662-5218</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnbot.2023.1129720</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>DualFlow: Generating imperceptible adversarial examples by flow field and normalize flow-based model</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Liu</surname> <given-names>Renyang</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2148730/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Jin</surname> <given-names>Xin</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1169523/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Hu</surname> <given-names>Dongting</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1996499/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Jinhong</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2177610/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Yuanyu</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2189861/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Jin</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2189604/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Zhou</surname> <given-names>Wei</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1916059/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>School of Information Science and Engineering, Yunnan University</institution>, <addr-line>Kunming</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Engineering Research Center of Cyberspace, Yunnan University</institution>, <addr-line>Kunming</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>National Pilot School of Software, Yunnan University</institution>, <addr-line>Kunming</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>School of Mathematics and Statistics, University of Melbourne</institution>, <addr-line>Melbourne, VIC</addr-line>, <country>Australia</country></aff>
<aff id="aff5"><sup>5</sup><institution>Kunming Institute of Physics, Yunnan University</institution>, <addr-line>Kunming</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Di Wu, Chongqing Institute of Green and Intelligent Technology (CAS), China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Ye Yuan, Southwest University, China; Linyuan Wang, National Digital Switching System Engineering and Technological Research Centre, China; Krishnaraj Nagappan, SRM Institute of Science and Technology, India</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Wei Zhou &#x02709; <email>zwei&#x00040;ynu.edu.cn</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>09</day>
<month>02</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>17</volume>
<elocation-id>1129720</elocation-id>
<history>
<date date-type="received">
<day>22</day>
<month>12</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>24</day>
<month>01</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2023 Liu, Jin, Hu, Zhang, Wang, Zhang and Zhou.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Liu, Jin, Hu, Zhang, Wang, Zhang and Zhou</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license></permissions>
<abstract>
<p>Recent adversarial attack research reveals the vulnerability of learning-based deep learning models (DNN) against well-designed perturbations. However, most existing attack methods have inherent limitations in image quality as they rely on a relatively loose noise budget, i.e., limit the perturbations by <italic>L</italic><sub><italic>p</italic></sub>-norm. Resulting that the perturbations generated by these methods can be easily detected by defense mechanisms and are easily perceptible to the human visual system (HVS). To circumvent the former problem, we propose a novel framework, called <bold>DualFlow</bold>, to craft adversarial examples by disturbing the image&#x00027;s latent representations with spatial transform techniques. In this way, we are able to fool classifiers with human imperceptible adversarial examples and step forward in exploring the existing DNN&#x00027;s fragility. For imperceptibility, we introduce the flow-based model and spatial transform strategy to ensure the calculated adversarial examples are perceptually distinguishable from the original clean images. Extensive experiments on three computer vision benchmark datasets (CIFAR-10, CIFAR-100 and ImageNet) indicate that our method can yield superior attack performance in most situations. Additionally, the visualization results and quantitative performance (in terms of six different metrics) show that the proposed method can generate more imperceptible adversarial examples than the existing imperceptible attack methods.</p></abstract>
<kwd-group>
<kwd>deep learning</kwd>
<kwd>adversarial attack</kwd>
<kwd>adversarial example</kwd>
<kwd>normalize flow</kwd>
<kwd>spatial transform</kwd>
</kwd-group>
<contract-num rid="cn001">62162067</contract-num>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content></contract-sponsor>
<counts>
<fig-count count="3"/>
<table-count count="7"/>
<equation-count count="13"/>
<ref-count count="63"/>
<page-count count="12"/>
<word-count count="8947"/>
</counts>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1. Introduction</title>
<p>Deep neural networks (DNNs) have achieved remarkable achievements in theories and applications. However, the DNNs have been proven to be easily fooled by adversarial examples (AEs), which are generated by adding well-designed unwanted perturbations to the original clean data (Zhou et al., <xref ref-type="bibr" rid="B62">2019</xref>). In these years, many studies dabbled in crafting adversarial examples and revealed that many DNN applications are vulnerable to them. Such as Computer Vision (CV) (Kurakin et al., <xref ref-type="bibr" rid="B31">2017</xref>; Eykholt et al., <xref ref-type="bibr" rid="B18">2018</xref>; Duan et al., <xref ref-type="bibr" rid="B17">2020</xref>), Neural Language Processing (NLP) (Xu H. et al., <xref ref-type="bibr" rid="B55">2020</xref>; Shao et al., <xref ref-type="bibr" rid="B47">2022</xref>; Yi et al., <xref ref-type="bibr" rid="B58">2022</xref>), and Autonomous Driving (Liu A. et al., <xref ref-type="bibr" rid="B35">2019</xref>; Zhao et al., <xref ref-type="bibr" rid="B61">2019</xref>; Yan et al., <xref ref-type="bibr" rid="B57">2022</xref>). Generally, in CV, the AE needs to meet the following two properties, one is that it can attack the target model successfully, resulting in the target model outputting wrong predictions; another one is its perturbations should be invisible to human eyes (Goodfellow et al., <xref ref-type="bibr" rid="B20">2015</xref>; Carlini and Wagner, <xref ref-type="bibr" rid="B6">2017</xref>).</p>
<p>Unfortunately, most existing works (Kurakin et al., <xref ref-type="bibr" rid="B31">2017</xref>; Dong et al., <xref ref-type="bibr" rid="B15">2018</xref>, <xref ref-type="bibr" rid="B16">2019</xref>) are focused on promoting the generated adversarial examples&#x00027; attack ability but ignored the visual aspects of the crafted evil examples. Typically, the calculated adversarial noise is limited by a small <italic>L</italic><sub><italic>p</italic></sub>-norm ball, which tries to keep the built adversarial examples looking like the original image as possible. However, the <italic>L</italic><sub><italic>p</italic></sub>-norm limited adversarial perturbations blur the images to a large extent and are so conspicuous to human eyes and not harmonious with the whole image. Furthermore, these <italic>L</italic><sub><italic>p</italic></sub>-norm-based methods, which modify the entire image at the pixel level, seriously affect the quality of the generated adversarial images. Resulting in the vivid details of the original image can not be preserved. Besides, the adversarial examples crafted in these settings can be easily detected by the defense mechanism or immediately discarded by the target model and further encounter the &#x0201C;denied to service.&#x0201D; All the mentioned above can lead the attack to be failed. Furthermore, most existing methods adopt <italic>L</italic><sub><italic>p</italic></sub>-norm, i.e., <italic>L</italic><sub>2</sub> and <italic>L</italic><sub><italic>inf</italic></sub>-norm, distance as the metrics to constraint the image&#x00027;s distortion. Indeed, the <italic>L</italic><sub><italic>p</italic></sub>-norm can ensure the similarity between the clean and adversarial images. However, it does not perform well in evaluating an adversarial example.</p>
<p>Recently, some studies have attempted to generate adversarial examples beyond the <italic>L</italic><sub><italic>p</italic></sub>-norm ball limited way. For instance, patch-based adversarial attacks, which usually extend into the physical world, do not limit the intensity of perturbation but the range scope. Such as adversarial-Yolo (Thys et al., <xref ref-type="bibr" rid="B50">2019</xref>), DPatch (Liu X. et al., <xref ref-type="bibr" rid="B36">2019</xref>), AdvCam (Duan et al., <xref ref-type="bibr" rid="B17">2020</xref>), Sparse-RS (Croce et al., <xref ref-type="bibr" rid="B9">2022</xref>). To obtain more human harmonious adversarial examples with acceptable attack success rate in the digital world, Xiao et al. (<xref ref-type="bibr" rid="B54">2018</xref>) proposed the stAdv to generate adversarial examples by spatial transform to modify each pixel&#x00027;s position in the whole image. The overall visual effect of the adversarial example generated by stAdv is good. However, the adversarial examples generated by stAdv usually have serration modifications and are visible to the naked eye. Later, the Chroma-Shift (Aydin et al., <xref ref-type="bibr" rid="B2">2021</xref>) made a forward step by applying the spatial transform to the image&#x00027;s YUV space rather than RGB space. Unfortunately, these attacks have destroyed the semantic information and data distribution of the image, resulting that the generated adversarial noise that can be easily detected by the defense mechanism (Arvinte et al., <xref ref-type="bibr" rid="B1">2020</xref>; Xu Z. et al., <xref ref-type="bibr" rid="B56">2020</xref>; Besnier et al., <xref ref-type="bibr" rid="B5">2021</xref>) and leading the attack failed.</p>
<p>To gap this bridge, we formulate the issue of synthesizing invisible adversarial examples beyond noise-adding at pixel level and propose a novel attack method called <bold>DualFlow</bold>. More specifically, DualFlow uses spatial transform techniques to disturb the latent representation of the image rather than directly adding well-designed noise to the benign image, which can significantly improve the adversarial noise&#x00027;s concealment and preserve the adversarial examples&#x00027; vivid details at the same time. The spatial transform can learn a smooth flow field vector <italic>f</italic> for each value&#x00027;s new location in the latent space to optimize an eligible adversarial example. Furthermore, the adversarial examples are not limited to <italic>L</italic><sub><italic>p</italic></sub>-norm rules, which can guarantee the image quality and details of the generated examples. Empirically, the proposed DualFlow can remarkably preserve the images&#x00027; vivid details while achieving an admirable attack success rate.</p>
<p>We conduct extensive experiments on three different computer vision benchmark datasets. Results illustrate that the adversarial perturbations generated by the proposed method take into account the data structure and only appear around the target object. We draw the adversarial examples and their corresponding noise from the noise-adding method MI-FGSM and the DualFlow in <xref ref-type="fig" rid="F1">Figure 1</xref>. As shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, our proposed method slightly alters this area around the target object, thus ensuring the invisibility of the adversarial perturbations. Furthermore, the statistical results demonstrate that the DualFlow can guarantee the generated adversarial examples&#x00027; image quality compared to the existing imperceptible attack methods on the target models while outperforming them both on the ordinary and defense models concerning attack success rate. The main contributions could be summarized as follows:</p>
<list list-type="bullet">
<list-item><p>We propose a novel attack method, named DualFlow, which generates adversarial examples by directly disturbing the latent representation of the clean examples rather than performing an attack on the pixel level.</p></list-item>
<list-item><p>We craft the adversarial examples by applying the spatial transform techniques to the latent value to preserve the details of original images and guarantee the adversarial images&#x00027; quality.</p></list-item>
<list-item><p>Comparing with the existing attack methods, experimental results show our method&#x00027;s superiority in synthesizing adversarial examples with the highest attack ability, best invisibility, and remarkable image quality.</p></list-item>
</list>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>The adversarial examples generated by the MI-FGSM (Aydin et al., <xref ref-type="bibr" rid="B2">2021</xref>) and the proposed DualFlow for the ResNet-152 (He et al., <xref ref-type="bibr" rid="B22">2016</xref>) model. Specifically, the first column and the second column are the adversarial examples and their corresponding adversarial perturbations generated by MI-FGSM, respectively. The middle column is the clean images. The last two columns are the adversarial perturbations and their corresponding adversarial examples, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-17-1129720-g0001.tif"/>
</fig>
<p>The rest of this paper is organized as follows. First, we briefly review the methods relating to adversarial attacks and imperceptible adversarial attacks in Section 2. Then, Sections 3 and 4, introduce the preliminary knowledge and the details of the proposed DualFlow framework. Finally, the experimental results are presented in Section 5, with the conclusion drawn in Section 6.</p></sec>
<sec id="s2">
<title>2. Related work</title>
<p>In this section, we briefly review the most pertinent attack methods to the proposed work: the adversarial attacks and the techniques used for crafting inconspicuous adversarial perturbations.</p>
<sec>
<title>2.1. Adversarial attack</title>
<p>Previous researchers contend that deep neural networks (DNN) are sensitive to adversarial examples (Goodfellow et al., <xref ref-type="bibr" rid="B20">2015</xref>), which are crafted by disturbing the clean data slightly but can fool the well-trained DNN models. The classical adversarial attack methods can be classified into two categories, white-box attacks (Kurakin et al., <xref ref-type="bibr" rid="B31">2017</xref>; Madry et al., <xref ref-type="bibr" rid="B41">2018</xref>) and black-box attacks (Narodytska and Kasiviswanathan, <xref ref-type="bibr" rid="B42">2017</xref>; Bai et al., <xref ref-type="bibr" rid="B3">2023</xref>). In white-box settings, the attackers can generate adversarial examples with a nearly 100% attack success rate because they can access the complete information of the target DNN model, while for the physical world, the black-box attack is more threatening to the DNN applications because they don&#x00027;t need too much information about the DNN models&#x00027; details (Ilyas et al., <xref ref-type="bibr" rid="B25">2018</xref>, <xref ref-type="bibr" rid="B26">2019</xref>; Guo et al., <xref ref-type="bibr" rid="B21">2019</xref>).</p></sec>
<sec>
<title>2.2. Imperceptible adversarial attacks</title>
<p>Recently, some studies have attempted to generate adversarial examples beyond the <italic>L</italic><sub><italic>p</italic></sub>-norm ball limit for obtaining humanly imperceptible adversarial examples. LowProFool (Ballet et al., <xref ref-type="bibr" rid="B4">2019</xref>) propose an imperceptibility attack to craft invisible adversarial examples in the tabular domain. Its empirical results show that LowProFool can generate imperceptible adversarial examples while keeping a high fooling rate. For computer vision tasks the attackers will also consider the human perception of the generated adversarial examples. In Luo et al. (<xref ref-type="bibr" rid="B37">2018</xref>), the authors propose a new approach to craft adversarial examples, which design a new distance metric that considers the human perceptual system and maximizes the noise tolerance of the generated adversarial examples. This metric evaluates the sensitivity of image pixels to the human eye and can ensure that the crafted adversarial examples are highly imperceptible and robust to the physical world. stAdv (Xiao et al., <xref ref-type="bibr" rid="B54">2018</xref>) focuses on generating different adversarial perturbations through spatial transform and claims that such adversarial examples are perceptually realistic and more challenging to defend against with existing defense systems. Later, the Chroma-Shift (Aydin et al., <xref ref-type="bibr" rid="B2">2021</xref>) made a forward step by applying the spatial transform to the image&#x00027;s YUV space rather than RGB space. AdvCam (Duan et al., <xref ref-type="bibr" rid="B17">2020</xref>) crafts and disguises adversarial examples of the physical world into natural styles to make them appear legitimate to a human observer. It transfers large adversarial perturbations into a custom style and then &#x0201C;hides&#x0201D; them in a background other than the target object. Moreover, its experimental results that AEs produced by AdvCam are well camouflaged and highly concealed in both digital and physical world scenarios while still being effective in deceiving state-of-the-art DNN image detectors. SSAH (Luo et al., <xref ref-type="bibr" rid="B38">2022</xref>) crafts adversarial examples and disguises adversarial noise in a low-frequency constraints manner. This method limits the adversarial perturbations to the high-frequency components of the specific image to ensure low human perceptual similarity. The SSAH also jumps out of the original <italic>L</italic><sub><italic>p</italic></sub>-norm constraint-based attack way and provides a new idea for calculating adversarial noise.</p>
<p>Therefore, crafting adversarial examples, especially for the imperceptible ones, poses the request for a method that can efficiently and effectively build adversarial examples with high invisibility and image quality efficiently and effectively. On the other hand, with the development of defense mechanisms, higher requirements are placed on the defense resistance of adversarial examples. To achieve these goals, we learn from the previous studies that adversarial examples can be gained beyond noise-adding ways. Hence, we are well motivated to develop a novel method to disturb the original image latent representation obtained by a well-trained normalizing flow-based model, and then apply a well-calculated flow field to it to generate adversarial examples. Our method can build adversarial examples with high invisibility and image quality without losing attack performance.</p></sec></sec>
<sec id="s3">
<title>3. Preliminary</title>
<p>Before introducing the details of the proposed framework, in this section, we first present the preliminary knowledge about adversarial attacks and normalizing flows.</p>
<sec>
<title>3.1. Adversarial attack</title>
<p>Given a well-trained DNN classifier <inline-formula><mml:math id="M1"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">C</mml:mi></mml:mrow></mml:math></inline-formula> and a correctly classified input (<italic>x, y</italic>)&#x0007E;<italic>D</italic>, we have <inline-formula><mml:math id="M2"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">C</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>y</mml:mi></mml:math></inline-formula>, where <italic>D</italic> denotes the accessible dataset. The adversarial example <italic>x</italic><sub><italic>adv</italic></sub> is a neighbor of <italic>x</italic> and satisfies that <inline-formula><mml:math id="M3"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">C</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02260;</mml:mo><mml:mi>y</mml:mi></mml:math></inline-formula> and ||<italic>x</italic><sub><italic>adv</italic></sub>&#x02212;<italic>x</italic>||<sub><italic>p</italic></sub> &#x02264; &#x003F5;, where the &#x02113;<sub><italic>p</italic></sub> norm is used as the metric function and &#x003F5; is usually a small value such as 8 and 16 with the image intensity [0, 255]. With this definition, the problem of calculating an adversarial example becomes a constrained optimization problem:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M4"><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>g</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mo>&#x02016;</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02212;</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo>&#x02016;</mml:mo></mml:mrow></mml:mrow><mml:mi>p</mml:mi></mml:msub><mml:mo>&#x02264;</mml:mo><mml:mi>&#x003F5;</mml:mi></mml:mrow></mml:munder><mml:mo stretchy='false'>(</mml:mo><mml:mi mathvariant='script'>C</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02260;</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>Where &#x02113; stands for a loss function that measures the confidence of the model outputs.</p>
<p>In the optimization-based methods, the above problem is solved by computing the gradients of the loss function in Equation (1) to generate the adversarial example. Furthermore, most traditional attack methods craft adversarial examples by optimizing a noise &#x003B4; and adding it to the clean image, i.e., <italic>x</italic><sub><italic>adv</italic></sub> &#x0003D; <italic>x</italic>&#x0002B;&#x003B4;. By contrast, in this work, we formulate the <italic>x</italic><sub><italic>adv</italic></sub> by disturbing the image&#x00027;s latent representation with spatial transform techniques.</p></sec>
<sec>
<title>3.2. Normalizing flow</title>
<p>The normalizing flows (Dinh et al., <xref ref-type="bibr" rid="B13">2015</xref>; Kingma and Dhariwal, <xref ref-type="bibr" rid="B29">2018</xref>; Xu H. et al., <xref ref-type="bibr" rid="B55">2020</xref>) are a class of probabilistic generative models, which are constructed based on a series of entirely reversible components. The reversible property allows to transform from the original distribution to a new one and vice versa. By optimizing the model, a simple distribution (such as the Gaussian distribution) can be transformed into a complex distribution of real data. The training process of normalizing flows is indeed an explicit likelihood maximization. Considering that the model is expressed by a fully invertible and differentiable function that transfers a random vector <italic>z</italic> from the Gaussian distribution to another vector <italic>x</italic>, we can employ such a model to generate high dimensional and complex data.</p>
<p>&#x022AE; Specifically, given a reversible function <italic>F</italic>:&#x0211D;<sup><italic>d</italic></sup> &#x02192; &#x0211D;<sup><italic>d</italic></sup> and two random variables <italic>z</italic>&#x0007E;<italic>p</italic>(<italic>z</italic>) and <italic>z</italic>&#x02032;&#x0007E;<italic>p</italic>(<italic>z</italic>&#x02032;) where <italic>z</italic>&#x02032; &#x0003D; <italic>f</italic>(<italic>z</italic>), the change of variable rule tells that</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M5"><mml:mrow><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msup><mml:mi>z</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>z</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>d</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mfrac><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:msup><mml:mstyle mathvariant="bold-italic"><mml:mtext>F</mml:mtext></mml:mstyle><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:msup><mml:mi>z</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup></mml:mrow></mml:mfrac></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<disp-formula id="E3"><label>(3)</label><mml:math id="M6"><mml:mrow><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>z</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msup><mml:mi>z</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>d</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mfrac><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:mstyle mathvariant="bold-italic"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:mfrac></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>Where <italic>det</italic> denotes the determinant operation. The above equation follows a chaining rule, in which a series of invertible mappings can be chained to approximate a sufficiently complex distribution, i.e.,</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02299;</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>&#x02299;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02299;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where each <italic>F</italic> is a reversible function called a flow step. Equation (4) is the shorthand of <italic>F</italic><sub><italic>K</italic></sub>(<italic>F</italic><sub><italic>k</italic>&#x02212;1</sub>(&#x02026;<italic>F</italic><sub>1</sub>(<italic>x</italic>))). Assuming that <italic>x</italic> is the observed example and <italic>z</italic> is the hidden representation, we write the generative process as</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>x</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where <italic>F</italic><sub>&#x003B8;</sub> is the accumulate sum of all <italic>F</italic> in Equation (4). Based on the change-of-variables theorem, we write the log-density function of <italic>x</italic> &#x0003D; <italic>z</italic><sub><italic>K</italic></sub> as follows:</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M9"><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mi>log</mml:mi><mml:msub><mml:mi>p</mml:mi><mml:mi>K</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mi>K</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>log</mml:mi><mml:msub><mml:mi>p</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>K</mml:mi></mml:munderover><mml:mrow><mml:mi>log</mml:mi></mml:mrow></mml:mstyle><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>d</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mfrac><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mrow><mml:mi>k</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>Where we use <italic>z</italic><sub><italic>k</italic></sub> &#x0003D; <italic>F</italic><sub><italic>k</italic></sub>(<italic>z</italic><sub><italic>k</italic>&#x02212;1</sub>) implicitly. The training process of normalizing flow is minimizing the above function, which exactly maximizes the likelihood of the observed training data. Hence, the optimization is stable and easy to implement.</p></sec>
<sec>
<title>3.3. Spatial transform</title>
<p>The concept of spatial transform is firstly mentioned in Fawzi and Frossard (<xref ref-type="bibr" rid="B19">2015</xref>), which indicates that the conventional neural networks are not robust to rotation, translation and dilation. Next, Xiao et al. (<xref ref-type="bibr" rid="B54">2018</xref>) utilized the spatial transform techniques and proposed the stAdv to craft adversarial examples with a high fooling rate and perceptually realistic beyond noise-adding way. StAdv changes each pixel position in the clean image by applying a well-optimized flow field matrix to the original image. Later, Zhang et al. (<xref ref-type="bibr" rid="B60">2020</xref>) proposed a new method to produce the universal adversarial examples by combining the spatial transform and pixel distortion, and it successfully increased the attack success rate against universal perturbation to more than 90%. In the literature (Aydin et al., <xref ref-type="bibr" rid="B2">2021</xref>), the authors applied spatial transform to the YUV space to generate adversarial examples with higher superiority in image quality.</p>
<p>We summarized the adopted symbols in <xref ref-type="table" rid="T1">Table 1</xref> to increase the readability.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>The notations used in this paper.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold><italic>x</italic></bold></th>
<th valign="top" align="left"><bold>clean example</bold></th>
<th valign="top" align="left"><bold><inline-formula><mml:math id="M10"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">C</mml:mi></mml:mrow></mml:math></inline-formula></bold></th>
<th valign="top" align="left"><bold>the classifier</bold></th>
<th valign="top" align="left"><bold><italic>z</italic><sub><italic>adv</italic></sub></bold></th>
<th valign="top" align="left"><bold>the disturbed latent value</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><italic>x</italic><sub><italic>adv</italic></sub></td>
<td valign="top" align="left">adversarial example</td>
<td valign="top" align="left"><inline-formula><mml:math id="M11"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">L</mml:mi></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="left">loss function</td>
<td valign="top" align="left">&#x003B4;</td>
<td valign="top" align="left">the noise</td>
</tr> <tr>
<td valign="top" align="left"><italic>y</italic></td>
<td valign="top" align="left">clean label</td>
<td valign="top" align="left"><italic>F</italic></td>
<td valign="top" align="left">Pretrained Flow Model</td>
<td valign="top" align="left"><italic>f</italic></td>
<td valign="top" align="left">the flow field</td>
</tr> <tr>
<td valign="top" align="left"><italic>t</italic></td>
<td valign="top" align="left">the target label</td>
<td valign="top" align="left"><italic>z</italic></td>
<td valign="top" align="left">the latent value</td>
<td valign="top" align="left"><inline-formula><mml:math id="M12"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>&#x000B7;</mml:mo></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="left">the four neighborhood</td>
</tr>
</tbody>
</table>
</table-wrap></sec></sec>
<sec id="s4">
<title>4. Methodology</title>
<p>In this section, we propose our attack method. First, we take an overview of our method. Next, we go over the detail of each part step by step. Finally, we discuss our objective function and summarize the whole process as <xref ref-type="table" rid="T8">Algorithm 1</xref>.</p>
<table-wrap position="float" id="T8">
<label>Algorithm 1</label>
<caption><p>DualFlow attack.</p></caption>
<table frame="hsides" rules="groups">
<tbody>
<tr><td align="left" valign="top"><bold>Input</bold>: &#x000A0;<italic>X</italic><sub><italic>tr</italic></sub>: a batch of clean examples used for training; &#x003B1;: the learning rate; <italic>T</italic>: the maximal training iterations; <italic>Q</italic>: the maximal steps for attack; &#x003BE;: the flow budget; <italic>X</italic><sub><italic>te</italic></sub>: a clean example used for test; <inline-formula><mml:math id="M27"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">C</mml:mi></mml:mrow></mml:math></inline-formula>: the target model to be attacked.</td></tr>
<tr><td align="left" valign="top"><bold>Output</bold>: &#x000A0;The adversarial example <italic>x</italic><sub><italic>adv</italic></sub> is used for attack.</td></tr>
<tr><td align="left" valign="top"><bold>Parameter</bold>: &#x000A0;The flow model <italic>F</italic><sub>&#x003B8;</sub>.</td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;1: &#x000A0;Initialize the parameters of the flow model <italic>F</italic><sub>&#x003B8;</sub>;</td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;2: &#x000A0;for <italic>i</italic> &#x0003D; 1 to <italic>T</italic> <bold>do</bold></td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;3: &#x000A0;&#x000A0;&#x000A0;Optimize <italic>F</italic><sub>&#x003B8;</sub> according to Equation (6);</td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;4: &#x000A0;&#x000A0;&#x000A0;<bold>if</bold> Convergence reached <bold>then</bold></td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;5: &#x000A0;&#x000A0;&#x000A0;&#x000A0;break;</td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;6: &#x000A0;&#x000A0;<bold>end if</bold></td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;7: &#x000A0;<bold>end for</bold></td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;8: &#x000A0;Obtain optimized <italic>F</italic><sub>&#x003B8;</sub>;</td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;9: &#x000A0;Compute the hidden representation of examples in <italic>X</italic><sub><italic>te</italic></sub> via <inline-formula><mml:math id="M28"><mml:mi>z</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>;</td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;10: &#x000A0;<inline-formula><mml:math id="M29"><mml:msubsup><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:msup><mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>z</mml:mi></mml:math></inline-formula></td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;11: &#x000A0;Initialize the flow filed <italic>f</italic> with zeros;</td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;12: &#x000A0;<bold>for</bold> <italic>i</italic> &#x0003D; 1 to <italic>Q</italic> <bold>do</bold></td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;13: &#x000A0;&#x000A0;&#x000A0; Optimize <italic>f</italic> via Equations (12) or 13;</td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;14: &#x000A0;&#x000A0;&#x000A0; Compute the adversarial example candidate <inline-formula><mml:math id="M30"><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msubsup></mml:math></inline-formula> via Equation (11);</td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;15: &#x000A0;&#x000A0;&#x000A0; <bold>if</bold> Successfully attack <inline-formula><mml:math id="M31"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">C</mml:mi></mml:mrow></mml:math></inline-formula> by <inline-formula><mml:math id="M32"><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msubsup></mml:math></inline-formula> <bold>then</bold></td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;16: &#x000A0;&#x000A0;&#x000A0; <inline-formula><mml:math id="M33"><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msubsup></mml:math></inline-formula></td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;17: &#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0; break.</td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;18: &#x000A0;&#x000A0;&#x000A0;&#x000A0;<bold>end if</bold></td></tr>
<tr><td align="left" valign="top">&#x000A0;&#x000A0;19: &#x000A0;&#x000A0;&#x000A0;<bold>end for</bold></td></tr>
</tbody>
</table>
</table-wrap><sec>
<title>4.1. The DualFlow framework</title>
<p>The proposed DualFlow attack framework can be divided into three parts, the first one is to map clean image <italic>x</italic> to its latent space <italic>z</italic> by the well-trained normalizing flow model. The second part is to optimize the flow field <italic>f</italic>, and apply it to the images&#x00027; latent representation <italic>z</italic> and inverse the transformed <italic>z</italic> to generate its corresponding RGB space counterpart <italic>x</italic><sub><italic>t</italic></sub>. Note that step 2 needs to be worked in an iterative manner to update the flow field <italic>f</italic> guided by the adv_loss until the adversarial candidate <italic>x</italic><sub><italic>t</italic></sub> can fool the target model. Finally, apply the optimized flow field <italic>f</italic> to the image&#x00027;s latent counterpart <italic>z</italic> and do the inverse operation of normalizing flow to obtain the adversarial image. The whole process is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>The framework of proposed DualFlow. <italic>x</italic> represent the image, among them, <italic>x</italic><sub>0</sub> is the benign image, <italic>x</italic><sub><italic>adv</italic></sub> is the corresponding adversarial counterpart; <italic>z</italic> is the hidden representation of the image; <italic>F</italic> is the well-trained Normalize Flow model and <inline-formula><mml:math id="M13"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">C</mml:mi></mml:mrow></mml:math></inline-formula> is the pre-trained classifier; <italic>f</italic> is the flow field need to be optimized and &#x02297; represents the spatial transform operation.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-17-1129720-g0002.tif"/>
</fig></sec>
<sec>
<title>4.2. Normalizing flow model training</title>
<p>As introduced in Section 3.2., the training of the normalizing flow is to maximize the likelihood function on the training data with respect to the model parameters. Formally, assume that the collected dataset is denoted by <italic>x</italic>&#x0007E;<italic>X</italic>. The hidden representation follows the Gaussian distribution, i.e., <inline-formula><mml:math id="M14"><mml:mi>z</mml:mi><mml:mo>&#x0007E;</mml:mo><mml:mrow><mml:mi mathvariant="-tex-caligraphic">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>. The flow model is denoted by <italic>F</italic>, parameterized &#x003B8;, which have <italic>x</italic> &#x0003D; <italic>F</italic><sub>&#x003B8;</sub>(<italic>z</italic>) and <italic>z</italic> &#x0003D; <italic>F</italic><sup>&#x02212;1</sup>(<italic>x</italic>). Then, the loss function to be minimized is expressed as:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M15"><mml:mrow><mml:mi>L</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x003B8;</mml:mi><mml:mo>;</mml:mo><mml:mi>z</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>log</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>x</mml:mi><mml:mo>&#x0007C;</mml:mo><mml:mi>z</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003B8;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>log</mml:mi><mml:msub><mml:mi>p</mml:mi><mml:mi>z</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mstyle mathvariant="bold-italic"><mml:mtext>F</mml:mtext></mml:mstyle><mml:mi>&#x003B8;</mml:mi><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:mo stretchy='false'>(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>log</mml:mi><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>d</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mfrac><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:msubsup><mml:mstyle mathvariant="bold-italic"><mml:mtext>F</mml:mtext></mml:mstyle><mml:mi>&#x003B8;</mml:mi><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:mo stretchy='false'>(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:mi>x</mml:mi></mml:mrow></mml:mfrac></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>By optimizing the above objective, the learned distribution <italic>p</italic>(<italic>x</italic>|<italic>z</italic>, &#x003B8;) characterizes the data distribution as expected.</p>
<p>In the training process, we use the Adam algorithm to optimize the model parameters; while the learning rate is set as 10<sup>&#x02212;4</sup>, the momentum is set to 0.999, and the maximal iteration number is 100,000.</p></sec>
<sec>
<title>4.3. Generating adversarial examples with DualFlow</title>
<p>For a clean image <italic>x</italic>, to obtain its corresponding adversarial example <italic>x</italic><sub><italic>adv</italic></sub>, we first calculate its corresponding latent space vector <italic>z</italic> by performing a forward flow process <italic>via</italic> <italic>z</italic> &#x0003D; <italic>F</italic><sub>&#x003B8;</sub>(<italic>x</italic>). Once the <italic>z</italic> is calculated, we can disturb it with the spatial transform techniques, the core is to optimize the flow filed vector <italic>f</italic>, which will be applied to <italic>z</italic> to get the transformed latent representation <italic>z</italic><sub><italic>st</italic></sub> according to <italic>x</italic>. In this paper, the flow filed vector <italic>f</italic> is directly optimized with the Adam optimizer iteratively. We will repeat the above process to optimize flow field <italic>f</italic> until <italic>z</italic><sub><italic>st</italic></sub> becomes an eligible adversarial latent value, that is, make the <italic>z</italic><sub><italic>st</italic></sub> becomes <italic>z</italic><sub><italic>adv</italic></sub>. Finally, when the optimal flow filed <italic>f</italic> is calculated, we restore the transformed latent representation <italic>z</italic><sub><italic>adv</italic></sub> to the image space through the inverse operation of the normalizing flow model, that is, <italic>x</italic><sub><italic>adv</italic></sub> &#x0003D; <italic>F</italic><sub>&#x003B8;</sub>(<italic>z</italic><sub><italic>adv</italic></sub>), to get its perturbed example <italic>x</italic><sub><italic>adv</italic></sub> in pixel level.</p>
<p>Moore specifically, the spatial transform techniques using a flow field matrix <italic>f</italic> &#x0003D; [2, <italic>h, w</italic>] to transform the original image <italic>x</italic> to <italic>x</italic><sub><italic>st</italic></sub> (Xiao et al., <xref ref-type="bibr" rid="B54">2018</xref>). In this paper, we adopt the spatial transform from the pixel level to the latent space. Specifically, assume the latent representation of input <italic>x</italic> is <italic>z</italic> and its transformed counterpart <italic>z</italic><sub><italic>st</italic></sub>, for the <italic>i</italic>-th value in <italic>z</italic><sub><italic>st</italic></sub> at the value location <inline-formula><mml:math id="M16"><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, we need to calculate the flow field matrix <inline-formula><mml:math id="M17"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x00394;</mml:mi><mml:msup><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>&#x00394;</mml:mi><mml:msup><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>. So, the <italic>i</italic>-th value <italic>z</italic><sup><italic>i</italic></sup>&#x00027;s location in the transformed image can be indicated as:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M18"><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msup><mml:mi>u</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>v</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>u</mml:mi><mml:mrow><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msubsup><mml:mo>+</mml:mo><mml:mi>&#x00394;</mml:mi><mml:msup><mml:mi>u</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>v</mml:mi><mml:mrow><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msubsup><mml:mo>+</mml:mo><mml:mi>&#x00394;</mml:mi><mml:msup><mml:mi>v</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula>
<p>To ensure the flow field <italic>f</italic> is differentiable, the bi-linear interpolation (Jaderberg et al., <xref ref-type="bibr" rid="B27">2015</xref>) is used to obtain the four neighboring values surrounding the location <inline-formula><mml:math id="M19"><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x00394;</mml:mi><mml:msup><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x00394;</mml:mi><mml:msup><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> for the transformed latent value <italic>z</italic><sub><italic>st</italic></sub> as:</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M20"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mrow><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>q</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi mathvariant='script'>N</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msup><mml:mi>u</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>v</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:munder><mml:mrow><mml:msup><mml:mi>z</mml:mi><mml:mi>q</mml:mi></mml:msup></mml:mrow></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mo>&#x0007C;</mml:mo><mml:msup><mml:mi>u</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mi>u</mml:mi><mml:mi>q</mml:mi></mml:msup><mml:mo>&#x0007C;</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mo>&#x0007C;</mml:mo><mml:msup><mml:mi>v</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mi>v</mml:mi><mml:mi>q</mml:mi></mml:msup><mml:mo>&#x0007C;</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>Where <inline-formula><mml:math id="M21"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> is the neighborhood, that is, the four positions (top-left, top-right, bottom-left, bottom-right) tightly surrounding the target value (<italic>u</italic><sup><italic>i</italic></sup>, <italic>v</italic><sup><italic>i</italic></sup>). In our adversarial attack settings, the calculated <italic>z</italic><sub><italic>st</italic></sub> is the final adversarial latent representation <italic>z</italic><sub><italic>adv</italic></sub>. Once the <italic>f</italic> has been computed, we can obtain the <italic>z</italic><sub><italic>adv</italic></sub> by applying the calculated flow field <italic>f</italic> to the original <italic>z</italic>, which is given by:</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M22"><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>q</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi mathvariant='script'>N</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msup><mml:mi>u</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>v</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:munder><mml:mrow><mml:msup><mml:mi>z</mml:mi><mml:mi>q</mml:mi></mml:msup></mml:mrow></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mo>&#x0007C;</mml:mo><mml:msup><mml:mi>u</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mi>u</mml:mi><mml:mi>q</mml:mi></mml:msup><mml:mo>&#x0007C;</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mo>&#x0007C;</mml:mo><mml:msup><mml:mi>v</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mi>v</mml:mi><mml:mi>q</mml:mi></mml:msup><mml:mo>&#x0007C;</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>and the adversarial examples <italic>x</italic><sub><italic>adv</italic></sub> can be obtained by:</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M23"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where <italic>clip</italic>(&#x000B7;) is the clip operation to keep the generated value belonging to [0, 1].</p></sec>
<sec>
<title>4.4. Objective functions</title>
<p>Taking the attack success rate and visual invisibility of the generated adversarial examples into account, we divide the objective function into two parts, where one is the adversarial loss and the other is a constraint for the flow field. Unlike other flow field-based attack methods, which constrain the flow field by the flow loss proposed in Xiao et al. (<xref ref-type="bibr" rid="B54">2018</xref>), in our method, we use a dynamically updated flow field budget &#x003BE; (a small number, like 1&#x0002A;10<sup>&#x02212;3</sup>) to regularize the flow field <italic>f</italic>. For adversarial attacks, the goal is making <inline-formula><mml:math id="M24"><mml:mrow><mml:mi mathvariant="-tex-caligraphic">C</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02260;</mml:mo><mml:mi>y</mml:mi></mml:math></inline-formula>. We give the objective function as follows:</p>
<p>for un-targeted attacks:</p>
<disp-formula id="E12"><label>(12)</label><mml:math id="M25"><mml:mrow><mml:msub><mml:mi>&#x02112;</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold-italic"><mml:mtext>f</mml:mtext></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mo stretchy='false'>[</mml:mo><mml:mi mathvariant='script'>C</mml:mi><mml:msub><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mi>y</mml:mi></mml:msub><mml:mo>&#x02212;</mml:mo><mml:munder><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>&#x02260;</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:munder><mml:mi mathvariant='script'>C</mml:mi><mml:msub><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy='false'>]</mml:mo><mml:mo>,</mml:mo><mml:mtext>&#x000A0;&#x000A0;</mml:mtext><mml:mi>s</mml:mi><mml:mo>.</mml:mo><mml:mi>t</mml:mi><mml:mo>.</mml:mo><mml:mo>&#x02016;</mml:mo><mml:mstyle mathvariant="bold-italic"><mml:mtext>f</mml:mtext></mml:mstyle><mml:mo>&#x02016;</mml:mo><mml:mo>&#x02264;</mml:mo><mml:mi>&#x003BE;</mml:mi><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula>
<p>for target attacks:</p>
<disp-formula id="E13"><label>(13)</label><mml:math id="M26"><mml:mrow><mml:msub><mml:mi>&#x02112;</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold-italic"><mml:mtext>f</mml:mtext></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mo stretchy='false'>[</mml:mo><mml:munder><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:munder><mml:mi mathvariant='script'>C</mml:mi><mml:msub><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo>&#x02212;</mml:mo><mml:mi mathvariant='script'>C</mml:mi><mml:msub><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mi>y</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy='false'>]</mml:mo><mml:mo>,</mml:mo><mml:mtext>&#x000A0;&#x000A0;</mml:mtext><mml:mi>s</mml:mi><mml:mo>.</mml:mo><mml:mi>t</mml:mi><mml:mo>.</mml:mo><mml:mo>&#x02016;</mml:mo><mml:mstyle mathvariant="bold-italic"><mml:mtext>f</mml:mtext></mml:mstyle><mml:mo>&#x02016;</mml:mo><mml:mo>&#x02264;</mml:mo><mml:mi>&#x003BE;</mml:mi><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula>
<p>The whole algorithm of LFFA is listed in <xref ref-type="table" rid="T8">Algorithm 1</xref> for easy reproducing of our results, where lines 11-18 depict the core optimization process.</p>
</sec></sec>
<sec id="s5">
<title>5. Experiments</title>
<p>In this section, we evaluate the proposed DualFlow on three benchmark image classification datasets. We first compare our proposed method with several baseline techniques concerned with Attack Success Rate (ASR) on clean models and robust models on three CV baseline datasets (CIFAR-10, CIFAR-100 and ImageNet). Then, we first provide a comparative experiment to the existing attack methods in image quality aspects with regard to LPIPS, DISTS, SCC, SSIM, VIPF and et al. Through these experimental results, we show the superiority of our method in attack ability, human inception and image quality.</p>
<sec>
<title>5.1. Settings</title>
<sec>
<title>Dataset</title>
<p>We verify the performance of our method on three benchmark datasets for computer vision task, named CIFAR-10<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref> (Krizhevsky and Hinton, <xref ref-type="bibr" rid="B30">2009</xref>), CIFAR-100<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref> (Krizhevsky and Hinton, <xref ref-type="bibr" rid="B30">2009</xref>) and ImageNet-1k<xref ref-type="fn" rid="fn0002"><sup>2</sup></xref> (Deng et al., <xref ref-type="bibr" rid="B10">2009</xref>). In detail, CIFAR-10 contains 50,000 training images and 10,000 testing images with the size of 3x32x32 from 10 classes; CIFAR-100 has 100 classes, including the same number of training and testing images as the CIFAR-10; ImageNet-1K has 1,000 categories, containing about 1.3M samples for training and 50,000 samples for validation. In particular, in this paper, we extend our attack on the whole images in testing datasets of CIFAR-10 and CIFAR-100, in terms of ImageNet-1k, we are using its subset datasets from ImageNet Adversarial Learning Challenge, which is commonly used in work related to adversarial attacks.</p>
<p>All the experiments are conducted on a GPU server with 4 &#x0002A; Tesla A100 40GB GPU, 2 &#x0002A; Xeon Glod 6112 CPU, and RAM 512GB.</p></sec>
<sec>
<title>Models</title>
<p>For CIFAR-10 and CIFAR-100, the pre-trained VGG-19 (Simonyan and Zisserman, <xref ref-type="bibr" rid="B49">2015</xref>), ResNet-56 (He et al., <xref ref-type="bibr" rid="B22">2016</xref>), MobileNet-V2 (Sandler et al., <xref ref-type="bibr" rid="B45">2018</xref>) and ShuffleNet-V2 (Ma N. et al., <xref ref-type="bibr" rid="B39">2018</xref>) are adopted, with top-1 classification accuracy 93.91, 94.37, 93.91, and 93.98% on CIFAR-10 and 73.87, 72.60, 71.13, and 75.49% on CIFAR-100, respectively, all the models&#x00027; parameters are provided in the GitHub Repository<xref ref-type="fn" rid="fn0003"><sup>3</sup></xref>. For ImageNet, we use the PyTorch pre-trained clean model VGG-16, VGG-19 (Simonyan and Zisserman, <xref ref-type="bibr" rid="B49">2015</xref>), ResNet-152 (He et al., <xref ref-type="bibr" rid="B22">2016</xref>), MobileNet-V2 (Sandler et al., <xref ref-type="bibr" rid="B45">2018</xref>) and DenseNet-121 (Huang et al., <xref ref-type="bibr" rid="B24">2017</xref>), achieving 87.40, 89.00, 94.40, 87.80, and 91.60% classification accuracy rate on ImageNet, respectively. And in terms of robust models, they include Hendrycks2019Using (Hendrycks et al., <xref ref-type="bibr" rid="B23">2019</xref>), Wu2020Adversarial (Wu et al., <xref ref-type="bibr" rid="B53">2020</xref>), Chen2020Efficient (Chen et al., <xref ref-type="bibr" rid="B7">2022</xref>) and Rice2020Overfitting (Rice et al., <xref ref-type="bibr" rid="B43">2020</xref>) for CIFAR-10 and CIFAR-100, And Engstrom2019Robustness (Croce et al., <xref ref-type="bibr" rid="B8">2021</xref>), Salman2020Do_R18 (Salman et al., <xref ref-type="bibr" rid="B44">2020</xref>), Salman2020Do_R50 (Salman et al., <xref ref-type="bibr" rid="B44">2020</xref>), and Wong2020Fast (Wong et al., <xref ref-type="bibr" rid="B52">2020</xref>) for ImageNet. All the models we use are implemented in the robustbench toolbox<xref ref-type="fn" rid="fn0004"><sup>4</sup></xref> (Croce et al., <xref ref-type="bibr" rid="B8">2021</xref>) and the models&#x00027; parameters are also provided in Croce et al. (<xref ref-type="bibr" rid="B8">2021</xref>). For all these models, we chose their <italic>L</italic><sub><italic>inf</italic></sub> version parameters due to most baselines being extended <italic>L</italic><sub><italic>inf</italic></sub> attacks in this paper.</p></sec>
<sec>
<title>Baselines</title>
<p>The baseline methods are FGSM (Goodfellow et al., <xref ref-type="bibr" rid="B20">2015</xref>), MI-FGSM (Dong et al., <xref ref-type="bibr" rid="B15">2018</xref>), TI-FGSM (Dong et al., <xref ref-type="bibr" rid="B16">2019</xref>), Jitter (Schwinn et al., <xref ref-type="bibr" rid="B46">2021</xref>), stAdv (Xiao et al., <xref ref-type="bibr" rid="B54">2018</xref>), Chroma-Shift (Aydin et al., <xref ref-type="bibr" rid="B2">2021</xref>), and GUAP (Zhang et al., <xref ref-type="bibr" rid="B60">2020</xref>). The experimental results of those methods are reproduced by the Torchattacks toolkit<xref ref-type="fn" rid="fn0005"><sup>5</sup></xref> and the code provided by the authors with default settings.</p></sec>
<sec>
<title>Metrics</title>
<p>Unlike the pixel-based attack methods, which only use <italic>L</italic><sub><italic>p</italic></sub> norm to evaluate the adversarial examples&#x00027; perceptual similarity to its corresponding benign image. The adversarial examples generated by spatial transform always use other metrics referring to image quality. To be exact, in this paper, we follow the work in Aydin et al. (<xref ref-type="bibr" rid="B2">2021</xref>) using the following perceptual metrics to evaluate the adversarial examples generated by our method, including Learned Perceptual Image Patch Similarity (LPIPS) metric (Zhang et al., <xref ref-type="bibr" rid="B59">2018</xref>) and Deep Image Structure and Texture Similarity (DISTS) index (Ding et al., <xref ref-type="bibr" rid="B12">2022</xref>). LPIPS is a technique that measures the Euclidean distance of deep representations (i.e., VGG network Simonyan and Zisserman, <xref ref-type="bibr" rid="B49">2015</xref>) calibrated by human perception. LPIPS has already been used on spatially transformed adversarial examples generating studies (Jordan et al., <xref ref-type="bibr" rid="B28">2019</xref>; Laidlaw and Feizi, <xref ref-type="bibr" rid="B32">2019</xref>; Aydin et al., <xref ref-type="bibr" rid="B2">2021</xref>). DISTS is a method that combines texture similarity with structure similarity (i.e., feature maps) using deep networks with the optimization of human perception. We used the implementation of Ding et al. for both perceptual metrics (Ding et al., <xref ref-type="bibr" rid="B11">2021</xref>). Moreover, we use other metrics like Spatial Correlation Coefficient (SCC) (Li, <xref ref-type="bibr" rid="B34">2000</xref>), Structure Similarity Index Measure (SSIM) and Pixel Based Visual Information Fidelity (VIFP) (Sheikh and Bovik, <xref ref-type="bibr" rid="B48">2004</xref>) to assess the generated images&#x00027; qualities. SCC reflects the indirect correlation based on the spatial contiguity between any two geographical entities. SSIM is used to assess the generated images&#x00027; qualities concerning luminance, contrast and structure. VIFP is used to assess the adversarial examples&#x00027; image quality. The primary toolkits we used in the experiments of this part are IQA_pytorch<xref ref-type="fn" rid="fn0006"><sup>6</sup></xref> and sewar<xref ref-type="fn" rid="fn0007"><sup>7</sup></xref>.</p></sec></sec>
<sec>
<title>5.2. Quantitative comparison with the existing attacks</title>
<p>In this subsection, we will evaluate the proposed DualFlow and the baselines FGSM, MI-FGSM, TI-FGSM (Dong et al., <xref ref-type="bibr" rid="B16">2019</xref>), Jitter, stAdv, Chroma-shift and GUAP in attack success rate on CIFAR-10, CIFAR-100 and the whole ImageNet dataset. We set the noise budget as &#x003F5; &#x0003D; 0.031 for all <italic>L</italic><sub><italic>inf</italic></sub>-based attacks baseline methods. The other attack methods, such as stAdv and Chroma-shift, follow their default settings in the code provided by the authors.</p>
<p><xref ref-type="table" rid="T2">Tables 2</xref>&#x02013;<xref ref-type="table" rid="T4">4</xref> show the ASR of DualFlow and the baselines on CIFAR-10, CIFAR-100 and ImageNet, respectively. As the results illustrated, DualFlow can perform better in most situations on the three benchmark datasets. Take the attack results on ImageNet as an example, refer to <xref ref-type="table" rid="T3">Table 3</xref>. The BIM, MI-FGSM, TI-FGSM, Jitter, stAdv, Chroma-shift and GUAP can achieve 91.954, 98.556, 93.94, 95.172, 97.356, 98.678, and 94.606% average attack success rate on ImageNet dataset, respectively, vice versa, our DualFlow can achieve 99.364% average attack success rate. On the other two benchmark datasets, CIFAR-10 and CIFAR-100, the DualFlow also can get a better average attack performance. To further explore the attack performance of the proposed DualFlow, we also extend the targeted attack on ImageNet, and the results are presented in <xref ref-type="table" rid="T4">Table 4</xref>. The empirical results show that DualFlow can generate more powerful adversarial examples and obtain a superior attack success rate in most cases. It can get an ASR range from 94.12 to 99.52% on five benchmark DL models, but the most competitive baseline MI-FGSM can achieve an ASR of 83.90 to 99.34%. It is indicated that the proposed method is more threatening to DNNs and meaningful for exploring the existing DNNs&#x00027; vulnerability and guiding the new DNNs&#x00027; design.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Experimental results on attack success rate (ASR) of un-targeted attack of CIFAR-10 and CIFAR-100.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919497">
<th/>
<th valign="top" align="center" colspan="4"><bold>CIFAR-10</bold></th>
<th valign="top" align="center" colspan="4"><bold>CIFAR-100</bold></th>
</tr>
</thead>
<tbody>
 <tr style="background-color:#919497">
<td/>
<td valign="top" align="center"><bold>VGG19</bold></td>
<td valign="top" align="center"><bold>ResNet56</bold></td>
<td valign="top" align="center"><bold>MobileNetV2</bold></td>
<td valign="top" align="center"><bold>ShuffleNetV2</bold></td>
<td valign="top" align="center"><bold>VGG19</bold></td>
<td valign="top" align="center"><bold>ResNet56</bold></td>
<td valign="top" align="center"><bold>MobileNetV2</bold></td>
<td valign="top" align="center"><bold>ShuffleNetV2</bold></td>
</tr> <tr>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="center">55.28</td>
<td valign="top" align="center">65.58</td>
<td valign="top" align="center">71.46</td>
<td valign="top" align="center">54.85</td>
<td valign="top" align="center">75.42</td>
<td valign="top" align="center">91.23</td>
<td valign="top" align="center">90.40</td>
<td valign="top" align="center">85.72</td>
</tr> <tr>
<td valign="top" align="left">MI-FGSM</td>
<td valign="top" align="center">76.43</td>
<td valign="top" align="center">93.11</td>
<td valign="top" align="center">94.12</td>
<td valign="top" align="center">78.47</td>
<td valign="top" align="center">87.69</td>
<td valign="top" align="center">99.78</td>
<td valign="top" align="center">99.47</td>
<td valign="top" align="center">93.68</td>
</tr> <tr>
<td valign="top" align="left">TI-FGSM</td>
<td valign="top" align="center">59.63</td>
<td valign="top" align="center">71.03</td>
<td valign="top" align="center">80.01</td>
<td valign="top" align="center">76.10</td>
<td valign="top" align="center">83.43</td>
<td valign="top" align="center">97.46</td>
<td valign="top" align="center">93.92</td>
<td valign="top" align="center">92.77</td>
</tr> <tr>
<td valign="top" align="left">Jitter</td>
<td valign="top" align="center">83.70</td>
<td valign="top" align="center">94.87</td>
<td valign="top" align="center"><bold>96.92</bold></td>
<td valign="top" align="center">86.25</td>
<td valign="top" align="center">98.31</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center"><bold>99.76</bold></td>
<td valign="top" align="center">94.63</td>
</tr> <tr>
<td valign="top" align="left">stAdv</td>
<td valign="top" align="center">86.04</td>
<td valign="top" align="center">63.77</td>
<td valign="top" align="center">69.43</td>
<td valign="top" align="center">66.11</td>
<td valign="top" align="center">97.66</td>
<td valign="top" align="center">93.26</td>
<td valign="top" align="center">93.55</td>
<td valign="top" align="center">95.61</td>
</tr> <tr>
<td valign="top" align="left">Chroma-shift</td>
<td valign="top" align="center">84.87</td>
<td valign="top" align="center">68.36</td>
<td valign="top" align="center">73.57</td>
<td valign="top" align="center">64.58</td>
<td valign="top" align="center">98.84</td>
<td valign="top" align="center">98.37</td>
<td valign="top" align="center">96.39</td>
<td valign="top" align="center">96.86</td>
</tr> <tr>
<td valign="top" align="left">GUAP</td>
<td valign="top" align="center">82.55</td>
<td valign="top" align="center">89.34</td>
<td valign="top" align="center">87.61</td>
<td valign="top" align="center">87.02</td>
<td valign="top" align="center">92.26</td>
<td valign="top" align="center">94.59</td>
<td valign="top" align="center">96.89</td>
<td valign="top" align="center">92.20</td>
</tr> <tr>
<td valign="top" align="left">DualFlow</td>
<td valign="top" align="center"><bold>97.07</bold></td>
<td valign="top" align="center"><bold>95.31</bold></td>
<td valign="top" align="center">93.65</td>
<td valign="top" align="center"><bold>96.19</bold></td>
<td valign="top" align="center"><bold>99.32</bold></td>
<td valign="top" align="center">99.02</td>
<td valign="top" align="center">98.83</td>
<td valign="top" align="center"><bold>97.36</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The victim models are VGG19, ResNet56, MobileNetV2 and ShuffleNetV2, respectively, pre-trained by a GitHub Repository, named pytorch-cifar-models. Note that for the FGSM-based baselines, we synthesize their adversarial examples under <italic>L</italic><sub><italic>inf</italic></sub>-norm=0.031 limitation; the others are not subject to the <italic>L</italic><sub><italic>inf</italic></sub>-norm restrictions. Bold values indicates the best result.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Experimental results on attack success rate (ASR) of un-targeted attack of ImageNet.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919497">
<th/>
<th valign="top" align="left"><bold>GSM</bold></th>
<th valign="top" align="left"><bold>MI-FGSM</bold></th>
<th valign="top" align="left"><bold>TI-FGSM</bold></th>
<th valign="top" align="left"><bold>Jitter</bold></th>
<th valign="top" align="left"><bold>stAdv</bold></th>
<th valign="top" align="center"><bold>Chroma-shift</bold></th>
<th valign="top" align="left"><bold>GUAP</bold></th>
<th valign="top" align="left"><bold>DualFlow</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">VGG16</td>
<td valign="top" align="left">93.56</td>
<td valign="top" align="left">98.64</td>
<td valign="top" align="left">97.16</td>
<td valign="top" align="left">95.27</td>
<td valign="top" align="left">97.62</td>
<td valign="top" align="center">98.62</td>
<td valign="top" align="left">97.73</td>
<td valign="top" align="left"><bold>99.37</bold></td>
</tr> <tr>
<td valign="top" align="left">VGG19</td>
<td valign="top" align="left">95.31</td>
<td valign="top" align="left">99.42</td>
<td valign="top" align="left">96.34</td>
<td valign="top" align="left">91.76</td>
<td valign="top" align="left">98.74</td>
<td valign="top" align="center">98.98</td>
<td valign="top" align="left">96.10</td>
<td valign="top" align="left"><bold>99.43</bold></td>
</tr> <tr>
<td valign="top" align="left">ResNet152</td>
<td valign="top" align="left">84</td>
<td valign="top" align="left">96.82</td>
<td valign="top" align="left">85.17</td>
<td valign="top" align="left">94.28</td>
<td valign="top" align="left">97.46</td>
<td valign="top" align="center">97.79</td>
<td valign="top" align="left">88.90</td>
<td valign="top" align="left"><bold>98.63</bold></td>
</tr> <tr>
<td valign="top" align="left">MobileNetV2</td>
<td valign="top" align="left">91.92</td>
<td valign="top" align="left">98.29</td>
<td valign="top" align="left">91.47</td>
<td valign="top" align="left">94.99</td>
<td valign="top" align="left">96.13</td>
<td valign="top" align="center">99.35</td>
<td valign="top" align="left">97.60</td>
<td valign="top" align="left"><bold>99.61</bold></td>
</tr> <tr>
<td valign="top" align="left">DenseNet121</td>
<td valign="top" align="left">94.98</td>
<td valign="top" align="left">99.61</td>
<td valign="top" align="left">99.56</td>
<td valign="top" align="left">99.56</td>
<td valign="top" align="left">96.83</td>
<td valign="top" align="center">98.65</td>
<td valign="top" align="left">92.70</td>
<td valign="top" align="left"><bold>99.78</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The victim models are VGG19, ResNet152, MobileNetV2 and DenseNet121, respectively, which are pre-trained by PyTorch. Note that for the FGSM-based baselines, we synthesize their adversarial examples under <italic>L</italic><sub><italic>inf</italic></sub>-norm=0.031 limitation; the others are not subject to the <italic>L</italic><sub><italic>inf</italic></sub>-norm restrictions. Bold values indicates the best result.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Experimental results on the attack success rate of targeted attack on dataset ImageNet.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919497">
<th valign="top" align="left"><bold>Methods</bold></th>
<th valign="top" align="left"><bold>FGSM</bold></th>
<th valign="top" align="left"><bold>MI-FGSM</bold></th>
<th valign="top" align="left"><bold>TI-FGSM</bold></th>
<th valign="top" align="left"><bold>Jitter</bold></th>
<th valign="top" align="left"><bold>stAdv</bold></th>
<th valign="top" align="center"><bold>Chroma-Shift</bold></th>
<th valign="top" align="left"><bold>DualFlow</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">VGG16</td>
<td valign="top" align="left">80.78</td>
<td valign="top" align="left">73.11</td>
<td valign="top" align="left">96.34</td>
<td valign="top" align="left">67.51</td>
<td valign="top" align="left">54.74</td>
<td valign="top" align="center">65.10</td>
<td valign="top" align="left"><bold>96.67</bold></td>
</tr> <tr>
<td valign="top" align="left">VGG19</td>
<td valign="top" align="left">60.59</td>
<td valign="top" align="left">49.36</td>
<td valign="top" align="left">83.90</td>
<td valign="top" align="left">46.50</td>
<td valign="top" align="left">53.23</td>
<td valign="top" align="center">55.39</td>
<td valign="top" align="left"><bold>98.85</bold></td>
</tr> <tr>
<td valign="top" align="left">ResNet152</td>
<td valign="top" align="left">80.22</td>
<td valign="top" align="left">73.93</td>
<td valign="top" align="left"><bold>94.72</bold></td>
<td valign="top" align="left">70.45</td>
<td valign="top" align="left">65.87</td>
<td valign="top" align="center">69.60</td>
<td valign="top" align="left">94.12</td>
</tr> <tr>
<td valign="top" align="left">MobileNetV2</td>
<td valign="top" align="left">72.70</td>
<td valign="top" align="left">63.94</td>
<td valign="top" align="left">92.38</td>
<td valign="top" align="left">60.86</td>
<td valign="top" align="left">70.63</td>
<td valign="top" align="center">76.00</td>
<td valign="top" align="left"><bold>99.52</bold></td>
</tr> <tr>
<td valign="top" align="left">DenseNet121</td>
<td valign="top" align="left">78.06</td>
<td valign="top" align="left">74.56</td>
<td valign="top" align="left"><bold>99.34</bold></td>
<td valign="top" align="left">63.86</td>
<td valign="top" align="left">75.94</td>
<td valign="top" align="center">80.79</td>
<td valign="top" align="left">99.06</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The baselines are FGSM, MI-FGSM, TI-FGSM, Jitter, stAdv, Chroma-shift and DualFlow. Note that for the FGSM-based baselines, we synthesize their adversarial examples under <italic>L</italic><sub><italic>inf</italic></sub>-norm=0.031 limitation; the others are not subject to the restrictions. Bold values indicates the best result.</p>
</table-wrap-foot>
</table-wrap></sec>
<sec>
<title>5.3. Attack on defense models</title>
<p>Next, we investigate the performance of the proposed method in attacking robust image classifiers. Thus we select some of the most recent defense techniques that are from the robustbench toolbox as follows, for CIFAR-10 and CIFAR-100 are Hendrycks2019Using (Hendrycks et al., <xref ref-type="bibr" rid="B23">2019</xref>), Wu2020Adversarial (Wu et al., <xref ref-type="bibr" rid="B53">2020</xref>), Chen2020Efficient (Chen et al., <xref ref-type="bibr" rid="B7">2022</xref>) and Rice2020Overfitting (Rice et al., <xref ref-type="bibr" rid="B43">2020</xref>); for ImageNet are Engstrom2019Robustness (Croce et al., <xref ref-type="bibr" rid="B8">2021</xref>), Salman2020Do_R18 (Salman et al., <xref ref-type="bibr" rid="B44">2020</xref>), Salman2020Do_R50 (Salman et al., <xref ref-type="bibr" rid="B44">2020</xref>) and Wong2020Fast (Wong et al., <xref ref-type="bibr" rid="B52">2020</xref>). We compare our proposed method with the baseline methods.</p>
<p>Following the results shown in <xref ref-type="table" rid="T5">Table 5</xref>, we derive that DualFlow exhibits the best performance of all the baseline methods in terms of the attack success rate in most cases. The attack success rate of the baseline method stAdv and Chroma-Shift range from 95.41 to 99.12% and 17.22% from 74.80 in ImageNet, respectively. However, the DualFlow can obtain a higher performance range from 97.50 to 100%. It demonstrates the superiority of our method when attacking robust models.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Experimental results on the attack success rate of un-targeted attack on CIFAR-10, CIFAR-100 and ImageNet dataset to robust models.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919497">
<th valign="top" align="left" colspan="2"></th>
<th valign="top" align="left"><bold>FGSM</bold></th>
<th valign="top" align="left"><bold>MIFGSM</bold></th>
<th valign="top" align="left"><bold>TIFGSM</bold></th>
<th valign="top" align="left"><bold>Jitter</bold></th>
<th valign="top" align="left"><bold>stAdv</bold></th>
<th valign="top" align="left"><bold>Chroma-shift</bold></th>
<th valign="top" align="center"><bold>DualFlow</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="4">CIFAR-10</td>
<td valign="top" align="left">Hendrycks2019Using</td>
<td valign="top" align="left">27.06</td>
<td valign="top" align="left">16.90</td>
<td valign="top" align="left">18.54</td>
<td valign="top" align="left">32.67</td>
<td valign="top" align="left">99.12</td>
<td valign="top" align="center">20.70</td>
<td valign="top" align="left"><bold>100</bold></td>
</tr>
 <tr>
<td valign="top" align="left">Wu2020Adversarial</td>
<td valign="top" align="left">25.63</td>
<td valign="top" align="left">16.28</td>
<td valign="top" align="left">19.10</td>
<td valign="top" align="left">31.02</td>
<td valign="top" align="left">99.12</td>
<td valign="top" align="center">18.36</td>
<td valign="top" align="left"><bold>100</bold></td>
</tr>
 <tr>
<td valign="top" align="left">Chen2020Efficient</td>
<td valign="top" align="left">28.59</td>
<td valign="top" align="left">18.93</td>
<td valign="top" align="left">20.94</td>
<td valign="top" align="left">35.59</td>
<td valign="top" align="left">99.02</td>
<td valign="top" align="center">24.90</td>
<td valign="top" align="left"><bold>100</bold></td>
</tr>
 <tr>
<td valign="top" align="left">Rice2020Overfitting</td>
<td valign="top" align="left">27.38</td>
<td valign="top" align="left">16.87</td>
<td valign="top" align="left">16.92</td>
<td valign="top" align="left">33.02</td>
<td valign="top" align="left">98.93</td>
<td valign="top" align="center">25.98</td>
<td valign="top" align="left"><bold>100</bold></td>
</tr> <tr>
<td valign="top" align="left" rowspan="4">CIFAR-100</td>
<td valign="top" align="left">Hendrycks2019Using</td>
<td valign="top" align="left">37.67</td>
<td valign="top" align="left">25.57</td>
<td valign="top" align="left">28.88</td>
<td valign="top" align="left">48.89</td>
<td valign="top" align="left">95.41</td>
<td valign="top" align="center">35.16</td>
<td valign="top" align="left"><bold>100</bold></td>
</tr>
 <tr>
<td valign="top" align="left">Wu2020Adversarial</td>
<td valign="top" align="left">40.13</td>
<td valign="top" align="left">27.06</td>
<td valign="top" align="left">30.71</td>
<td valign="top" align="left">50.13</td>
<td valign="top" align="left">97.66</td>
<td valign="top" align="center">30.86</td>
<td valign="top" align="left"><bold>100</bold></td>
</tr>
 <tr>
<td valign="top" align="left">Chen2020Efficient</td>
<td valign="top" align="left">42.24</td>
<td valign="top" align="left">30.51</td>
<td valign="top" align="left">34.24</td>
<td valign="top" align="left">54.66</td>
<td valign="top" align="left">97.75</td>
<td valign="top" align="center">34.57</td>
<td valign="top" align="left"><bold>100</bold></td>
</tr>
 <tr>
<td valign="top" align="left">Rice2020Overfitting</td>
<td valign="top" align="left">52.55</td>
<td valign="top" align="left">38.92</td>
<td valign="top" align="left">46.63</td>
<td valign="top" align="left">62.66</td>
<td valign="top" align="left">97.75</td>
<td valign="top" align="center">34.67</td>
<td valign="top" align="left"><bold>100</bold></td>
</tr> <tr>
<td valign="top" align="left" rowspan="4">ImageNet</td>
<td valign="top" align="left">Engstrom2019Robustness</td>
<td valign="top" align="left">62.92</td>
<td valign="top" align="left">51.03</td>
<td valign="top" align="left">65.50</td>
<td valign="top" align="left">83.85</td>
<td valign="top" align="left">95.41</td>
<td valign="top" align="center">22.61</td>
<td valign="top" align="left"><bold>97.50</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Salman2020Do_R18</td>
<td valign="top" align="left">65.61</td>
<td valign="top" align="left">51.82</td>
<td valign="top" align="left">62.44</td>
<td valign="top" align="left">82.09</td>
<td valign="top" align="left">97.66</td>
<td valign="top" align="center">42.16</td>
<td valign="top" align="left"><bold>100</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Salman2020Do_R50</td>
<td valign="top" align="left">57.58</td>
<td valign="top" align="left">44.99</td>
<td valign="top" align="left">55.66</td>
<td valign="top" align="left">76.48</td>
<td valign="top" align="left">97.75</td>
<td valign="top" align="center">17.22</td>
<td valign="top" align="left"><bold>99.19</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Wong2020Fast</td>
<td valign="top" align="left">61.24</td>
<td valign="top" align="left">50.08</td>
<td valign="top" align="left">70.02</td>
<td valign="top" align="left">82.30</td>
<td valign="top" align="left"><bold>97.75</bold></td>
<td valign="top" align="center">74.80</td>
<td valign="top" align="left">97.5</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Bold values indicates the best result.</p>
</table-wrap-foot>
</table-wrap></sec>
<sec>
<title>5.4. Evaluation of human perceptual and image quality</title>
<p>Unlike the noise-adding attack methods, which usually use <italic>L</italic><sub><italic>p</italic></sub> norm to evaluate the victim examples&#x00027; perceptual similarity to its corresponding benign image. The adversarial examples generated by noise-beyond ways always use other metrics referring to image quality. To be exact, we follow the work in Aydin et al. (<xref ref-type="bibr" rid="B2">2021</xref>) using the following perceptual metrics to evaluate the adversarial examples generated by baseline methods and the proposed method, including Learned Perceptual Image Patch Similarity (LPIPS) metric (Zhang et al., <xref ref-type="bibr" rid="B59">2018</xref>) and Deep Image Structure and Texture Similarity (DISTS) index (Ding et al., <xref ref-type="bibr" rid="B12">2022</xref>). In addition, <italic>L</italic><sub><italic>inf</italic></sub>-norm, Spatial Correlation Coefficient (SCC) (Li, <xref ref-type="bibr" rid="B34">2000</xref>), Structure Similarity Index Measure (SSIM) (Wang et al., <xref ref-type="bibr" rid="B51">2004</xref>), and Pixel Based Visual Information Fidelity (VIFP) (Sheikh and Bovik, <xref ref-type="bibr" rid="B48">2004</xref>) are also involved in evaluating the difference between the generated adversarial examples and their benign counterparts and the quality of the generated adversarial examples.</p>
<p>The generated images&#x00027; quality results can be seen in <xref ref-type="table" rid="T6">Table 6</xref>, which indicated that the proposed method has the lowest LPIPS, DISTS perceptual loss and <italic>L</italic><sub><italic>inf</italic></sub> (the lower is better) are 0.0188, 0.0324 and 0.1642, respectively, on VGG-19 model; and has the highest SCC, SSIM and VIFP (the higher is better), achieving 0.9452, 0.7876 and 0.8192, respectively, on VGG-19 model. All the empirical data are obtained on the ImageNet dataset. The results show that the proposed method is superior to the existing attack methods.</p>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Perceptual distances were calculated on fooled examples by FGSM, MI-FGSM, TI-FGSM, Jitter, stAdv, Chroma-shift, GUAP, and the proposed DualFlow on ImageNet.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919497">
<th/>
<th valign="top" align="center" colspan="6"><bold>VGG19</bold></th>
<th valign="top" align="center" colspan="6"><bold>ResNet152</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919497">
<td/>
<td valign="top" align="center"><bold>LPIPS</bold></td>
<td valign="top" align="center"><bold>DISTS</bold></td>
<td valign="top" align="center"><italic>L</italic><sub><italic>inf</italic></sub></td>
<td valign="top" align="center"><bold>SCC</bold></td>
<td valign="top" align="center"><bold>SSIM</bold></td>
<td valign="top" align="center"><bold>VIFP</bold></td>
<td valign="top" align="center"><bold>LPIPS</bold></td>
<td valign="top" align="center"><bold>DISTS</bold></td>
<td valign="top" align="center"><italic>L</italic><sub><italic>inf</italic></sub></td>
<td valign="top" align="center"><bold>SCC</bold></td>
<td valign="top" align="center"><bold>SSIM</bold></td>
<td valign="top" align="center"><bold>VIFP</bold></td>
</tr> <tr>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="center">0.3036</td>
<td valign="top" align="center">0.1916</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">0.5572</td>
<td valign="top" align="center">0.8273</td>
<td valign="top" align="center">0.4705</td>
<td valign="top" align="center">0.2688</td>
<td valign="top" align="center">0.1679</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">0.5796</td>
<td valign="top" align="center">0.8348</td>
<td valign="top" align="center">0.4753</td>
</tr> <tr>
<td valign="top" align="left">MI-FGSM</td>
<td valign="top" align="center">0.1962</td>
<td valign="top" align="center">0.1444</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">0.7135</td>
<td valign="top" align="center">0.9474</td>
<td valign="top" align="center">0.6575</td>
<td valign="top" align="center">0.1589</td>
<td valign="top" align="center">0.1078</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">0.7180</td>
<td valign="top" align="center">0.9466</td>
<td valign="top" align="center">0.6597</td>
</tr> <tr>
<td valign="top" align="left">TI-FGSM</td>
<td valign="top" align="center">0.2179</td>
<td valign="top" align="center">0.1849</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">0.8153</td>
<td valign="top" align="center">0.9199</td>
<td valign="top" align="center">0.5576</td>
<td valign="top" align="center">0.1684</td>
<td valign="top" align="center">0.1451</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">0.8216</td>
<td valign="top" align="center">0.9330</td>
<td valign="top" align="center">0.5943</td>
</tr> <tr>
<td valign="top" align="left">Jitter</td>
<td valign="top" align="center">0.2461</td>
<td valign="top" align="center">0.1617</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">0.6342</td>
<td valign="top" align="center">0.9076</td>
<td valign="top" align="center">0.5864</td>
<td valign="top" align="center">0.2001</td>
<td valign="top" align="center">0.1305</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">0.6480</td>
<td valign="top" align="center">0.9107</td>
<td valign="top" align="center">0.5792</td>
</tr> <tr>
<td valign="top" align="left">stAdv</td>
<td valign="top" align="center">0.0581</td>
<td valign="top" align="center">0.0757</td>
<td valign="top" align="center">0.2420</td>
<td valign="top" align="center">0.8954</td>
<td valign="top" align="center">0.9873</td>
<td valign="top" align="center">0.7290</td>
<td valign="top" align="center">0.0490</td>
<td valign="top" align="center">0.0690</td>
<td valign="top" align="center">0.2420</td>
<td valign="top" align="center">0.8954</td>
<td valign="top" align="center">0.9873</td>
<td valign="top" align="center">0.7290</td>
</tr> <tr>
<td valign="top" align="left">Chroma-shift</td>
<td valign="top" align="center">0.0231</td>
<td valign="top" align="center">0.5943</td>
<td valign="top" align="center">0.0275</td>
<td valign="top" align="center">0.9142</td>
<td valign="top" align="center">0.9834</td>
<td valign="top" align="center">0.8079</td>
<td valign="top" align="center">0.0.0203</td>
<td valign="top" align="center">0.0246</td>
<td valign="top" align="center">0.0.2250</td>
<td valign="top" align="center">0.9126</td>
<td valign="top" align="center">0.0.9848</td>
<td valign="top" align="center">0.0.8027</td>
</tr> <tr>
<td valign="top" align="left">GUAP</td>
<td valign="top" align="center">0.4349</td>
<td valign="top" align="center">0.2838</td>
<td valign="top" align="center">0.4984</td>
<td valign="top" align="center">0.2768</td>
<td valign="top" align="center">0.7630</td>
<td valign="top" align="center">0.2955</td>
<td valign="top" align="center">0.4205</td>
<td valign="top" align="center">0.2501</td>
<td valign="top" align="center">0.6443</td>
<td valign="top" align="center">0.2289</td>
<td valign="top" align="center">0.7274</td>
<td valign="top" align="center">0.2674</td>
</tr> <tr>
<td valign="top" align="left">DualFlow</td>
<td valign="top" align="center"><bold>0.0188</bold></td>
<td valign="top" align="center"><bold>0.0324</bold></td>
<td valign="top" align="center"><bold>0.1642</bold></td>
<td valign="top" align="center"><bold>0.9451</bold></td>
<td valign="top" align="center"><bold>0.9876</bold></td>
<td valign="top" align="center"><bold>0.8192</bold></td>
<td valign="top" align="center"><bold>0.0169</bold></td>
<td valign="top" align="center"><bold>0.0312</bold></td>
<td valign="top" align="center"><bold>0.1550</bold></td>
<td valign="top" align="center"><bold>0.9451</bold></td>
<td valign="top" align="center"><bold>0.9876</bold></td>
<td valign="top" align="center"><bold>0.8192</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Note that for the FGSM-based baselines, we synthesize their adversarial examples under <italic>L</italic><sub><italic>inf</italic></sub>-norm=0.031 limitation; the others are not subject to the restrictions. Bold values indicates the best result.</p>
</table-wrap-foot>
</table-wrap>
<p>To visualize the difference between the adversarial examples generated by our method and the baselines, we also draw the adversarial perturbation generated on NIPS2107 by FGSM, MI-FGSM, TI-FGSM, Jitter stAdv, Chroma-shift, GUAP and the proposed method in <xref ref-type="fig" rid="F3">Figure 3</xref>, the target model is pre-trained VGG-19. The first two columns is the adversarial examples and the following are the adversarial noises of FGSM, MI-FGSM, TI-FGSM, Jitter stAdv, Chroma-shift, GUAP and our method, respectively. Noted that, for better observation, we magnified the noise by a factor of 10. From <xref ref-type="fig" rid="F3">Figure 3</xref>, we can clearly observe that stAdv and Chroma-Shift distort the whole image. In contrast, the adversarial examples generated by our method are focused on the salient region and its noise is milder, and they are similar to the original clean counterparts and are more imperceptible to human eyes. These simulations of the proposed method take place under diverse aspects and the outcome verified the betterment of the presented method over the compared baselines.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Adversarial examples and their corresponding perturbations. The first two columns are the adversarial examples, and the followings are the adversarial noise of FGSM, MI-FGSM, TI-FGSM, Jitter, stAdv, Chroma-shift, GUAP and our method, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-17-1129720-g0003.tif"/>
</fig></sec>
<sec>
<title>5.5. Detectability</title>
<p>Adversarial examples can be viewed as data outside the clean data distribution, so the defender can easily check whether each input is an adversarial example. Therefore, generating adversarial examples with high concealment means that they have the same or similar distribution as the original data (Ma X. et al., <xref ref-type="bibr" rid="B40">2018</xref>; Dolatabadi et al., <xref ref-type="bibr" rid="B14">2020</xref>). To verify whether the carefully crafted examples satisfy this rule, we follow (Dolatabadi et al., <xref ref-type="bibr" rid="B14">2020</xref>) and select LID (Ma X. et al., <xref ref-type="bibr" rid="B40">2018</xref>), Mahalanobis (Lee et al., <xref ref-type="bibr" rid="B33">2018</xref>), and Res-Flow (Zisselman and Tamar, <xref ref-type="bibr" rid="B63">2020</xref>) adversarial attack detectors to evaluate the performance of the adversarial examples crafted by DualFlow. For comparison, we choose FGSM (Goodfellow et al., <xref ref-type="bibr" rid="B20">2015</xref>), MI-FGSM (Dong et al., <xref ref-type="bibr" rid="B15">2018</xref>), stAdv (Xiao et al., <xref ref-type="bibr" rid="B54">2018</xref>), and Chroma-Shift (Aydin et al., <xref ref-type="bibr" rid="B2">2021</xref>) as baseline methods. The test results are shown in the <xref ref-type="table" rid="T7">Table 7</xref>, including the area under the receiver operating characteristic curve (AUROC) and detection accuracy. <xref ref-type="table" rid="T7">Table 7</xref>, we can find that these adversarial detectors struggle to detect malicious examples constructed with DualFlow, compared to the baseline in all cases. Empirical results precisely demonstrate the superiority of our method, which generates adversarial examples closer to the distribution of original clean images than other methods, and the optimized adversarial perturbations have better hiding ability. The classifier is ResNet-34, and the code used in this experiment is modified from deep_Mahalanobis_detector<xref ref-type="fn" rid="fn0008"><sup>8</sup></xref> and Residual-Flow<xref ref-type="fn" rid="fn0009"><sup>9</sup></xref>, respectively.</p>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>The detect results of DualFlow and the baselines on CIFAR-10 and CIFAR-100, Where the Chroma represent the Chroma-Shift.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919497">
<th valign="top" align="left" rowspan="2"><bold>Datasets</bold></th>
<th valign="top" align="left" rowspan="2"><bold>Methods</bold></th>
<th valign="top" align="center" colspan="5"><bold>AUROC (%)</bold> &#x02191;</th>
<th valign="top" align="center" colspan="5"><bold>Detection Acc. (%)</bold> &#x02191;</th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919497">
<td valign="top" align="center"><bold>FGSM</bold></td>
<td valign="top" align="center"><bold>MI-FGSM</bold></td>
<td valign="top" align="center"><bold>stAdv</bold></td>
<td valign="top" align="center"><bold>Chroma</bold></td>
<td valign="top" align="center"><bold>DualFlow</bold></td>
<td valign="top" align="center"><bold>FGSM</bold></td>
<td valign="top" align="center"><bold>MI-FGSM</bold></td>
<td valign="top" align="center"><bold>stAdv</bold></td>
<td valign="top" align="center"><bold>Chroma</bold></td>
<td valign="top" align="center"><bold>DualFlow</bold></td>
</tr> <tr>
<td valign="top" align="left">CIFAR-10</td>
<td valign="top" align="left">LID</td>
<td valign="top" align="center">99.67</td>
<td valign="top" align="center">95.36</td>
<td valign="top" align="center">82.13</td>
<td valign="top" align="center">70.61</td>
<td valign="top" align="center"><bold>52.23</bold></td>
<td valign="top" align="center">99.73</td>
<td valign="top" align="center">90.42</td>
<td valign="top" align="center">78.95</td>
<td valign="top" align="center">65.42</td>
<td valign="top" align="center"><bold>58.42</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Mahalanobis</td>
<td valign="top" align="center">96.54</td>
<td valign="top" align="center">98.54</td>
<td valign="top" align="center">85.64</td>
<td valign="top" align="center">75.61</td>
<td valign="top" align="center"><bold>58.49</bold></td>
<td valign="top" align="center">90.42</td>
<td valign="top" align="center">97.26</td>
<td valign="top" align="center">79.67</td>
<td valign="top" align="center">76.13</td>
<td valign="top" align="center"><bold>64.23</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Res-Flow</td>
<td valign="top" align="center">94.47</td>
<td valign="top" align="center">97.59</td>
<td valign="top" align="center">78.96</td>
<td valign="top" align="center">72.37</td>
<td valign="top" align="center"><bold>64.95</bold></td>
<td valign="top" align="center">88.56</td>
<td valign="top" align="center">91.54</td>
<td valign="top" align="center">76.38</td>
<td valign="top" align="center">73.64</td>
<td valign="top" align="center"><bold>59.78</bold></td>
</tr> <tr>
<td valign="top" align="left">CIFAR-100</td>
<td valign="top" align="left">LID</td>
<td valign="top" align="center">97.86</td>
<td valign="top" align="center">91.67</td>
<td valign="top" align="center">75.85</td>
<td valign="top" align="center">73.84</td>
<td valign="top" align="center"><bold>62.37</bold></td>
<td valign="top" align="center">93.34</td>
<td valign="top" align="center">82.6</td>
<td valign="top" align="center">76.71</td>
<td valign="top" align="center">69.57</td>
<td valign="top" align="center"><bold>57.78</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Mahalanobis</td>
<td valign="top" align="center">99.61</td>
<td valign="top" align="center">97.64</td>
<td valign="top" align="center">76.17</td>
<td valign="top" align="center">72.32</td>
<td valign="top" align="center"><bold>65.48</bold></td>
<td valign="top" align="center">98.62</td>
<td valign="top" align="center">92.49</td>
<td valign="top" align="center">80.65</td>
<td valign="top" align="center">71.48</td>
<td valign="top" align="center"><bold>63.15</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Res-Flow</td>
<td valign="top" align="center">99.07</td>
<td valign="top" align="center">99.76</td>
<td valign="top" align="center">78.53</td>
<td valign="top" align="center">78.56</td>
<td valign="top" align="center"><bold>65.74</bold></td>
<td valign="top" align="center">95.92</td>
<td valign="top" align="center">96.99</td>
<td valign="top" align="center">83.43</td>
<td valign="top" align="center">69.72</td>
<td valign="top" align="center"><bold>62.94</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>&#x02191; means that the larger the value, the better the detection method. Bold values indicates the best result.</p>
</table-wrap-foot>
</table-wrap></sec></sec>
<sec id="s6">
<title>6. Conclusions</title>
<p>In this paper, we propose a novel framework named Dual-Flow for generating imperceptible adversarial examples with strong attack ability. It aims to perturb images by disturbing their latent representation space rather than adding noise to the clean image at the pixel level. Combining the normalizing flow and the spatial transform techniques, DualFlow can attack images&#x00027; latent representations by changing the position of each value in the latent vector to craft adversarial examples. Besides, the empirical results of defense models show that DualFlow has stronger attack capability than noise-adding-based methods, which is meaningful for exploring the DNN&#x00027;s vulnerability sufficiently. Therefore, developing a more effective method to generate invisible, both for human eyes and the machine, is fascinating. Extensive experiments show that the adversarial examples obtained by DualFlow have superiority in imperceptibility and attack ability compared with the existing methods.</p></sec>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found at: CIFAR-10 and CIFAR-100, <ext-link ext-link-type="uri" xlink:href="http://www.cs.toronto.edu/&#x0007E;kriz/cifar.html">http://www.cs.toronto.edu/&#x0007E;kriz/cifar.html</ext-link>; ImageNet, <ext-link ext-link-type="uri" xlink:href="https://image-net.org/">https://image-net.org/</ext-link>.</p></sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>YW and JinZ performed computer simulations. DH analyzed the data. RL and JinhZ wrote the original draft. RL and XJ revised and edited the manuscript. WZ polished the manuscript. All authors confirmed the submitted version.</p></sec>
</body>
<back>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>This work was supported in part by the National Natural Science Foundation of China under Grant Nos. 62162067 and 62101480, in part by the Yunnan Province Science Foundation under Grant Nos. 202005AC160007, 202001BB050076, and Research and Application of Object detection based on Artificial Intelligence.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn0001"><p><sup>1</sup><ext-link ext-link-type="uri" xlink:href="http://www.cs.toronto.edu/&#x0007E;kriz/cifar.html">http://www.cs.toronto.edu/&#x0007E;kriz/cifar.html</ext-link></p></fn>
<fn id="fn0002"><p><sup>2</sup><ext-link ext-link-type="uri" xlink:href="https://image-net.org/">https://image-net.org/</ext-link></p></fn>
<fn id="fn0003"><p><sup>3</sup><ext-link ext-link-type="uri" xlink:href="https://github.com/chenyaofo/pytorch-cifar-models">https://github.com/chenyaofo/pytorch-cifar-models</ext-link></p></fn>
<fn id="fn0004"><p><sup>4</sup><ext-link ext-link-type="uri" xlink:href="https://github.com/RobustBench/robustbench">https://github.com/RobustBench/robustbench</ext-link></p></fn>
<fn id="fn0005"><p><sup>5</sup><ext-link ext-link-type="uri" xlink:href="https://github.com/Harry24k/adversarial-attacks-pytorch">https://github.com/Harry24k/adversarial-attacks-pytorch</ext-link></p></fn>
<fn id="fn0006"><p><sup>6</sup><ext-link ext-link-type="uri" xlink:href="https://www.cnpython.com/pypi/iqa-pytorch">https://www.cnpython.com/pypi/iqa-pytorch</ext-link></p></fn>
<fn id="fn0007"><p><sup>7</sup><ext-link ext-link-type="uri" xlink:href="https://github.com/andrewekhalel/sewar">https://github.com/andrewekhalel/sewar</ext-link></p></fn>
<fn id="fn0008"><p><sup>8</sup><ext-link ext-link-type="uri" xlink:href="https://github.com/pokaxpoka/deep_Mahalanobis_detector">https://github.com/pokaxpoka/deep_Mahalanobis_detector</ext-link></p></fn>
<fn id="fn0009"><p><sup>9</sup><ext-link ext-link-type="uri" xlink:href="https://github.com/EvZissel/Residual-Flow">https://github.com/EvZissel/Residual-Flow</ext-link></p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Arvinte</surname> <given-names>M.</given-names></name> <name><surname>Tewfik</surname> <given-names>A. H.</given-names></name> <name><surname>Vishwanath</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>Detecting patch adversarial attacks with image residuals</article-title>. <source>CoRR</source>, abs/2002.12504. <pub-id pub-id-type="doi">10.48550/arXiv.2002.12504</pub-id></citation>
</ref>
<ref id="B2">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Aydin</surname> <given-names>A.</given-names></name> <name><surname>Sen</surname> <given-names>D.</given-names></name> <name><surname>Karli</surname> <given-names>B. T.</given-names></name> <name><surname>Hanoglu</surname> <given-names>O.</given-names></name> <name><surname>Temizel</surname> <given-names>A.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Imperceptible adversarial examples by spatial chroma-shift,&#x0201D;</article-title> in <source>ADVM &#x00027;21: Proceedings of the 1st International Workshop on Adversarial Learning for Multimedia</source>, eds D. Song, D. Tao, A. L. Yuille, A. Anandkumar, A. Liu, X. Chen, Y. Li, C. Xiao, X. Yang, and X. Liu (<publisher-loc>Beijing</publisher-loc>: <publisher-name>ACM</publisher-name>) <fpage>8</fpage>&#x02013;<lpage>14</lpage>.<pub-id pub-id-type="pmid">33286969</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bai</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Zeng</surname> <given-names>Y.</given-names></name> <name><surname>Jiang</surname> <given-names>Y.</given-names></name> <name><surname>Xia</surname> <given-names>S.</given-names></name></person-group> (<year>2023</year>). <article-title>Query efficient black-box adversarial attack on deep neural networks</article-title>. <source>Pattern Recognit</source>. <volume>133</volume>, <fpage>109037</fpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2022.109037</pub-id><pub-id pub-id-type="pmid">35468057</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ballet</surname> <given-names>V.</given-names></name> <name><surname>Renard</surname> <given-names>X.</given-names></name> <name><surname>Aigrain</surname> <given-names>J.</given-names></name> <name><surname>Laugel</surname> <given-names>T.</given-names></name> <name><surname>Frossard</surname> <given-names>P.</given-names></name> <name><surname>Detyniecki</surname> <given-names>M.</given-names></name></person-group> (<year>2019</year>). <article-title>Imperceptible adversarial attacks on tabular data</article-title>. <source>CoRR</source>, abs/1911.03274. <pub-id pub-id-type="doi">10.48550/arXiv.1911.03274</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Besnier</surname> <given-names>V.</given-names></name> <name><surname>Bursuc</surname> <given-names>A.</given-names></name> <name><surname>Picard</surname> <given-names>D.</given-names></name> <name><surname>Briot</surname> <given-names>A.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Triggering failures: out-of-distribution detection by learning from local adversarial attacks in semantic segmentation,&#x0201D;</article-title> in <source>2021 IEEE/CVF International Conference on Computer Vision (ICCV)</source> (<publisher-loc>Montreal, QC</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>15681</fpage>&#x02013;<lpage>15690</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Carlini</surname> <given-names>N.</given-names></name> <name><surname>Wagner</surname> <given-names>D. A.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Towards evaluating the robustness of neural networks,&#x0201D;</article-title> in <source>2017 IEEE Symposium on Security and Privacy</source> (<publisher-loc>San Jose, CA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>39</fpage>&#x02013;<lpage>57</lpage>.</citation>
</ref>
<ref id="B7">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Cheng</surname> <given-names>Y.</given-names></name> <name><surname>Gan</surname> <given-names>Z.</given-names></name> <name><surname>Gu</surname> <given-names>Q.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Efficient robust training via backward smoothing,&#x0201D;</article-title> in <source>Thirty-Sixth AAAI Conference on Artificial Intelligence, (AAAI) 2022, Thirty-Fourth Conference on Innovative Applications of Artificial Intelligence, IAAI 2022, The Twelveth Symposium on Educational Advances in Artificial Intelligence</source> (<publisher-loc>AAAI Press</publisher-loc>), <fpage>6222</fpage>&#x02013;<lpage>6230</lpage>.</citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Croce</surname> <given-names>F.</given-names></name> <name><surname>Andriushchenko</surname> <given-names>M.</given-names></name> <name><surname>Sehwag</surname> <given-names>V.</given-names></name> <name><surname>Debenedetti</surname> <given-names>E.</given-names></name> <name><surname>Flammarion</surname> <given-names>N.</given-names></name> <name><surname>Chiang</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;Robustbench: a standardized adversarial robustness benchmark,&#x0201D;</article-title> in <source>Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks 1</source>, eds J. Vanschoren and S.-K. Yeung.</citation>
</ref>
<ref id="B9">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Croce</surname> <given-names>F.</given-names></name> <name><surname>Andriushchenko</surname> <given-names>M.</given-names></name> <name><surname>Singh</surname> <given-names>N. D.</given-names></name> <name><surname>Flammarion</surname> <given-names>N.</given-names></name> <name><surname>Hein</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Sparse-rs: a versatile framework for query-efficient sparse black-box adversarial attacks,&#x0201D;</article-title> in <source>Thirty-Sixth AAAI Conference on Artificial Intelligence, AAAI 2022, Thirty-Fourth Conference on Innovative Applications of Artificial Intelligence, IAAI 2022, The Twelveth Symposium on Educational Advances in Artificial Intelligence</source> (<publisher-loc>AAAI Press</publisher-loc>), <fpage>6437</fpage>&#x02013;<lpage>6445</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Deng</surname> <given-names>J.</given-names></name> <name><surname>Dong</surname> <given-names>W.</given-names></name> <name><surname>Socher</surname> <given-names>R.</given-names></name> <name><surname>Li</surname> <given-names>L.</given-names></name> <name><surname>Li</surname> <given-names>K.</given-names></name> <name><surname>Fei-Fei</surname> <given-names>L.</given-names></name></person-group> (<year>2009</year>). <article-title>&#x0201C;Imagenet: a large-scale hierarchical image database,&#x0201D;</article-title> in <source>2009 IEEE Computer Society Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Miami, FL</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>248</fpage>&#x02013;<lpage>255</lpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>K.</given-names></name> <name><surname>Ma</surname> <given-names>K.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Simoncelli</surname> <given-names>E. P.</given-names></name></person-group> (<year>2021</year>). <article-title>Comparison of full-reference image quality models for optimization of image processing systems</article-title>. <source>Int. J. Comput. Vis</source>. <volume>129</volume>, <fpage>1258</fpage>&#x02013;<lpage>1281</lpage>. <pub-id pub-id-type="doi">10.1007/s11263-020-01419-7</pub-id><pub-id pub-id-type="pmid">33495671</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>K.</given-names></name> <name><surname>Ma</surname> <given-names>K.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Simoncelli</surname> <given-names>E. P.</given-names></name></person-group> (<year>2022</year>). <article-title>Image quality assessment: unifying structure and texture similarity</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>44</volume>, <fpage>2567</fpage>&#x02013;<lpage>2581</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2020.3045810</pub-id><pub-id pub-id-type="pmid">33338012</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Dinh</surname> <given-names>L.</given-names></name> <name><surname>Krueger</surname> <given-names>D.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;NICE: non-linear independent components estimation,&#x0201D;</article-title> in <source>3rd International Conference on Learning Representations</source> (<publisher-loc>San Diego, CA</publisher-loc>: <publisher-name>ICLR</publisher-name>).</citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dolatabadi</surname> <given-names>H. M.</given-names></name> <name><surname>Erfani</surname> <given-names>S. M.</given-names></name> <name><surname>Leckie</surname> <given-names>C.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;AdvFlow: Inconspicuous black-box adversarial attacks using normalizing flows,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020</source>, eds H. Larochelle, M. A. Rantzato, R. Hadsell, M.-F. Balcan, and H.-T. Lin.</citation>
</ref>
<ref id="B15">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Dong</surname> <given-names>Y.</given-names></name> <name><surname>Liao</surname> <given-names>F.</given-names></name> <name><surname>Pang</surname> <given-names>T.</given-names></name> <name><surname>Su</surname> <given-names>H.</given-names></name> <name><surname>Zhu</surname> <given-names>J.</given-names></name> <name><surname>Hu</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>&#x0201C;Boosting adversarial attacks with momentum,&#x0201D;</article-title> in <source>2018 IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Salt Lake City, UT</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>9185</fpage>&#x02013;<lpage>9193</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Dong</surname> <given-names>Y.</given-names></name> <name><surname>Pang</surname> <given-names>T.</given-names></name> <name><surname>Su</surname> <given-names>H.</given-names></name> <name><surname>Zhu</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Evading defenses to transferable adversarial examples by translation-invariant attacks,&#x0201D;</article-title> in <source>IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Long Beach, CA</publisher-loc>: <publisher-name>Computer Vision Foundation; IEEE</publisher-name>), <fpage>4312</fpage>&#x02013;<lpage>4321</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Duan</surname> <given-names>R.</given-names></name> <name><surname>Ma</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Bailey</surname> <given-names>J.</given-names></name> <name><surname>Qin</surname> <given-names>A. K.</given-names></name> <name><surname>Yang</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Adversarial camouflage: Hiding physical-world attacks with natural styles,&#x0201D;</article-title> in <source>2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Seattle, WA</publisher-loc>: <publisher-name>Computer Vision Foundation; IEEE</publisher-name>), <fpage>97</fpage>&#x02013;<lpage>1005</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Eykholt</surname> <given-names>K.</given-names></name> <name><surname>Evtimov</surname> <given-names>I.</given-names></name> <name><surname>Fernandes</surname> <given-names>E.</given-names></name> <name><surname>Li</surname> <given-names>B.</given-names></name> <name><surname>Rahmati</surname> <given-names>A.</given-names></name> <name><surname>Xiao</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>&#x0201C;Robust physical-world attacks on deep learning visual classification,&#x0201D;</article-title> in <source>2018 IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Salt Lake City, UT</publisher-loc>: <publisher-name>Computer Vision Foundation; IEEE</publisher-name>), <fpage>1625</fpage>&#x02013;<lpage>1634</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Fawzi</surname> <given-names>A.</given-names></name> <name><surname>Frossard</surname> <given-names>P.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Manitest: are classifiers really invariant?,&#x0201D;</article-title> in <source>Proceedings of the British Machine Vision Conference 2015</source> (<publisher-loc>Swansea</publisher-loc>), <fpage>106.1</fpage>&#x02013;<lpage>106.13</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Goodfellow</surname> <given-names>I. J.</given-names></name> <name><surname>Shlens</surname> <given-names>J.</given-names></name> <name><surname>Szegedy</surname> <given-names>C.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Explaining and harnessing adversarial examples,&#x0201D;</article-title> in <source>3rd International Conference on Learning Representations</source>, eds Y. Bengio and Y. Lecum (<publisher-loc>San Diego, CA</publisher-loc>).</citation>
</ref>
<ref id="B21">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>C.</given-names></name> <name><surname>Gardener</surname> <given-names>J. R.</given-names></name> <name><surname>You</surname> <given-names>Y.</given-names></name> <name><surname>Wilson</surname> <given-names>A. G.</given-names></name> <name><surname>Weinberger</surname> <given-names>K. Q.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Simple black-box adversarial attacks,&#x0201D;</article-title> in <source>Proceedings of the 36th International Conference on Machine Learning</source>, eds K. Chaudhari and R. Salakhutdinov (<publisher-loc>Long Beach, CA</publisher-loc>: <publisher-name>ICML</publisher-name>).</citation>
</ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Deep residual learning for image recognition,&#x0201D;</article-title> in <source>2016 IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>770</fpage>&#x02013;<lpage>778</lpage>.<pub-id pub-id-type="pmid">32166560</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hendrycks</surname> <given-names>D.</given-names></name> <name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Mazeika</surname> <given-names>M.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Using pre-training can improve model robustness and uncertainty,&#x0201D;</article-title> in <source>in Proceedings of the 36th International Conference on Machine Learning</source>, eds K. Chaudhuri and R. Salakhutdinov (<publisher-loc>Long Beach, CA</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>2712</fpage>&#x02013;<lpage>2721</lpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>G.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>van der Maaten</surname> <given-names>L.</given-names></name> <name><surname>Weinberger</surname> <given-names>K. Q.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Densely connected convolutional networks,&#x0201D;</article-title> in <source>2017 IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Honolulu, HI</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>2261</fpage>&#x02013;<lpage>2269</lpage>.</citation>
</ref>
<ref id="B25">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ilyas</surname> <given-names>A.</given-names></name> <name><surname>Engstrom</surname> <given-names>L.</given-names></name> <name><surname>Athalye</surname> <given-names>A.</given-names></name> <name><surname>Lin</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Black-box adversarial attacks with limited queries and information,&#x0201D;</article-title> in <source>Proceedings of the 35th International Conference on Machine Learning</source>, eds J. G. Dy and A. Krause (<publisher-loc>Stockholm</publisher-loc>: <publisher-name>PMLR</publisher-name>) <fpage>2142</fpage>&#x02013;<lpage>2151</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ilyas</surname> <given-names>A.</given-names></name> <name><surname>Engstrom</surname> <given-names>L.</given-names></name> <name><surname>Madry</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Prior convictions: black-box adversarial attacks with bandits and priors,&#x0201D;</article-title> in <source>7th International Conference on Learning Representations</source> (<publisher-loc>New Orleans, LA</publisher-loc>: <publisher-name>OpenReview.net</publisher-name>).</citation>
</ref>
<ref id="B27">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Jaderberg</surname> <given-names>M.</given-names></name> <name><surname>Simonyan</surname> <given-names>K.</given-names></name> <name><surname>Zisserman</surname> <given-names>A.</given-names></name> <name><surname>Kavukcuoglu</surname> <given-names>K.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Spatial transformer networks,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015</source>, eds C. Cortes, N. D. Lawrence, D. D. Lee, M. Sugiyama, and R. Garnett (<publisher-loc>Montreal, QC</publisher-loc>), <fpage>2017</fpage>&#x02013;<lpage>2025</lpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jordan</surname> <given-names>M.</given-names></name> <name><surname>Manoj</surname> <given-names>N.</given-names></name> <name><surname>Goel</surname> <given-names>S.</given-names></name> <name><surname>Dimakis</surname> <given-names>A. G.</given-names></name></person-group> (<year>2019</year>). <article-title>Quantifying perceptual distortion of adversarial examples</article-title>. <source>CoRR</source>, abs/1902.08265. <pub-id pub-id-type="doi">10.48550/arXiv.1902.08265</pub-id></citation>
</ref>
<ref id="B29">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kingma</surname> <given-names>D. P.</given-names></name> <name><surname>Dhariwal</surname> <given-names>P.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Glow: generative flow with invertible 1x1 convolutions,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 31: Annual Conference on Neural Information Processing Systems 2018</source> (<publisher-loc>Montreal, QC</publisher-loc>), <fpage>10236</fpage>&#x02013;<lpage>10245</lpage>.</citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krizhevsky</surname> <given-names>A.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2009</year>). <source>Learning multiple layers of features from tiny images</source>. Computer Science Department, University of Toronto, Techchnical Report 1.<pub-id pub-id-type="pmid">33561989</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kurakin</surname> <given-names>A.</given-names></name> <name><surname>Goodfellow</surname> <given-names>I. J.</given-names></name> <name><surname>Bengio</surname> <given-names>S.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Adversarial examples in the physical world,&#x0201D;</article-title> in <source>5th International Conference on Learning Representations</source> (<publisher-loc>Toulon</publisher-loc>).<pub-id pub-id-type="pmid">34851825</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Laidlaw</surname> <given-names>C.</given-names></name> <name><surname>Feizi</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Functional adversarial attacks,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 32: Annual Conference on Neural Information Processing Systems 2019</source>, eds H. M. Wallach, H. Larochelle, A. Beydelzimer, F. d&#x00027;Alche-Bec, E. B. Fox, and R. Garnett (<publisher-loc>Vancouver, BC</publisher-loc>), <fpage>10408</fpage>&#x02013;<lpage>10418</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Lee</surname> <given-names>H.</given-names></name> <name><surname>Shin</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;A simple unified framework for detecting out-of-distribution samples and adversarial attacks,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 31: Annual Conference on Neural Information Processing Systems 2018</source>, eds S. Bengio, H. M. Wallach, H. Larochelle, K. Grauman, N. Cesa-Bianchi, and R. Garnett (<publisher-loc>Montreal, QC</publisher-loc>), <fpage>7167</fpage>&#x02013;<lpage>7177</lpage>.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>J.</given-names></name></person-group> (<year>2000</year>). <article-title>Spatial quality evaluation of fusion of different resolution images</article-title>. <source>Int. Arch. Photogramm. Remot. Sens.</source> <volume>33</volume>, <fpage>339</fpage>&#x02013;<lpage>346</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>A.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Fan</surname> <given-names>J.</given-names></name> <name><surname>Ma</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>A.</given-names></name> <name><surname>Xie</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;Perceptual-sensitive gan for generating adversarial patches,&#x0201D;</article-title> in <source>The Thirty-Third AAAI Conference on Artificial Intelligence 2019, The Thirty-First Innovative Applications of Artificial Intelligence</source> (<publisher-loc>Honolulu, HI</publisher-loc>: <publisher-name>AAAI Press</publisher-name>), <fpage>1028</fpage>&#x02013;<lpage>1035</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Yang</surname> <given-names>H.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Song</surname> <given-names>L.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;DPATCH: an adversarial patch attack on object detectors,&#x0201D;</article-title> in <source>Workshop on Artificial Intelligence Safety 2019 co-located with the Thirty-Third AAAI Conference on Artificial Intelligence 2019</source> (<publisher-loc>Honolulu, HI</publisher-loc>).</citation>
</ref>
<ref id="B37">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>B.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Wei</surname> <given-names>L.</given-names></name> <name><surname>Xu</surname> <given-names>Q.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Towards imperceptible and robust adversarial example attacks against neural networks,&#x0201D;</article-title> in <source>Proceedings of the Thirty-Second Conference on Artificial Intelligence, (AAAI-18), the 30th Innovative Applications of Artificial Intelligence (IAAI-18), and the 8th AAAI Symposium on Educational Advances in Artificial Intelligence (EAAI-18)</source> (<publisher-loc>New Orleans, LA</publisher-loc>: <publisher-name>AAAI Press</publisher-name>), <fpage>1652</fpage>&#x02013;<lpage>1659</lpage>.</citation>
</ref>
<ref id="B38">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>C.</given-names></name> <name><surname>Lin</surname> <given-names>Q.</given-names></name> <name><surname>Xie</surname> <given-names>W.</given-names></name> <name><surname>Wu</surname> <given-names>B.</given-names></name> <name><surname>Xie</surname> <given-names>J.</given-names></name> <name><surname>Shen</surname> <given-names>L.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Frequency-driven imperceptible adversarial attack on semantic similarity,&#x0201D;</article-title> in <source>2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-loc>New Orleans, LA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>15294</fpage>&#x02013;<lpage>15303</lpage>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>N.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Zheng</surname> <given-names>H.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Shufflenet V2: practical guidelines for efficient CNN architecture design,&#x0201D;</article-title> in <source>ECCV, Vol. 11218</source>, <fpage>122</fpage>&#x02013;<lpage>138</lpage>.</citation>
</ref>
<ref id="B40">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>B.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Erfani</surname> <given-names>S. M.</given-names></name> <name><surname>Wijewickrema</surname> <given-names>S. N. R.</given-names></name> <name><surname>Schoenebeck</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>&#x0201C;Characterizing adversarial subspaces using local intrinsic dimensionality,&#x0201D;</article-title> in <source>6th International Conference on Learning Representations</source> (<publisher-loc>Vancouver, BC</publisher-loc>: <publisher-name>OpenReview.net</publisher-name>).</citation>
</ref>
<ref id="B41">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Madry</surname> <given-names>A.</given-names></name> <name><surname>Makelov</surname> <given-names>A.</given-names></name> <name><surname>Schmidt</surname> <given-names>L.</given-names></name> <name><surname>Tsipras</surname> <given-names>D.</given-names></name> <name><surname>Vladu</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Towards deep learning models resistant to adversarial attacks,&#x0201D;</article-title> in <source>6th International Conference on Learning Representations</source> (<publisher-loc>Vancouver, BC</publisher-loc>: <publisher-name>OpenReview.net</publisher-name>).</citation>
</ref>
<ref id="B42">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Narodytska</surname> <given-names>N.</given-names></name> <name><surname>Kasiviswanathan</surname> <given-names>S. P.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Simple black-box adversarial attacks on deep neural networks,&#x0201D;</article-title> in <source>2017 IEEE Conference on Computer Vision and Pattern Recognition Workshops</source> (<publisher-loc>IEEE Computer Society</publisher-loc>), <fpage>1310</fpage>&#x02013;<lpage>1318</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Rice</surname> <given-names>L.</given-names></name> <name><surname>Wong</surname> <given-names>E.</given-names></name> <name><surname>Kolter</surname> <given-names>J. Z.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Overfitting in adversarially robust deep learning,&#x0201D;</article-title> in <source>2017 IEEE Conference on Computer Vision and Pattern Recognition Workshops</source> (<publisher-loc>PMLR</publisher-loc>), <fpage>8093</fpage>&#x02013;<lpage>8104</lpage>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Salman</surname> <given-names>H.</given-names></name> <name><surname>Ilyas</surname> <given-names>A.</given-names></name> <name><surname>Engstrom</surname> <given-names>L.</given-names></name> <name><surname>Kapoor</surname> <given-names>A.</given-names></name> <name><surname>Madry</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Do adversarially robust imagenet models transfer better?,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020</source>, eds H. Larochelle, M. Ranzato, R. Hadsell, M.-F. Balcan, and H.-T. Lin.</citation>
</ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sandler</surname> <given-names>M.</given-names></name> <name><surname>Howard</surname> <given-names>A. G.</given-names></name> <name><surname>Zhu</surname> <given-names>M.</given-names></name> <name><surname>Zhmoginov</surname> <given-names>A.</given-names></name> <name><surname>Chen</surname> <given-names>L.</given-names></name></person-group> (<year>2018</year>). <article-title>Inverted residuals and linear bottlenecks: Mobile networks for classification, detection and segmentation</article-title>. <source>CoRR</source>, abs/1801.04381. <pub-id pub-id-type="doi">10.1109/CVPR.2018.00474</pub-id></citation>
</ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schwinn</surname> <given-names>L.</given-names></name> <name><surname>Raab</surname> <given-names>R.</given-names></name> <name><surname>Nguyen</surname> <given-names>A.</given-names></name> <name><surname>Zanca</surname> <given-names>D.</given-names></name> <name><surname>Eskofier</surname> <given-names>B. M.</given-names></name></person-group> (<year>2021</year>). <article-title>Exploring misclassifications of robust neural networks to enhance adversarial attacks</article-title>. <source>CoRR</source>, abs/2105.10304. <pub-id pub-id-type="doi">10.48550/arXiv.2105.10304</pub-id></citation>
</ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shao</surname> <given-names>Z.</given-names></name> <name><surname>Wu</surname> <given-names>Z.</given-names></name> <name><surname>Huang</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>Advexpander: Generating natural language adversarial examples by expanding text</article-title>. <source>IEEE ACM Trans. Audio Speech Lang. Process</source>. <volume>30</volume>, <fpage>1184</fpage>&#x02013;<lpage>1196</lpage>. <pub-id pub-id-type="doi">10.1109/TASLP.2021.3129339</pub-id></citation>
</ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sheikh</surname> <given-names>H. R.</given-names></name> <name><surname>Bovik</surname> <given-names>A. C.</given-names></name></person-group> (<year>2004</year>). <article-title>&#x0201C;Image information and visual quality,&#x0201D;</article-title> in <source>ICASSP</source>, <fpage>709</fpage>&#x02013;<lpage>712</lpage>.<pub-id pub-id-type="pmid">16479813</pub-id></citation></ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Simonyan</surname> <given-names>K.</given-names></name> <name><surname>Zisserman</surname> <given-names>A.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Very deep convolutional networks for large-scale image recognition,&#x0201D;</article-title> in <source>ICLR</source>.</citation>
</ref>
<ref id="B50">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Thys</surname> <given-names>S.</given-names></name> <name><surname>Ranst</surname> <given-names>W. V.</given-names></name> <name><surname>Goedem&#x000E9;</surname> <given-names>T.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Fooling automated surveillance cameras: adversarial patches to attack person detection,&#x0201D;</article-title> in <source>CVPR</source> (<publisher-loc>IEEE; Computer Vision Foundation</publisher-loc>)<fpage>49</fpage>&#x02013;<lpage>55</lpage>.</citation>
</ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Bovik</surname> <given-names>A. C.</given-names></name> <name><surname>Sheikh</surname> <given-names>H. R.</given-names></name> <name><surname>Simoncelli</surname> <given-names>E. P.</given-names></name></person-group> (<year>2004</year>). <article-title>Image quality assessment: from error visibility to structural similarity</article-title>. <source>IEEE Trans. on Image Process</source>. <volume>13</volume>, <fpage>600</fpage>&#x02013;<lpage>612</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2003.819861</pub-id><pub-id pub-id-type="pmid">15376593</pub-id></citation></ref>
<ref id="B52">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wong</surname> <given-names>E.</given-names></name> <name><surname>Rice</surname> <given-names>L.</given-names></name> <name><surname>Kolter</surname> <given-names>J. Z.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Fast is better than free: revisiting adversarial training,&#x0201D;</article-title> in <source>ICLR</source>.</citation>
</ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>D.</given-names></name> <name><surname>Xia</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Adversarial weight perturbation helps robust generalization,&#x0201D;</article-title> in <source>NeurIPS</source>.</citation>
</ref>
<ref id="B54">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiao</surname> <given-names>C.</given-names></name> <name><surname>Zhu</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>B.</given-names></name> <name><surname>He</surname> <given-names>W.</given-names></name> <name><surname>Liu</surname> <given-names>M.</given-names></name> <name><surname>Song</surname> <given-names>D.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Spatially transformed adversarial examples,&#x0201D;</article-title> in <source>ICLR</source>.</citation>
</ref>
<ref id="B55">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>H.</given-names></name> <name><surname>Ma</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Deb</surname> <given-names>D.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Tang</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Adversarial attacks and defenses in images, graphs and text: a review</article-title>. <source>Inte. J. Autom. Comput</source>. <volume>17</volume>, <fpage>151</fpage>&#x02013;<lpage>178</lpage>. <pub-id pub-id-type="doi">10.1007/s11633-019-1211-x</pub-id></citation>
</ref>
<ref id="B56">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>Z.</given-names></name> <name><surname>Yu</surname> <given-names>F.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Lance: a comprehensive and lightweight CNN defense methodology against physical adversarial attacks on embedded multimedia applications,&#x0201D;</article-title> in <source>ASP-DAC</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>470</fpage>&#x02013;<lpage>475</lpage>.</citation>
</ref>
<ref id="B57">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yan</surname> <given-names>C.</given-names></name> <name><surname>Xu</surname> <given-names>Z.</given-names></name> <name><surname>Yin</surname> <given-names>Z.</given-names></name> <name><surname>Ji</surname> <given-names>X.</given-names></name> <name><surname>Xu</surname> <given-names>W.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Rolling colors: adversarial laser exploits against traffic light recognition,&#x0201D;</article-title> in <source>USENIX Security</source>, <fpage>1957</fpage>&#x02013;<lpage>1974</lpage>.</citation>
</ref>
<ref id="B58">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yi</surname> <given-names>Z.</given-names></name> <name><surname>Yu</surname> <given-names>J.</given-names></name> <name><surname>Tan</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>Q.</given-names></name></person-group> (<year>2022</year>). <article-title>Fine-tuning more stable neural text classifiers for defending word level adversarial attacks</article-title>. <source>Appl. Intell</source>. <volume>52</volume>, <fpage>11948</fpage>&#x02013;<lpage>11965</lpage>. <pub-id pub-id-type="doi">10.1007/s10489-021-02800-w</pub-id></citation>
</ref>
<ref id="B59">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>R.</given-names></name> <name><surname>Isola</surname> <given-names>P.</given-names></name> <name><surname>Efros</surname> <given-names>A. A.</given-names></name> <name><surname>Shechtman</surname> <given-names>E.</given-names></name> <name><surname>Wang</surname> <given-names>O.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;The unreasonable effectiveness of deep features as a perceptual metric,&#x0201D;</article-title> in <source>CVPR</source>, <fpage>586</fpage>&#x02013;<lpage>595</lpage>.</citation>
</ref>
<ref id="B60">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Ruan</surname> <given-names>W.</given-names></name> <name><surname>Wang</surname> <given-names>F.</given-names></name> <name><surname>Huang</surname> <given-names>X.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Generalizing universal adversarial attacks beyond additive perturbations,&#x0201D;</article-title> in <source>ICDM</source>, <fpage>1412</fpage>&#x02013;<lpage>1417</lpage>.</citation>
</ref>
<ref id="B61">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>Y.</given-names></name> <name><surname>Zhu</surname> <given-names>H.</given-names></name> <name><surname>Liang</surname> <given-names>R.</given-names></name> <name><surname>Shen</surname> <given-names>Q.</given-names></name> <name><surname>Zhang</surname> <given-names>S.</given-names></name> <name><surname>Chen</surname> <given-names>K.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Seeing isn&#x00027;t believing: Towards more robust adversarial attack against real world object detectors,&#x0201D;</article-title> in <source>CCS</source>, eds L. Cavallaro, J. Kinder, X. Wang, and J. Katz, <fpage>1989</fpage>&#x02013;<lpage>2004</lpage>.</citation>
</ref>
<ref id="B62">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Y.</given-names></name> <name><surname>Han</surname> <given-names>M.</given-names></name> <name><surname>Liu</surname> <given-names>L.</given-names></name> <name><surname>He</surname> <given-names>J.</given-names></name> <name><surname>Gao</surname> <given-names>X.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;The adversarial attacks threats on computer vision: a survey,&#x0201D;</article-title> in <source>MASS</source>, <fpage>25</fpage>&#x02013;<lpage>30</lpage>.</citation>
</ref>
<ref id="B63">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zisselman</surname> <given-names>E.</given-names></name> <name><surname>Tamar</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Deep residual flow for out of distribution detection,&#x0201D;</article-title> in <source>CVPR</source>, <fpage>13991</fpage>&#x02013;<lpage>14000</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>