<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="methods-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Comput. Sci.</journal-id>
<journal-title>Frontiers in Computer Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Comput. Sci.</abbrev-journal-title>
<issn pub-type="epub">2624-9898</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fcomp.2025.1631561</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Computer Science</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>MeetSafe: enhancing robustness against white-box adversarial examples</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Stenhuis</surname> <given-names>Ruben</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Liu</surname> <given-names>Dazhuang</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3067120/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Qiao</surname> <given-names>Yanqi</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3130281/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Conti</surname> <given-names>Mauro</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/548275/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Panaousis</surname> <given-names>Manos</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Liang</surname> <given-names>Kaitai</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2585183/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Cybersecurity Group, Faculty of Electrical Engineering, Mathematics and Computer Science, Delft University of Technology</institution>, <addr-line>Delft</addr-line>, <country>Netherlands</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Mathematics, SPRITZ Security and Privacy Research Group, University of Padua</institution>, <addr-line>Padua</addr-line>, <country>Italy</country></aff>
<aff id="aff3"><sup>3</sup><institution>Faculty of Engineering and Science, School of Computing and Mathematical Sciences, Center for Sustainable Cyber Security, University of Greenwich</institution>, <addr-line>London</addr-line>, <country>United Kingdom</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Christos Xenakis, University of Piraeus, Greece</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Christoforos Ntantogian, Ionian University, Greece</p>
<p>Vaios Bolgouras, Unisystems, Luxembourg</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Dazhuang Liu <email>d.liu-8&#x00040;tudelft.nl</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>13</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>7</volume>
<elocation-id>1631561</elocation-id>
<history>
<date date-type="received">
<day>19</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>16</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Stenhuis, Liu, Qiao, Conti, Panaousis and Liang.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Stenhuis, Liu, Qiao, Conti, Panaousis and Liang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Convolutional neural networks (CNNs) are vulnerable to adversarial attacks in computer vision tasks. Current adversarial detections are ineffective against white-box attacks and inefficient when deep CNNs generate high-dimensional hidden features. This study proposes MeetSafe, an effective and scalable adversarial example (AE) detection against white-box attacks. MeetSafe identifies AEs using critical hidden features rather than the entire feature space. We observe a non-uniform distribution of Z-scores between clean samples and adversarial examples (AEs) among hidden features and propose two utility functions to select those most relevant to AEs. We process critical hidden features using feature engineering methods: local outlier factor (LOF), feature squeezing, and whitening, which estimate feature density relative to its k-neighbors, reduce redundancy, and normalize features. To deal with the curse of dimensionality and smooth statistical fluctuations in high-dimensional features, we propose local reachability density (LRD). Our LRD iteratively selects a bag of engineered features with random cardinality and quantifies their average density by its k-nearest neighbors. Finally, MeetSafe constructs a Gaussian Mixture Model (GMM) with the processed features and detects AEs if it is seen as a local outlier, shown by a low density from GMM. Experimental results show that MeetSafe achieves 74%, 96%, and 79% of detection accuracy against adaptive, classic, and white-box attacks, respectively, and at least 2.3&#x000D7; faster than comparison methods.</p></abstract>
<kwd-group>
<kwd>adversarial attack</kwd>
<kwd>convolutional neural network</kwd>
<kwd>Gaussian Mixture Model</kwd>
<kwd>adversarial example</kwd>
<kwd>local reachability density</kwd>
</kwd-group>
<counts>
<fig-count count="5"/>
<table-count count="5"/>
<equation-count count="8"/>
<ref-count count="46"/>
<page-count count="14"/>
<word-count count="10025"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computer Security</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Deep neural networks (DNNs) have emerged as highly effective models in machine learning (ML) tasks. Among DNNs, convolutional neural networks (CNNs) revolutionized various computer vision applications, such as medical image recognition (<xref ref-type="bibr" rid="B26">Litjens et al., 2017</xref>) and facial recognition (<xref ref-type="bibr" rid="B44">Zhao et al., 2003</xref>). However, the robustness of CNNs remains a significant concern, as even a slight and imperceptible perturbations deliberately designed to manipulate images can result in high misclassification rates (<xref ref-type="bibr" rid="B36">Szegedy et al., 2014</xref>). Therefore, adversarial detections are in urgent demand to guarantee the integrity of CNN models.</p>
<p>A plethora of adversarial detections (<xref ref-type="bibr" rid="B11">Feinman et al., 2017</xref>; <xref ref-type="bibr" rid="B16">Hu et al., 2019</xref>; <xref ref-type="bibr" rid="B15">Hendrycks and Gimpel, 2017</xref>; <xref ref-type="bibr" rid="B33">Raghuram et al., 2021</xref>; <xref ref-type="bibr" rid="B28">Ma et al., 2018</xref>; <xref ref-type="bibr" rid="B2">Aldahdooh et al., 2022</xref>) have been proposed to identify adversarial examples (AEs). However, these methods remain vulnerable to white-box adversarial attacks (<xref ref-type="bibr" rid="B7">Carlini and Wagner, 2017a</xref>; <xref ref-type="bibr" rid="B4">Athalye et al., 2018</xref>; <xref ref-type="bibr" rid="B3">Athalye and Carlini, 2018</xref>; <xref ref-type="bibr" rid="B40">Tramer et al., 2020</xref>), which assume full access to the model and training process. Several defenses (<xref ref-type="bibr" rid="B33">Raghuram et al., 2021</xref>; <xref ref-type="bibr" rid="B16">Hu et al., 2019</xref>) have been developed against white-box AEs. They obscure the detector&#x00027;s gradients, leading to: (1) diminished security, as gradient obfuscation is proven to be an ineffective strategy for enhancing robustness (<xref ref-type="bibr" rid="B4">Athalye et al., 2018</xref>); and (2) inefficient for large CNNs, as computing exact gradients becomes prohibitively expensive.</p>
<p>It has been reported (<xref ref-type="bibr" rid="B15">Hendrycks and Gimpel, 2017</xref>; <xref ref-type="bibr" rid="B2">Aldahdooh et al., 2022</xref>) that the integration of multiple detections to limit adversary capabilities, a strategy termed &#x0201C;<italic>meet the defense&#x0201D;</italic>, is promising in countering adversarial attacks. However, these studies did not include any implementation or experimental results. Indeed, exploiting synergistic effects of multiple detections is challenging due to their ineffectiveness. For example, certified methods (<xref ref-type="bibr" rid="B41">Weng et al., 2018</xref>; <xref ref-type="bibr" rid="B32">Raghunathan et al., 2018</xref>), a widely studied adversarial defense that employs minimum distance decoding (<xref ref-type="bibr" rid="B39">Tramer, 2022</xref>) for AEs detection, are generally effective only for AEs with small &#x02113;<sub><italic>p</italic></sub> distances from clean samples. This limitation renders them ineffective against semantically stealthy adversarial examples (<xref ref-type="bibr" rid="B12">Ghiasi et al., 2020</xref>), which achieve substantially large but visually imperceptible perturbations by manipulating image factors such as color or shadows. As such, <xref ref-type="bibr" rid="B12">Ghiasi et al. (2020)</xref> show that any perturbation on semantic attributes such as shadows is as effective as contrived noise. However, the vast number of semantics in images renders supervised detection inadequate for adversarial attacks (<xref ref-type="bibr" rid="B46">Zheng and Hong, 2018</xref>) due to its limited generalization and dependence on patterns specific to the existing dataset.</p>
<p>Meanwhile, many effective adversarial defenses (<xref ref-type="bibr" rid="B46">Zheng and Hong, 2018</xref>; <xref ref-type="bibr" rid="B11">Feinman et al., 2017</xref>) fail to scale efficiently as CNNs deepen and their number of parameters increases. For instance, the full covariance matrix &#x003A3; in I-Defender (<xref ref-type="bibr" rid="B46">Zheng and Hong, 2018</xref>) scales as <inline-formula><mml:math id="M01"><mml:mrow><mml:mi mathvariant="script">O</mml:mi></mml:mrow></mml:math></inline-formula>(<italic>d</italic><sup>2</sup>) w.r.t. the input dimension <italic>d</italic> of the hidden features extracted by CNN. Similarly, an increase in the number of features exponentially reduces the efficiency of Euclidean distance computations, as noted by <xref ref-type="bibr" rid="B11">Feinman et al. (2017)</xref>, due to the curse of dimensionality. The complexity of distance calculation also impacts density-based outlier detection methods, such as local outlier factor (LOF) (<xref ref-type="bibr" rid="B6">Breunig et al., 2000</xref>), which require repeated distance evaluations between data points and their neighbors in the feature space.</p>
<p>This study proposes MeetSafe, a scalable and effective detection for strong white-box adversarial attacks. MeetSafe selects critical hidden features obtained by convolutional layers, applies feature engineering techniques, and utilizes a Gaussian Mixture Model (GMM) to estimate their distribution. AEs are then identified by the GMM as outliers as they deviate from the distribution of benign hidden features.</p>
<p>In detail, we first observe that the Z-scores of hidden features from selected neurons are non-uniformly distributed (see <xref ref-type="fig" rid="F1">Figure 1b</xref>) in each CNN layer, with not all layers actively extracting features from AEs (see <xref ref-type="fig" rid="F3">Figure 3a</xref>&#x02013;<xref ref-type="fig" rid="F3">d</xref>). We propose two utility functions to identify the layers most sensitive to adversarial perturbations and the neurons with the largest Z-score differences between benign and adversarial features. By leveraging only the hidden features from the selected neurons, we significantly reduce the feature dimension. Then, our GMM estimates the distribution of selected features processed by three feature engineering techniques: feature squeezing (<xref ref-type="bibr" rid="B42">Xu et al., 2018</xref>), which compares the model&#x00027;s predictions on the original and feature-squeezed inputs;whitening (<xref ref-type="bibr" rid="B15">Hendrycks and Gimpel, 2017</xref>), captures the principal component of the covariance of inputs; and LOF, which estimates the sparsity of images based on their neighbors in the processed feature space. LOF is ineffective and inefficient in high-dimensional spaces as increased sample distances reduce critical feature impact and raise computational costs for density estimation. To enhance LOF&#x00027;s scalability in high-dimensional feature spaces and reduce statistical fluctuations for improved precision, we propose reachability density (LRD) for local outlier detection. LRD iteratively selects feature subsets with random cardinality and estimate the density of images based on their k-nearest neighbors in the feature space. Finally, an AE is identified if its sparsity, as estimated by the GMM, exceeds the 90th percentile. Experimental results on real-world datasets show that MeetSafe attains a 74%&#x0002B; detection accuracy against adaptive adversaries, 96%&#x0002B; against classic adversarial attacks, 79%&#x0002B; accuracy under white-box attacks, and at least 2.3&#x000D7; faster speed.</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>LRD for a MNIST model learned with RCE, utilizing the 10 hidden units with the largest, absolute Z-scores. <bold>(a)</bold> LRD metric, which uses the averaged maximum of the &#x02113;<sub>2</sub>- (<italic>d</italic>) and k-distance (<italic>d</italic>&#x02032;) among <italic>k</italic> nearest neighbors. <bold>(b)</bold> The distribution for test- and FGSM samples with &#x02113;<sub>2</sub> &#x02248; 5 and <italic>k</italic> &#x0003D; 8.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1631561-g0001.tif">
<alt-text>Diagram showing a neural network process and histogram. (a) Illustrates a neural network analyzing images of digits with arrows indicating layers and transformations. (b) Displays a histogram comparing cardinality and local reachability density for original (blue) and adversarial (orange) images.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2">
<title>2 Related work</title>
<sec>
<title>2.1 Notation</title>
<p>A deep neural network can be expressed as the mapping function <italic>f</italic><sup>(<italic>l</italic>)</sup>(<italic>X</italic>):&#x0211D;<sup><italic>m</italic></sup> &#x02192; &#x0211D;<sup><italic>L</italic></sup>, where the hidden units at layer <italic>l</italic> are <inline-formula><mml:math id="M02"><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> for the input <italic>X</italic> &#x02208; {&#x1D507; | &#x1D507; &#x02286; &#x0211D;<sup><italic>m</italic></sup>} in dataset &#x1D507;. For simplicity, we define units of the last layer of this network (<italic>i.e.</italic>, logits) to be <italic>z</italic><sub><italic>i</italic></sub> &#x02208; <italic>Z</italic>(<italic>X</italic>) and the predictions to be <italic>y</italic><sub><italic>i</italic></sub> &#x02208; <italic>Y</italic>(<italic>X</italic>). Neural networks often minimize the empirical risk with a loss function <inline-formula><mml:math id="M03"><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:math></inline-formula><sub><italic>f</italic></sub>(<bold>X</bold>) with a batch of &#x0211D;<sup><italic>B</italic>&#x000D7;<italic>m</italic></sup> as input. Our method also utilizes GMMs for detection, which is a linear superposition of Gaussians with the form <inline-formula><mml:math id="M04"><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow></mml:math></inline-formula>(<italic>X</italic>|&#x003BC;<sub><italic>i</italic></sub>, <bold>&#x003A3;</bold><sub><italic>i</italic></sub>) where <bold>&#x003A3;</bold> and &#x003BC; denote its covariance matrix and mean, respectively. Each Gaussian has a mixing coefficient &#x003C0;<sub><italic>i</italic></sub> that equals the probability <italic>p</italic>(&#x003BE;<sub><italic>i</italic></sub>) of a latent variable &#x003BE;<sub><italic>i</italic></sub>.</p>
</sec>
<sec>
<title>2.2 Adversarial attacks</title>
<p>One can define at least three threat models for adversarial attacks: the white-, gray-, and black-box scenario. The white-box setting indicates that the adversary has perfect information about the system. The detector should thus be deterministic for the adversary (<xref ref-type="bibr" rid="B4">Athalye et al., 2018</xref>). A weaker assumption is a gray-box model in which the attacker has no knowledge about the defenses. Black-box attacks only assume knowledge of the output and input space, possibly with access to a querying oracle. The empirical risk of an actual threat is often measured with the &#x02113;<sub><italic>p</italic></sub>-norm required by adversarial attacks like the ones below.</p>
<sec>
<title>2.2.1 Fast gradient sign method (FGSM) (<xref ref-type="bibr" rid="B13">Goodfellow et al., 2015</xref>)</title>
<p>FGSM is a one-step &#x02113;<sub>&#x0221E;</sub> perturbation toward the gradient of the loss function &#x02207;<sub><italic>X</italic></sub><inline-formula><mml:math id="M05"><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:math></inline-formula><sub><italic>f</italic></sub>. FGSMs perturbation is &#x003F5;&#x000B7;sign(&#x02207;<sub><italic>X</italic></sub><inline-formula><mml:math id="M06"><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:math></inline-formula><sub><italic>f</italic></sub>), where &#x003F5; is the &#x02113;<sub>&#x0221E;</sub> norm of the perturbation. The method assumes linearity in the proximate region of sample <italic>X</italic>.</p>
</sec>
<sec>
<title>2.2.2 Carlini &#x00026; Wagner (C&#x00026;W) (<xref ref-type="bibr" rid="B8">Carlini and Wagner, 2017b</xref>)</title>
<p>C&#x00026;W is a first-order constrained optimization that closely resembles <xref ref-type="bibr" rid="B36">Szegedy et al. (2014)</xref> method for adversarial example generation. Both define the objective to be <inline-formula><mml:math id="M07"><mml:mo>&#x02016;</mml:mo><mml:mi>X</mml:mi><mml:msub><mml:mrow><mml:mo>&#x02016;</mml:mo></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>c</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>. This objective function includes the &#x02113;<sub><italic>p</italic></sub> distance with a custom criterion <inline-formula><mml:math id="M08"><mml:mover accent="true"><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, modulated by the sensitivity parameter <italic>c</italic> and confidence parameter &#x003BA;. C&#x00026;W uses <inline-formula><mml:math id="M09"><mml:mover accent="true"><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mo class="qopname">max</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02260;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003BA;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x0002B;</mml:mo></mml:mrow></mml:msup></mml:math></inline-formula> where <italic>t</italic> is the targeted class.</p>
</sec>
<sec>
<title>2.2.3 DeepFool (<xref ref-type="bibr" rid="B30">Moosavi-Dezfooli et al., 2016</xref>)</title>
<p>DeepFool fits a hyperplane on the target model. The hyperplane is an aggregate of binary classifiers, which encloses the true class <italic>k</italic>. The algorithm applies Newton&#x00027;s method on the probits to move to the closest non-maximal class <italic>t</italic>. To misclassify the sample, a small overshoot &#x003B7; is added as scalar.</p>
<p>We describe the perturbations generated by the three methods as near-optimal as they are optimized within the constraints of the &#x02113;<sub><italic>p</italic></sub>-ball. However, recent studies on semantic perturbations have identified approaches that produce adversarial examples more closely aligned with human perception (<xref ref-type="bibr" rid="B27">Luo et al., 2022</xref>; <xref ref-type="bibr" rid="B10">Duan et al., 2021</xref>; <xref ref-type="bibr" rid="B45">Zhao et al., 2020</xref>; <xref ref-type="bibr" rid="B12">Ghiasi et al., 2020</xref>). For instance, PerC uses color differences, which considerably increases the &#x02113;<sub><italic>p</italic></sub> distance of adversarial examples. The primary focus in this study is on adaptive near-optimal perturbations on state-of-the-art defenses that <italic>do not</italic> rely on obfuscated gradients.</p>
</sec>
</sec>
<sec>
<title>2.3 Adversarial detection</title>
<sec>
<title>2.3.1 Adversarial pockets</title>
<p>A common intuition of adversarial perturbation is that it pushes examples off the manifold of training data. <xref ref-type="bibr" rid="B36">Szegedy et al. (2014)</xref> were the first to conjecture the idea with the Lipschitz constant. A high constant enables the manifold to be dense, with low-probability pockets containing adversarial examples. Therefore, generative classifiers may detect these adversarial pockets (<xref ref-type="bibr" rid="B22">Lee et al., 2018</xref>; <xref ref-type="bibr" rid="B33">Raghuram et al., 2021</xref>; <xref ref-type="bibr" rid="B43">Yin et al., 2019</xref>; <xref ref-type="bibr" rid="B11">Feinman et al., 2017</xref>; <xref ref-type="bibr" rid="B46">Zheng and Hong, 2018</xref>; <xref ref-type="bibr" rid="B24">Li et al., 2019</xref>). An example of this is Deep Bayes (<xref ref-type="bibr" rid="B24">Li et al., 2019</xref>), which uses a deep latent variable model on the logits to estimate a joint distribution. JTLA (<xref ref-type="bibr" rid="B33">Raghuram et al., 2021</xref>) aggregates class-conditional probabilities from each layer by computing <italic>k</italic>NN class counts. Others trained a more simple GMM (<xref ref-type="bibr" rid="B46">Zheng and Hong, 2018</xref>) and utilized Kernel Density Estimation (KDE) (<xref ref-type="bibr" rid="B11">Feinman et al., 2017</xref>) on deep layers. <xref ref-type="bibr" rid="B22">Lee et al. (2018)</xref> performed a density estimation with the Mahalanobis distance.</p>
</sec>
<sec>
<title>2.3.2 Boundary tilting</title>
<p>A geometric analysis renders a different perspective on adversarial examples. When the decision boundary tilts too much toward a submanifold of one class, then the distance of another classification is relatively close. <xref ref-type="bibr" rid="B37">Tanay and Griffin (2016)</xref> therefore measured adversarial strength as the deviation angle with a bisecting boundary that maximizes the inter-class distance. This angle can, without major performance hits, be higher along directions of low variance. Near-optimal perturbations may thus be detected by manipulating such components with semantic-preserving image filters (<xref ref-type="bibr" rid="B42">Xu et al., 2018</xref>; <xref ref-type="bibr" rid="B38">Tian et al., 2021</xref>; <xref ref-type="bibr" rid="B25">Liang et al., 2018</xref>). In particular, feature squeezing (<xref ref-type="bibr" rid="B42">Xu et al., 2018</xref>) uses median smoothing and bit-depth reduction. <xref ref-type="bibr" rid="B38">Tian et al. (2021)</xref> train a dual model on the sample&#x00027;s wavelet transform. Others (<xref ref-type="bibr" rid="B35">Song et al., 2018</xref>; <xref ref-type="bibr" rid="B16">Hu et al., 2019</xref>) propose denoisers which perturb samples with optimizers. Scene statistics may also detect the perturbation, like whitening (<xref ref-type="bibr" rid="B15">Hendrycks and Gimpel, 2017</xref>) that measures the variance of low-rank eigenvectors. <xref ref-type="bibr" rid="B23">Li and Li (2017)</xref> also use low-rank eigenvectors with their extremal value to detect extreme deviations, both (<xref ref-type="bibr" rid="B17">Kherchouche et al., 2020</xref>; <xref ref-type="bibr" rid="B1">Akhtar et al., 2018</xref>) train simple classifiers on BRISQUE&#x00027;s (<xref ref-type="bibr" rid="B29">Mittal et al., 2012</xref>) features, and Local Intrinsic Dimensionality (LID) (<xref ref-type="bibr" rid="B28">Ma et al., 2018</xref>) directly calculates the dimensionality. However, current adversarial detection methods are ineffective at identifying hidden anomalies in high-dimensional spaces and are not efficient for large dataset.</p>
<p><bold>Contributions</bold> of this study are as follows: (<bold>i</bold>) We propose MeetSafe, a scalable detection algorithm for adaptive adversarial examples. (<bold>ii</bold>) Two utility functions that allow LRD and other detectors to scale based on a unit&#x00027;s Z-scores or rate of change under perturbation. (<bold>iii</bold>) Extensive empirical evaluations on 4 datasets and 14 models that show effectiveness of whitening and MeetSafe under adaptive white-box attacks.</p>
</sec>
</sec>
</sec>
<sec id="s3">
<title>3 Method</title>
<p>The main idea of MeetSafe is to combine discrepant detectors in an ensemble. In particular, we use the scores of whitening (<xref ref-type="bibr" rid="B15">Hendrycks and Gimpel, 2017</xref>), feature squeezing (<xref ref-type="bibr" rid="B42">Xu et al., 2018</xref>), and a density estimation, called LRD, within a GMM. LRD makes two novel improvements on existing density estimates. First, we noticed that the activation&#x00027;s Z-score of hidden features is not uniform under perturbation (see <xref ref-type="fig" rid="F1">Figure 1b</xref>); we therefore use two utility functions to select the 10 units that were most anomalous under perturbations. Second, kernel density estimation does not adjust for local densities, which carries the risk of over-smoothing as illustrated by <xref ref-type="bibr" rid="B28">Ma et al. (2018)</xref>. Like Ma et al., LRD uses an extension of the <italic>k</italic>-distance.</p>
<p>We now turn to LRD and its relation to non-parametric methods. Then, we explain the used features and feature selection of LRD. Finally, this section introduces MeetSafe.</p>
<sec>
<title>3.1 Density estimation with k-distances</title>
<p>Non-parametric methods model the distribution <italic>p</italic>(<italic>X</italic>) with limited assumptions for the true distribution. This makes the models flexible. Distribution <italic>p</italic>(<italic>X</italic>) can, for instance, be generalized with its volume <italic>V</italic> and cardinality <italic>K</italic> of <italic>X</italic>&#x00027;s proximite region, given enough observations. In contrast to KDE, the <italic>k</italic>NN method fixes the cardinality and finds the appropriate volume from the data. For a sample <italic>X</italic>, one can then estimate <italic>p</italic>(<italic>X</italic>) using the frequentist notion:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>K</mml:mi><mml:mo>/</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mo>&#x1D507;</mml:mo><mml:mo>|</mml:mo><mml:mo>&#x000B7;</mml:mo><mml:mi>V</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the dataset &#x1D507; is sampled from <italic>p</italic>(<italic>X</italic>). The volume is defined by a sphere with the <italic>k</italic>-distance as radius, which makes the <italic>k</italic>-distance of a test sample <italic>X</italic> sufficient to approximate <italic>p</italic>(<italic>X</italic>). The estimate would only be shallow with limited information from its neighbors. Recursive calls on neighbors may improve the estimate due to greater depth.</p>
<p>Notice that BRISQUE hidden features may be affected by unequal standard deviations as these are not normalized. Some have thus more influence on the <italic>k</italic>-distance than others. This is not favorable, especially because we stated earlier that the low variance components may be an important characteristic of some adversarial examples. Our method will therefore use the scaled Euclidean distance. This normalizes the <italic>k</italic>-distance with respect to a diagonal covariance matrix <bold>&#x003A3;</bold>. In addition, we will lower the memory burden of the <italic>k</italic>NN algorithm in Section 3.2, after we discuss the details and idea of LRD.</p>
<sec>
<title>3.1.1 Local reachability density (LRD)</title>
<p>The intuition behind LRD comes from <xref ref-type="bibr" rid="B6">Breunig et al. (2000)</xref>, in which they propose LOF, a heuristic for finding local outliers. LRD extends the <italic>k</italic>-distance as it smooths out statistical fluctuations in at least two ways. First, the actual distance, called reachability, used to estimate the volume <italic>V</italic> is capped by the <italic>k</italic>-distance of the neighbor. Second, the average is taken among the <italic>k</italic> nearest neighbors. The reachability measure <italic>reach</italic>(<italic>X</italic><sub>1</sub>, <italic>X</italic><sub>2</sub>) of two nodes (<italic>A</italic> &#x00026; <italic>D</italic>) is demonstrated in <xref ref-type="fig" rid="F1">Figure 1</xref>, and the value depends on the volume of node <italic>D</italic> and its Euclidean distance to <italic>A</italic>. Whichever value is bigger equals the reachability from <italic>D</italic> to <italic>A</italic>:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M11"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mo class="qopname">max</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup></mml:mstyle><mml:mo>,</mml:mo><mml:msqrt><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mo>&#x003A3;</mml:mo></mml:mstyle></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msqrt></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>d</italic>&#x02032; is the <italic>k</italic>-distance. Substituting the reachability from <italic>A</italic> to its neighbors <bold>N</bold><sub><italic>A</italic></sub> in <xref ref-type="disp-formula" rid="E1">Equation 1</xref> yields its reachability density (<xref ref-type="disp-formula" rid="E3">Equation 3</xref>). Using reachability as a measure to assess the density in the proximate region of node <italic>A</italic> has the advantage that only a fixed amount of neighbors needs to be considered.</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>L</mml:mi><mml:mi>R</mml:mi><mml:mi>D</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo stretchy="false">|</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>N</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">|</mml:mo><mml:mo>/</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>N</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mstyle><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>h</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>We add one novel improvement called feature bagging (<xref ref-type="bibr" rid="B19">Lazarevic and Kumar, 2005</xref>). This enables LRD to capture higher dimensions. The generalizability of <italic>k</italic>NN degrades under these circumstances, as the distance between all data points becomes larger and individual features have less of an impact. Bagging is a popular approach to limit this issue. It takes a subset (with random cardinality) of the features for multiple iterations and returns a combined LRD score.</p>
</sec>
</sec>
<sec>
<title>3.2 Feature engineering for LRD</title>
<p>We now consider two possible Points Of Interests (POI) for LRD: the layer after convolution and the pixel values, where we refer to the former as Learned Feature Analysis (LFA). Specifically, we explain how we select its features for both options as without limiting its feature space, LRD would be space inefficient and suffer from sparse data.</p>
<p>On raw pixel values, we advise the use of BRISQUE. BRISQUE fits a Gaussian-like distribution on the raw image while maintaining structural information, which can evaluate the naturalness of an image. Moreover, BRISQUE is 149 times faster than wavelet methods such as DIIVINE and performs almost similar on white noise (<xref ref-type="bibr" rid="B29">Mittal et al., 2012</xref>).</p>
<p>On hidden layers, we extract a random set of hidden features. Depending on the chosen POI, the Z-scores of FGSM examples <italic>X</italic>&#x02032; will be used to select the best 10 features of BRISQUE or the best 10 hidden features. We will further explain this feature selection more formally.</p>
<sec>
<title>3.2.1 Selecting hidden features</title>
<p>For the hidden features <italic>f</italic><sup>(<italic>l</italic>)</sup>, a pool <inline-formula><mml:math id="M13"><mml:mi mathvariant="bold">P</mml:mi><mml:msub><mml:mrow><mml:mo>&#x02286;</mml:mo></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:math></inline-formula> is defined, so that its members are chosen randomly at initialization and preserved during execution. The detector then follows a watching scheme upon <inline-formula><mml:math id="M14"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula> and its utilities (<xref ref-type="disp-formula" rid="E4">Equation 4</xref>, <xref ref-type="disp-formula" rid="E5">5</xref>). The first equation calculates the difference in Z-scores of all features in the pool. It estimates the units that were relevant under perturbation. The second estimates the rate of change of one layer. The first utility is calculated with FGSM after every epoch, and this is the fastest evasion method we know, limiting the constraints on scalability or parameter updates. The second utility showed most potential in the final layers, making that our preferred choice (Section 5.1). Because of this, we believe the second utility is optional.</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M15"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mo>&#x1D507;</mml:mo><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>X</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mo>&#x1D507;</mml:mo></mml:mrow></mml:msub></mml:mstyle><mml:mfrac><mml:mrow><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E5"><label>(5)</label><mml:math id="M16"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo>&#x02016;</mml:mo><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi><mml:mo>&#x0007E;</mml:mo><mml:mo>&#x1D507;</mml:mo></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi><mml:mo>&#x0007E;</mml:mo><mml:mo>&#x1D507;</mml:mo></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mo>&#x02016;</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>&#x02016;</mml:mo><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi><mml:mo>&#x0007E;</mml:mo><mml:mo>&#x1D507;</mml:mo></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mo>&#x02016;</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
</sec>
<sec>
<title>3.3 MeetSafe</title>
<p>Our MeetSafe combines LRD, using hidden features as POI, with variance-based anomaly detection&#x02014;whitening (<xref ref-type="bibr" rid="B15">Hendrycks and Gimpel, 2017</xref>)&#x02014;and feature squeezing (<xref ref-type="bibr" rid="B42">Xu et al., 2018</xref>) in a GMM, for which an ablation study is given in Section 4.3. The scores of the three heuristics are learned through Expectation-Maximization (EM).</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M17"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mo>-</mml:mo><mml:mo class="qopname">log</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo class="qopname">^</mml:mo></mml:mover><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mover class="msup"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:mover><mml:msub><mml:mrow><mml:mi>&#x003C0;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mi mathvariant="script">N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="script">H</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mo>&#x003A3;</mml:mo></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Concretely, to detect a sample, we first select the best features based on <italic>U</italic><sub><inline-formula><mml:math id="M18"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula></sub> for LRD and the eigenvectors of the training data (<xref ref-type="table" rid="T6">Algorithm 1</xref>). Then, we evaluate the three heuristics (denoted as <inline-formula><mml:math id="M19"><mml:mrow><mml:mi mathvariant="script">H</mml:mi></mml:mrow></mml:math></inline-formula><sub><italic>X</italic></sub>). Assuming normality, a three-dimensional GMM fitted on benign data can classify the sample as malicious when it exceeds the 90th percentile of the <xref ref-type="disp-formula" rid="E6">Equation 6</xref>. The workflow of MeetSafe is shown in <xref ref-type="table" rid="T7">Algorithm 2</xref>.</p>
<table-wrap position="float" id="T6">
<label>Algorithm 1</label>
<caption><p>MeetSafe&#x00027;s feature extraction.</p></caption>
<table frame="lhs" rules="none">
<tbody>
<tr>
<td align="left" valign="top" style="font-family:courier;"><bold>Require</bold>: &#x000A0;&#x1D507;: Dataset of benign samples; <inline-formula><mml:math id="M20"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula>: Set of basis functions that maximizes <italic>U</italic><sub><inline-formula><mml:math id="M21"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula></sub>; <inline-formula><mml:math id="M22"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula><sub><italic>max</italic></sub>: Maximum amount of features; <italic>X</italic>: A benign or adversarial sample.</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;"><bold>Ensure</bold>: &#x000A0;<inline-formula><mml:math id="M23"><mml:mrow><mml:mi mathvariant="script">H</mml:mi></mml:mrow></mml:math></inline-formula><sub><inline-formula><mml:math id="M24"><mml:mrow><mml:mi mathvariant="script">X</mml:mi></mml:mrow></mml:math></inline-formula></sub><italic>X</italic>&#x00027;s Features</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">1: &#x000A0;<bold>if</bold> <italic>k</italic>NN or SVD is not initialized <bold>then</bold> &#x02003;&#x02003;&#x022B3; Prepare heuristics</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">2: &#x000A0;&#x02003; <inline-formula><mml:math id="M25"><mml:mi>k</mml:mi><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mo>&#x02190;</mml:mo><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>k</mml:mi><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x1D507;</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> where {<italic>f</italic><sup>(<italic>l</italic>)</sup>} &#x02208; <inline-formula><mml:math id="M26"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">3: &#x000A0;&#x02003; <inline-formula><mml:math id="M27"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>U</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>S</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>V</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02190;</mml:mo><mml:mi>S</mml:mi><mml:mi>V</mml:mi><mml:mi>D</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>&#x1D507;</mml:mo></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> &#x02003;&#x02003;&#x022B3; Truncated to <inline-formula><mml:math id="M28"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula><sub><italic>max</italic></sub></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">4: &#x000A0;&#x02003; <bold>for</bold> <italic>X</italic> &#x02208; &#x1D507; <bold>do</bold></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">5: &#x000A0;&#x02003;&#x02003; <inline-formula><mml:math id="M29"><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02190;</mml:mo><mml:mi>X</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003F5;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mstyle class="textit" mathvariant="italic"><mml:mtext>sign</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mo>&#x02207;</mml:mo></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi mathvariant="bold">L</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">6: &#x000A0;&#x02003;&#x02003; <inline-formula><mml:math id="M30"><mml:msubsup><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msubsup><mml:mo>&#x02190;</mml:mo><mml:mi>X</mml:mi><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>V</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">7: &#x000A0;&#x02003;&#x02003; <inline-formula><mml:math id="M31"><mml:msubsup><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msubsup><mml:mo>&#x02190;</mml:mo><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>V</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">8: &#x000A0;&#x000A0; <bold>end for</bold></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">9: &#x000A0;&#x000A0; <inline-formula><mml:math id="M32"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi mathvariant="bold">P</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mo>&#x1D507;</mml:mo><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>X</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mo>&#x1D507;</mml:mo></mml:mrow></mml:munder><mml:mfrac><mml:mrow><mml:mo>|</mml:mo><mml:msubsup><mml:mrow><mml:mi mathvariant="bold">P</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msubsup><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mi mathvariant="bold">P</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msubsup><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi mathvariant="bold">P</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:math></inline-formula> &#x02003;&#x02003;&#x022B3; <xref ref-type="disp-formula" rid="E4">Equation 4</xref></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">10: &#x000A0;&#x000A0; <inline-formula><mml:math id="M33"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>V</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:mi>t</mml:mi><mml:mi>o</mml:mi><mml:mi>p</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>n</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> &#x02003;&#x02003;&#x022B3; Pick the best eigenvectors</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">11: &#x000A0;<bold>end if</bold></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">12: &#x000A0;<inline-formula><mml:math id="M34"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>N</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:mi>k</mml:mi><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> where {<italic>f</italic><sup>(<italic>l</italic>)</sup>} &#x02208; <inline-formula><mml:math id="M35"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">13: &#x000A0;<bold>&#x003A3;</bold>&#x02190;<italic>Diag</italic>(&#x003C3;<sub><inline-formula><mml:math id="M36"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula></sub>)</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">14: &#x000A0;<inline-formula><mml:math id="M37"><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>R</mml:mi><mml:mi>D</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>N</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mo>&#x003A3;</mml:mo></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:math></inline-formula> where {<italic>f</italic><sup>(<italic>l</italic>)</sup>} &#x02208; <inline-formula><mml:math id="M38"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula> &#x02003;&#x02003;&#x022B3; <xref ref-type="disp-formula" rid="E2">Equation 2</xref>, <xref ref-type="disp-formula" rid="E3">3</xref>, with feature bagging</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">15: &#x000A0;<italic>X</italic><sub><italic>bd</italic></sub>, <italic>X</italic><sub><italic>bl</italic></sub> &#x02190; <italic>Reduce</italic>_<italic>Bit</italic>_<italic>Depth</italic>(<italic>X</italic>), <italic>Blur</italic>(<italic>X</italic>)</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">16: &#x000A0;<inline-formula><mml:math id="M39"><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:mo class="qopname">max</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mo>&#x02016;</mml:mo><mml:mi>Y</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>Y</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mo>&#x02016;</mml:mo></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x02016;</mml:mo><mml:mi>Y</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>Y</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mo>&#x02016;</mml:mo></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">17: &#x000A0;<italic>H</italic><sub>2</sub> &#x02190; <italic>Var</italic>(<italic>X</italic><bold>V</bold><sub><italic>t</italic></sub>) &#x02003;&#x02003;&#x022B3; Whitening <xref ref-type="bibr" rid="B15">Hendrycks and Gimpel (2017)</xref></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">18: &#x000A0;<bold>return</bold> {<italic>H</italic><sub>0</sub>, <italic>H</italic><sub>1</sub>, <italic>H</italic><sub>2</sub>} &#x02003;&#x02003;&#x022B3; <italic>H</italic><sub>1</sub> is Feature Squeezing <xref ref-type="bibr" rid="B42">Xu et al. (2018)</xref></td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap position="float" id="T7">
<label>Algorithm 2</label>
<caption><p>MeetSafe.</p></caption>
<table frame="lhs" rules="none">
<tbody>
<tr>
<td align="left" valign="top" style="font-family:courier;"><bold>Require</bold>: &#x000A0;&#x1D507;: dataset of benign samples; <inline-formula><mml:math id="M40"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula><sub><italic>max</italic></sub>: maximum amount of features; <italic>K</italic>, &#x003C4;<sub>90</sub>: Gaussian components and threshold; <italic>f</italic><sup>(<italic>l</italic>)</sup>: basis function of the neural network; <bold>X</bold>&#x02032;: suspicious samples.</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;"><bold>Ensure</bold>: &#x000A0;<inline-formula><mml:math id="M41"><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>X</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:math></inline-formula>: MeetSafe classifications.</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">1: &#x000A0;<inline-formula><mml:math id="M42"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>0</mml:mtext></mml:mstyle></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">2: &#x000A0;<bold>for</bold> <italic>X</italic> &#x02208; &#x1D507; and basic block <italic>l</italic> <bold>do</bold> &#x02003;&#x02003;&#x022B3; Get the utility of each layer (optional)</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">3: &#x000A0;&#x02003; <inline-formula><mml:math id="M43"><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02190;</mml:mo><mml:mi>X</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003F5;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mstyle class="textit" mathvariant="italic"><mml:mtext>sign</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mo>&#x02207;</mml:mo></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi mathvariant="bold">L</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">4: &#x000A0;&#x02003; <inline-formula><mml:math id="M44"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:mi>S</mml:mi><mml:mi>e</mml:mi><mml:mi>q</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>A</mml:mi><mml:mi>v</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> &#x02003;&#x02003;&#x022B3; <xref ref-type="disp-formula" rid="E5">Equation 5</xref></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">5: &#x000A0;<bold>end for</bold></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">6: &#x000A0;Let {<italic>l</italic>} have the largest <inline-formula><mml:math id="M45"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:math></inline-formula> value and let <inline-formula><mml:math id="M46"><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mo>&#x02286;</mml:mo></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:math></inline-formula>.</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">7: &#x000A0;<bold>for</bold> <italic>X</italic> &#x02208; &#x1D507; <bold>do</bold> &#x02003;&#x02003;&#x022B3; Get the utility of each hidden unit in <inline-formula><mml:math id="M47"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">8: &#x000A0;&#x02003; <inline-formula><mml:math id="M48"><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02190;</mml:mo><mml:mi>X</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003F5;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mstyle class="textit" mathvariant="italic"><mml:mtext>sign</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mo>&#x02207;</mml:mo></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi mathvariant="bold">L</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">9: &#x000A0;&#x02003; <inline-formula><mml:math id="M49"><mml:msub><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> where {<italic>f</italic><sup>(<italic>l</italic>)</sup>} &#x02208; <inline-formula><mml:math id="M50"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">10: &#x000A0;&#x02003; <inline-formula><mml:math id="M51"><mml:msub><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> where {<italic>f</italic><sup>(<italic>l</italic>)</sup>} &#x02208; <inline-formula><mml:math id="M52"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">11: &#x000A0;<bold>end for</bold></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">12: &#x000A0;<inline-formula><mml:math id="M53"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="bold">P</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mo>&#x1D507;</mml:mo><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>X</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mo>&#x1D507;</mml:mo></mml:mrow></mml:munder><mml:mfrac><mml:mrow><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">P</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">P</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mo>&#x003C3;</mml:mo></mml:mstyle></mml:mrow><mml:mrow><mml:mi mathvariant="bold">P</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:math></inline-formula> &#x02003;&#x02003;&#x022B3; <xref ref-type="disp-formula" rid="E4">Equation 4</xref></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">13: &#x000A0;<inline-formula><mml:math id="M54"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula>&#x02190;<italic>top</italic>_<italic>n</italic>(<inline-formula><mml:math id="M55"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula><sub><italic>max</italic></sub>, <inline-formula><mml:math id="M56"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula>) &#x02003;&#x02003;&#x022B3; Pick the best hidden units</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">14: &#x000A0;<bold>for</bold> <italic>X</italic> &#x02208; &#x1D507; <bold>do</bold> &#x02003;&#x02003;&#x022B3; Initialize the GMM</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">15: &#x000A0;&#x02003; Initialize {&#x003BC;<sub><italic>i</italic></sub>} &#x02208; &#x0211D; via the K-means algorithm and <inline-formula><mml:math id="M57"><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C0;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mo>&#x003A3;</mml:mo></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:mi>&#x0211D;</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> uniformly at random.</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">16: &#x000A0;&#x02003; <inline-formula><mml:math id="M58"><mml:mrow><mml:mi mathvariant="script">H</mml:mi></mml:mrow></mml:math></inline-formula><sub><italic>X</italic></sub> &#x02190; <italic>Extract</italic>_<italic>Features</italic>(&#x1D507;, <inline-formula><mml:math id="M59"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M60"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula><sub><italic>max</italic></sub>, <italic>X</italic>) &#x02003;&#x02003;&#x022B3; <xref ref-type="table" rid="T6">Algorithm 1</xref></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">17: &#x000A0;<bold>end for</bold></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">18: &#x000A0;&#x003BC;<sub><italic>i</italic></sub>, <bold>&#x003A3;</bold><sub><italic>i</italic></sub>, &#x003C0;<sub><italic>i</italic></sub> &#x02190; <italic>EM</italic>(&#x003BC;<sub><italic>i</italic></sub>, <bold>&#x003A3;</bold><sub><italic>i</italic></sub>, &#x003C0;<sub><italic>i</italic></sub>, <inline-formula><mml:math id="M61"><mml:mrow><mml:mi mathvariant="script">H</mml:mi></mml:mrow></mml:math></inline-formula><sub><italic>X</italic></sub>) <italic>i</italic> &#x02208; [0..<italic>K</italic>), &#x02200; <italic>X</italic> &#x02208; &#x1D507;</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">19: &#x000A0;<bold>for</bold> <italic>X</italic>&#x02032; &#x02208; <bold>X</bold>&#x02032; <bold>do</bold> &#x02003;&#x02003;&#x022B3; Classify samples</td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">20: &#x000A0;&#x02003; <inline-formula><mml:math id="M62"><mml:msub><mml:mrow><mml:mi mathvariant="bold">H</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:mi>E</mml:mi><mml:mi>x</mml:mi><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>F</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>&#x1D507;</mml:mo><mml:mo>,</mml:mo><mml:mi mathvariant="script">P</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">21: &#x000A0;<bold>end for</bold></td>
</tr>
<tr>
<td align="left" valign="top" style="font-family:courier;">22: &#x000A0;<bold>return</bold> <inline-formula><mml:math id="M63"><mml:mo>-</mml:mo><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mover class="msup"><mml:mrow><mml:mo class="qopname">&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:mover><mml:msub><mml:mrow><mml:mi>&#x003C0;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mi mathvariant="bold">N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="bold">H</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mo>&#x003A3;</mml:mo></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x0003E;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mn>90</mml:mn></mml:mrow></mml:msub><mml:mtext>&#x000A0;&#x000A0;</mml:mtext><mml:mo>&#x02200;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>X</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> &#x02003;&#x02003;&#x022B3; <xref ref-type="disp-formula" rid="E6">Equation 6</xref></td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s4">
<title>4 Experiments</title>
<p>We evaluate MeetSafe and LRD against several adversarial attacks including FGSM, DeepFool, and C&#x00026;W. The experiments will demonstrate white- and/or gray-box performance for four datasets: Tiny-ImageNet (<xref ref-type="bibr" rid="B20">Le and Yang, 2015</xref>), CIFAR-10 (<xref ref-type="bibr" rid="B18">Krizhevsky and Hinton, 2009</xref>), MNIST (<xref ref-type="bibr" rid="B21">LeCun, 1998</xref>), and STL-10 (<xref ref-type="bibr" rid="B9">Coates et al., 2011</xref>). Only for Tiny-ImageNet, we resized the samples to be in &#x0211D;<sup>3&#x000D7;64&#x000D7;64</sup>. The attacks are restricted to an &#x02113;<sub>2</sub> distance to ensure a fair comparison across datasets and &#x02113;<sub>&#x0221E;</sub>-based methods. For instance, an &#x02113;<sub>&#x0221E;</sub> distance permits more noise for higher resolution images. Consequently, we use the &#x003F5; parameter given by <xref ref-type="disp-formula" rid="E7">Equation 7</xref>, so that the maximum allowed perturbation of FGSM equals that of &#x02113;<sub>2</sub> methods (&#x003B4;<sub>max</sub>); where the image <italic>X</italic> is given by a &#x0211D;<sup><italic>m</italic></sup> flattened matrix.</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M64"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>&#x003F5;</mml:mi><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mo>&#x02016;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mtext>max</mml:mtext></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mo>&#x02016;</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>/</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msqrt></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Each attack, defense, and target model is re-implemented in PyTorch. Herewith, we evaluated 14 models based on ResNet-50 (<xref ref-type="bibr" rid="B14">He et al., 2016</xref>) and VGG-13 (<xref ref-type="bibr" rid="B34">Simonyan and Zisserman, 2015</xref>). Ten of them are trained with robust optimization techniques, utilizing gradient smoothing (RCE) (<xref ref-type="bibr" rid="B31">Pang et al., 2018</xref>) or adversarial training with FGSM (&#x02113;<sub>2</sub>-radii of 5) (AL) (<xref ref-type="bibr" rid="B13">Goodfellow et al., 2015</xref>).</p>
<p>The experiments also include some related methods that will be compared to MeetSafe and LRD. The baseline for MeetSafe (MS) is KDE with predictive uncertainty (KDE&#x0002B;BU) (<xref ref-type="bibr" rid="B11">Feinman et al., 2017</xref>), I-Defender (I-Def) (<xref ref-type="bibr" rid="B46">Zheng and Hong, 2018</xref>), and the Mahalanobis measure (MAH) (<xref ref-type="bibr" rid="B22">Lee et al., 2018</xref>). Additional work that we tested are LID (<xref ref-type="bibr" rid="B28">Ma et al., 2018</xref>), Whitening (PCA) (<xref ref-type="bibr" rid="B15">Hendrycks and Gimpel, 2017</xref>), Feature Squeezing (FSQ) (<xref ref-type="bibr" rid="B42">Xu et al., 2018</xref>), extremal value (EXM) (<xref ref-type="bibr" rid="B23">Li and Li, 2017</xref>), and <xref ref-type="bibr" rid="B17">Kherchouche et al. (2020)</xref>&#x00027;s third model for BRISQUE (SVM).</p>
<sec>
<title>4.1 Experimental setup</title>
<p>We trained the models for 150 epochs on predetermined training sets. During training, a batch size was used of 256, learning rate of 0.01 with momentum 0.9 under a cosine annealing schedule, and 1e &#x02212; 4 weight decay. The model is also adapted to exhibit required invariances. All training samples are normalized on each channel, randomly flipped horizontally, and randomly cropped within a padding of 4. Adversarially learned models were additionally trained half-on-half on benign and perturbed data.</p>
<p>We evaluated the detectors on unseen test images and their perturbed variants as follows. First, we evaluate the defense and model on non-adaptive gray-box perturbations. That includes the one-step FGSM perturbation as well as the DeepFool and C&#x00026;W-&#x02113;<sub>2</sub>. Second, for each defense, we evaluate its best-performing technique (RCE or AL) on DeepFool against adaptive white-box attacks.</p>
<p>White-box attacks can be generated by adding the detector&#x00027;s likelihood function (<xref ref-type="disp-formula" rid="E6">Equation 6</xref>) to C&#x00026;W&#x00027;s objective (<xref ref-type="bibr" rid="B7">Carlini and Wagner, 2017a</xref>). In essence, this optimizes a multi-objective gradient with Adam that considers both the gradient of the detector&#x00027;s internals and the confidence of the target model:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M65"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mo>&#x02016;</mml:mo><mml:mi>X</mml:mi><mml:msub><mml:mrow><mml:mo>&#x02016;</mml:mo></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>c</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>c</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msup><mml:mo>&#x000B7;</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:msup><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mo class="qopname">log</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo class="qopname">^</mml:mo></mml:mover><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003BA;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x0002B;</mml:mo></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003C4; is a given threshold and <italic>c</italic><sup>&#x0002A;</sup> a constant that controls the sensitivity toward the detector&#x00027;s gradient, optimized with binary search. The sample is updated with perturbation &#x003B4; when its aggregate <italic>X</italic> &#x0002B; &#x003B4; fools successfully. For our experiments, we limit the perturbation to a &#x02113;<sub>2</sub> distance of 5.</p>
<p>Our white-box attack follows an all-or-nothing criterion: the batch with adversarial examples is either clean or fully successful. For this reason, we assume that the detector is successful if either the white-box perturbation is detected or classified by the target model. Its true positives are thus in the set <inline-formula><mml:math id="M66"><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>X</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mo>|</mml:mo><mml:mtext>&#x000A0;argmax&#x000A0;</mml:mtext><mml:mi>Y</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02260;</mml:mo><mml:mi>k</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x02228;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mo>-</mml:mo><mml:mo class="qopname">log</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo class="qopname">^</mml:mo></mml:mover><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02264;</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula> for true class <italic>k</italic> and threshold &#x003C4;.</p>
<p>The performance of the defenses is measured using its overall detection accuracy of one test run in both adversarial and benign situations, where the detector classifies at a TNR of 90%&#x0002B;. The adversarial setting may include samples without perturbation when the target model already misclassifies the clean sample. We therefore have an optimal detection accuracy of <inline-formula><mml:math id="M67"><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="bold">R</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac></mml:math></inline-formula> with standard empirical risk <inline-formula><mml:math id="M68"><mml:mrow><mml:mi mathvariant="script">R</mml:mi></mml:mrow></mml:math></inline-formula><sub><italic>f</italic></sub> of the target model.</p>
<p>During test runs, we reduced the batch size to 128 (64 for white-box); other hyperparameters, used for the attacks and defenses, were as follows. The magnitude &#x003F5; of FGSM is deduced from <xref ref-type="disp-formula" rid="E7">Equation 7</xref>, DeepFool had an overshoot of 0.02, and C&#x00026;W executed 5 steps with 500 iterations (10 and 1000 for white-box) under a 0.05 confidence &#x003BA;. For all defenses, we applied the same feature selection. That took the best 10 in a pool of at most 500 features, given <italic>U</italic><sub><inline-formula><mml:math id="M69"><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula></sub>. We choose the <italic>k</italic> of kNN to be 8 for LRD and LOF based on the ablation study in Sec. 5.3. See Section 5.3 for details. The experiments were conducted on AMD Ryzen 7 7700X and Nvidia RTX 4070 Ti.</p>
</sec>
<sec>
<title>4.2 Model performance</title>
<p>The accuracy of the models is shown in <xref ref-type="table" rid="T1">Table 1</xref>. The CIFAR-10 ResNet-50 model is able to reach an average cross-entropy of 0.0249 and a test accuracy of 93.2%. The cross-entropy for robust optimization techniques is notably higher, and this increased to 0.47 for adversarial training and 431.1 for RCE. A higher value for RCE was expected as almost each class now adds to the error instead of only the true class. We also observe that the fit and convergence changes dramatically when the amount of classes is increased. Take Tiny-ImageNet which has 200 classes, where the others have 10, a RCE model trained on ImageNet does only reach an accuracy of 3%. When we test the STL-10 subset, RCE does not show this behavior.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Accuracies and proportion of stationary points for FGSM of the trained target models on the CIFAR-10, Tiny-ImageNet, STL-10, and MNIST testing set, respectively.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left" rowspan="2"><bold>Model</bold></th>
<th valign="top" align="center" rowspan="2"><bold>Learn. rate</bold></th>
<th valign="top" align="center" rowspan="2"><bold>Robust optim.</bold></th>
<th valign="top" align="center" rowspan="2"><bold>Evasions (&#x02113;<sub>2</sub> &#x02264; 5) &#x02192;</bold><break/><bold>Stat. Points (%)</bold> &#x02193;</th>
<th valign="top" align="center" colspan="4"><bold>Model&#x00027;s top-1 accuracy</bold> &#x021D1;</th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="center"><bold>Benign</bold></th>
<th valign="top" align="center"><bold>FGSM</bold><break/> <bold><xref ref-type="disp-formula" rid="E7">Equation 7</xref></bold></th>
<th valign="top" align="center"><bold>C&#x00026;W</bold><break/>&#x003BA; &#x0003D; 0.05</th>
<th valign="top" align="center"><bold>DeepFool</bold><break/>&#x003B7; &#x0003D; 0.02</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">ResNet-50</td>
<td valign="top" align="center">0.01</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">50.5, 0.0</td>
<td valign="top" align="center">0.932, 0.581</td>
<td valign="top" align="center">0.619, 0.013</td>
<td valign="top" align="center">0.000, 0.000</td>
<td valign="top" align="center">0.061, 0.188</td>
</tr> <tr>
<td/>
<td/>
<td/>
<td valign="top" align="center">13.7, 12.8</td>
<td valign="top" align="center">0.711, 0.992</td>
<td valign="top" align="center">0.111, 0.307</td>
<td valign="top" align="center">0.008, 0.000</td>
<td valign="top" align="center">0.214, 0.130</td>
</tr> <tr>
<td valign="top" align="left">ResNet-50</td>
<td valign="top" align="center">0.01</td>
<td valign="top" align="center">AL</td>
<td valign="top" align="center">37.9, 0.0</td>
<td valign="top" align="center">0.916, 0.559</td>
<td valign="top" align="center">0.640, 0.176</td>
<td valign="top" align="center">0.000, 0.003</td>
<td valign="top" align="center">0.086, 0.149</td>
</tr> <tr>
<td/>
<td/>
<td/>
<td valign="top" align="center">2.6, 8.1</td>
<td valign="top" align="center">0.612, 0.991</td>
<td valign="top" align="center">0.229, 0.2704</td>
<td valign="top" align="center">0.200, 0.000</td>
<td valign="top" align="center">0.221, 0.385</td>
</tr> <tr>
<td valign="top" align="left">ResNet-50</td>
<td valign="top" align="center">0.01</td>
<td valign="top" align="center">RCE</td>
<td valign="top" align="center">0.0, 0.0</td>
<td valign="top" align="center">0.885, 0.030</td>
<td valign="top" align="center">0.485, 0.003</td>
<td valign="top" align="center">0.000, 0.000</td>
<td valign="top" align="center">0.106, 0.015</td>
</tr> <tr>
<td/>
<td/>
<td/>
<td valign="top" align="center">0.0, 0.0</td>
<td valign="top" align="center">0.614, 0.990</td>
<td valign="top" align="center">0.081, 0.085</td>
<td valign="top" align="center">0.000, 0.000</td>
<td valign="top" align="center">0.147, 0.228</td>
</tr> <tr>
<td valign="top" align="left">VGG-13</td>
<td valign="top" align="center">0.01</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">22.4</td>
<td valign="top" align="center">0.928</td>
<td valign="top" align="center">0.294</td>
<td valign="top" align="center">0.000</td>
<td valign="top" align="center">0.096</td>
</tr> <tr>
<td valign="top" align="left">VGG-13</td>
<td valign="top" align="center">0.01</td>
<td valign="top" align="center">AL</td>
<td valign="top" align="center">2.9</td>
<td valign="top" align="center">0.885</td>
<td valign="top" align="center">0.578</td>
<td valign="top" align="center">0.000</td>
<td valign="top" align="center">0.158</td>
</tr> <tr>
<td valign="top" align="left">VGG-13</td>
<td valign="top" align="center">0.01</td>
<td valign="top" align="center">RCE</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">0.883</td>
<td valign="top" align="center">0.329</td>
<td valign="top" align="center">0.000</td>
<td valign="top" align="center">0.060</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>VGG-13 is only trained on CIFAR-10. More details in-text.</p>
</table-wrap-foot>
</table-wrap>
<p>Robust optimization shows a descent mitigation of FGSM for all models and no meaningful mitigation of C&#x00026;W attacks. The results of DeepFool show that the optimized models are often more robust than the plain ones under smaller perturbations. Still, the high adversarial accuracy of the plain model is somewhat unexpected. However, this may have a clear reason. Namely, the confidence of this model is higher, which causes vanishing gradients.</p>
<sec>
<title>4.2.1 Vanishing gradients</title>
<p>The confidence score of the plain ResNet-50 is near a unit vector toward the correct class for some images (<xref ref-type="table" rid="T1">Table 1</xref>), which makes that gradient zero due to rounding errors. Such images sit on a stationary point for the current parameters. This is a major drawback of FGSM, but not present for DeepFool and C&#x00026;W which use gradients of different loss function. Vanishing gradients do give a sense of robustness for the plain model, while it is probably not.</p>
</sec>
<sec>
<title>4.2.2 Utility of hidden layers</title>
<p>The utilities discussed in Section 3.2 grow more or less each layer for the CIFAR-10 ResNet. The RCE and plain model exhibit the largest normalized &#x02113;<sub>2</sub> distance at the fourth bottleneck and the smallest distance at the raw input. The utility <inline-formula><mml:math id="M70"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:math></inline-formula> at the first and last bottleneck differs significantly with RCE; as for 5 samples, the 95% t-confidence interval (CI) is 0.29 &#x000B1; 0.003 and 0.51 &#x000B1; 0.01, respectively. This supports the unfolding intuition of Bengio <italic>et al</italic>. (<xref ref-type="bibr" rid="B5">Bengio et al., 2013</xref>). Although, we find the behavior of adversarial learned models to be different. There, the first layers seem to be the most sensitive, with the first bottleneck (0.66 &#x000B1; 0.14) having a higher utility than the fourth (0.48 &#x000B1; 0.13). Detection methods may thus be fine-tuned by utilizing different layers.</p>
</sec>
</sec>
<sec>
<title>4.3 Detection results</title>
<p>The following section will primarily discuss the performance on near-optimal perturbations. We showcase more results in Section 4.1 regarding the accuracies under semantic adversarial attacks such as shadow attack (<xref ref-type="bibr" rid="B12">Ghiasi et al., 2020</xref>) and PerC (<xref ref-type="bibr" rid="B45">Zhao et al., 2020</xref>).</p>
<sec>
<title>4.3.1 Against gray-box attacks</title>
<p>We start by examining gray-box attacks on CIFAR-10. This provides a more comprehensive understanding of our method&#x00027;s performance. <xref ref-type="table" rid="T2">Table 2</xref> shows LRD in addition to various other works. Instance-based methods similar to ours are KDE&#x0002B;BU, LID, and MAH. We see that LID does not lead to practical results. On the other hand, LRD reaches an accuracy above 85% for FGSM perturbations, and this outperforms similar methods.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Accuracies of several detection algorithms against gray- and white-box (GB/WB) adversaries with a &#x02113;<sub>2</sub>-radii of 5.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th rowspan="3"/>
<th valign="top" align="left" rowspan="3"><bold>Evasion attack</bold></th>
<th valign="top" align="left" rowspan="3"><bold>Robust optim</bold>.</th>
<th valign="top" align="left" rowspan="3"><bold>POI &#x02192; Stat. points (%) &#x02193;</bold></th>
<th valign="top" align="center" colspan="9"><bold>Detection accuracy</bold> &#x021D1;</th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="center" colspan="3"><bold>Units of the fourth bottleneck</bold></th>
<th valign="top" align="center"><bold>max</bold> <inline-formula><mml:math id="M71"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:math></inline-formula></th>
<th valign="top" align="center" colspan="4"><bold>Scene statistics</bold></th>
<th valign="top" align="center"><bold>Logits</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="center"><bold>LRD</bold></th>
<th valign="top" align="center"><bold>KDE&#x0002B;BU</bold></th>
<th valign="top" align="center"><bold>LID</bold></th>
<th valign="top" align="center"><bold>EXM</bold></th>
<th valign="top" align="center"><bold>LRD</bold></th>
<th valign="top" align="center"><bold>SVM</bold></th>
<th valign="top" align="center"><bold>PCA</bold></th>
<th valign="top" align="center"><bold>FSQ</bold></th>
<th valign="top" align="center"><bold>MAH</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left" rowspan="8"><italic>CIFAR-10</italic></td>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center">0.680</td>
<td valign="top" align="center">0.545</td>
<td valign="top" align="center">0.508</td>
<td valign="top" align="center">0.584</td>
<td valign="top" align="center"><bold>0.698</bold></td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.725</bold></td>
<td valign="top" align="center"><bold>0.722</bold></td>
<td valign="top" align="center">0.570</td>
<td valign="top" align="center">0.596</td>
</tr>
<tr>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="left">AL</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center">0.852</td>
<td valign="top" align="center">0.663</td>
<td valign="top" align="center">0.586</td>
<td valign="top" align="center"><bold>0.863</bold></td>
<td valign="top" align="center">0.815</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.884</bold></td>
<td valign="top" align="center"><bold>0.878</bold></td>
<td valign="top" align="center">0.757</td>
<td valign="top" align="center">0.792</td>
</tr>
<tr>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="left">RCE</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center">0.644</td>
<td valign="top" align="center">0.818</td>
<td valign="top" align="center">0.545</td>
<td valign="top" align="center">0.726</td>
<td valign="top" align="center"><bold>0.875</bold></td>
<td valign="top" align="center"><bold>0.987</bold></td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.990</bold></td>
<td valign="top" align="center">0.636</td>
<td valign="top" align="center">0.639</td>
</tr>
<tr>
<td valign="top" align="left">DeepFool</td>
<td valign="top" align="left">AL</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center">0.516</td>
<td valign="top" align="center">0.596</td>
<td valign="top" align="center">0.511</td>
<td valign="top" align="center">0.499</td>
<td valign="top" align="center">0.500</td>
<td valign="top" align="center">0.502</td>
<td valign="top" align="center"><bold>0.663</bold></td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.890</bold></td>
<td valign="top" align="center"><bold>0.676</bold></td>
</tr>
<tr>
<td valign="top" align="left">DeepFool</td>
<td valign="top" align="left">RCE</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center"><bold>0.775</bold></td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.785</bold></td>
<td valign="top" align="center">0.513</td>
<td valign="top" align="center">0.564</td>
<td valign="top" align="center">0.498</td>
<td valign="top" align="center">0.501</td>
<td valign="top" align="center"><underline>0.533</underline></td>
<td valign="top" align="center"><bold>0.759</bold></td>
<td valign="top" align="center">0.756</td>
</tr>
<tr>
<td valign="top" align="left">C&#x00026;W</td>
<td valign="top" align="left">AL</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center">0.520</td>
<td valign="top" align="center">0.561</td>
<td valign="top" align="center">0.501</td>
<td valign="top" align="center">0.500</td>
<td valign="top" align="center">0.502</td>
<td valign="top" align="center"><underline>0.500</underline></td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.758</bold></td>
<td valign="top" align="center"><bold>0.744</bold></td>
<td valign="top" align="center"><bold>0.635</bold></td>
</tr>
<tr>
<td valign="top" align="left">C&#x00026;W</td>
<td valign="top" align="left">RCE</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center">0.621</td>
<td valign="top" align="center"><bold>0.707</bold></td>
<td valign="top" align="center">0.513</td>
<td valign="top" align="center">0.558</td>
<td valign="top" align="center"><underline>0.497</underline></td>
<td valign="top" align="center">0.500</td>
<td valign="top" align="center">0.582</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.780</bold></td>
<td valign="top" align="center"><bold>0.647</bold></td>
</tr>
<tr>
<td valign="top" align="left">C&#x00026;W</td>
<td valign="top" align="left">Best Perf.</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="center"><underline>0.516</underline></td>
<td valign="top" align="center"><underline>0.489</underline></td>
<td valign="top" align="center"><underline>0.453</underline></td>
<td valign="top" align="center"><underline>0.450</underline></td>
<td valign="top" align="center"><bold>0.660</bold></td>
<td valign="top" align="center"><bold>0.619</bold></td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.792</bold></td>
<td valign="top" align="center"><underline>0.484</underline></td>
<td valign="top" align="center"><underline>0.474</underline></td>
</tr> <tr>
<td valign="middle" align="left" rowspan="6">MNIST</td>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="left">AL</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center"><bold>0.928</bold></td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.957</bold></td>
<td valign="top" align="center">0.856</td>
<td valign="top" align="center">0.901</td>
<td valign="top" align="center"><bold>0.930</bold></td>
<td valign="top" align="center">0.927</td>
<td valign="top" align="center">0.614</td>
<td valign="top" align="center"><underline>0.828</underline></td>
<td valign="top" align="center">0.915</td>
</tr>
<tr>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="left">RCE</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center"><bold>0.955</bold></td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.975</bold></td>
<td valign="top" align="center">0.936</td>
<td valign="top" align="center">0.952</td>
<td valign="top" align="center"><bold>0.955</bold></td>
<td valign="top" align="center"><bold>0.974</bold></td>
<td valign="top" align="center">0.678</td>
<td valign="top" align="center">0.915</td>
<td valign="top" align="center"><underline>0.659</underline></td>
</tr>
<tr>
<td valign="top" align="left">DeepFool</td>
<td valign="top" align="left">AL</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center"><underline>0.744</underline></td>
<td valign="top" align="center"><underline><bold>0.829</bold></underline></td>
<td valign="top" align="center">0.592</td>
<td valign="top" align="center">0.684</td>
<td valign="top" align="center">0.745</td>
<td valign="top" align="center">0.820</td>
<td valign="top" align="center">0.645</td>
<td valign="top" align="center"><bold>0.915</bold></td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.927</bold></td>
</tr>
<tr>
<td valign="top" align="left">DeepFool</td>
<td valign="top" align="left">RCE</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center">0.925</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.960</bold></td>
<td valign="top" align="center">0.468</td>
<td valign="top" align="center"><underline>0.498</underline></td>
<td valign="top" align="center">0.770</td>
<td valign="top" align="center">0.533</td>
<td valign="top" align="center"><underline>0.505</underline></td>
<td valign="top" align="center"><bold>0.947</bold></td>
<td valign="top" align="center"><bold>0.945</bold></td>
</tr>
<tr>
<td valign="top" align="left">C&#x00026;W</td>
<td valign="top" align="left">Best Perf.</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center"><bold>0.894</bold></td>
<td valign="top" align="center">0.866</td>
<td valign="top" align="center">0.522</td>
<td valign="top" align="center">0.538</td>
<td valign="top" align="center"><underline>0.655</underline></td>
<td valign="top" align="center"><underline>0.505</underline></td>
<td valign="top" align="center">0.609</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.949</bold></td>
<td valign="top" align="center"><bold>0.947</bold></td>
</tr>
<tr>
<td valign="top" align="left">C&#x00026;W</td>
<td valign="top" align="left">Best Perf.</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="center">0.940</td>
<td valign="top" align="center">0.912</td>
<td valign="top" align="center"><underline>0.457</underline></td>
<td valign="top" align="center">0.558</td>
<td valign="top" align="center">0.983</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.990</bold></td>
<td valign="top" align="center">0.599</td>
<td valign="top" align="center"><bold>0.986</bold></td>
<td valign="top" align="center"><bold>0.988</bold></td>
</tr> <tr>
<td valign="middle" align="left" rowspan="6">STL-10</td>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="left">AL</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center"><bold>0.517</bold></td>
<td valign="top" align="center">0.498</td>
<td valign="top" align="center">0.505</td>
<td valign="top" align="center">0.503</td>
<td valign="top" align="center">0.502</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.524</bold></td>
<td valign="top" align="center"><bold>0.506</bold></td>
<td valign="top" align="center"><underline>0.497</underline></td>
<td valign="top" align="center">0.505</td>
</tr>
<tr>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="left">RCE</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center">0.528</td>
<td valign="top" align="center">0.571</td>
<td valign="top" align="center">0.511</td>
<td valign="top" align="center">0.498</td>
<td valign="top" align="center">0.511</td>
<td valign="top" align="center"><bold>0.540</bold></td>
<td valign="top" align="center">0.531</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.667</bold></td>
<td valign="top" align="center"><bold>0.635</bold></td>
</tr>
<tr>
<td valign="top" align="left">DeepFool</td>
<td valign="top" align="left">AL</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center"><bold>0.509</bold></td>
<td valign="top" align="center">0.503</td>
<td valign="top" align="center">0.504</td>
<td valign="top" align="center">0.502</td>
<td valign="top" align="center"><underline>0.495</underline></td>
<td valign="top" align="center"><underline>0.498</underline></td>
<td valign="top" align="center">0.500</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.588</bold></td>
<td valign="top" align="center"><bold>0.510</bold></td>
</tr>
<tr>
<td valign="top" align="left">DeepFool</td>
<td valign="top" align="left">RCE</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.647</bold></td>
<td valign="top" align="center"><bold>0.611</bold></td>
<td valign="top" align="center">0.515</td>
<td valign="top" align="center">0.506</td>
<td valign="top" align="center">0.505</td>
<td valign="top" align="center">0.493</td>
<td valign="top" align="center">0.500</td>
<td valign="top" align="center"><bold>0.634</bold></td>
<td valign="top" align="center">0.524</td>
</tr>
<tr>
<td valign="top" align="left">C&#x00026;W</td>
<td valign="top" align="left">Best Perf.</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center">0.512</td>
<td valign="top" align="center"><bold>0.524</bold></td>
<td valign="top" align="center">0.507</td>
<td valign="top" align="center">0.505</td>
<td valign="top" align="center">0.513</td>
<td valign="top" align="center">0.499</td>
<td valign="top" align="center">0.500</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.586</bold></td>
<td valign="top" align="center"><bold>0.522</bold></td>
</tr>
<tr>
<td valign="top" align="left">C&#x00026;W</td>
<td valign="top" align="left">Best Perf.</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="center"><underline><bold>0.507</bold></underline></td>
<td valign="top" align="center"><underline>0.498</underline></td>
<td valign="top" align="center"><underline>0.450</underline></td>
<td valign="top" align="center"><underline>0.470</underline></td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.806</bold></td>
<td valign="top" align="center">0.498</td>
<td valign="top" align="center"><underline>0.456</underline></td>
<td valign="top" align="center"><bold>0.547</bold></td>
<td valign="top" align="center"><underline>0.479</underline></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>White-box attacks are evaluated on the detector&#x00027;s best performing robust optimization under DeepFool. DeepFool&#x00027;s and C&#x00026;W&#x00027;s results are dependent on the error rate <inline-formula><mml:math id="M72"><mml:mrow><mml:mi mathvariant="script">R</mml:mi></mml:mrow></mml:math></inline-formula><sub><italic>f</italic></sub> of ResNet-50 (more details in-text). Top-3 results are bolded, and the worst-case of a detection algorithm is underlined. Top-1 results are highlighted in blue.</p>
</table-wrap-foot>
</table-wrap>
<p>We also consider Robust optimization beneficial. The detection accuracy is frequently higher with one of these methods. Particularly for PCA, its accuracy against DeepFool and FGSM shows a respective difference of 15% and 26%. Moreover, PCA shows strong and similar results as the supervised method SVM on FGSM; the extremal measure, that also uses PCA, is less effective. Finally, we consider feature squeezing&#x00027;s performance limited for FGSM perturbations. It achieves the lowest accuracy of 76% after LID.</p>
<p>These results largely change for smaller perturbations. SVM drops from 90%&#x0002B; accuracy to a random classifier. In fact, almost all methods suffer from smaller perturbations, except for purification measures such as feature squeezing. Its situation is reverse for smaller perturbations and does improve in this setting, which suggests that most methods do not have a sufficient scope to cover all adversarial attacks.</p>
</sec>
<sec>
<title>4.3.2 Against adaptive attacks</title>
<p>We test MeetSafe against adaptive adversaries, a challenging AEs detection scenario, and also show in <xref ref-type="table" rid="T3">Table 3</xref>. We find that an adaptive attacker can break most methods. In particular, the results for KDE&#x0002B;BU, LID, EXM, and MAH showed a true positive rate close to 0% on the CIFAR-10 and STL-10 datasets, which is consistent with prior works (<xref ref-type="bibr" rid="B7">Carlini and Wagner, 2017a</xref>; <xref ref-type="bibr" rid="B4">Athalye et al., 2018</xref>).</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Accuracy of comparison detections evaluated against gray- and white-box (GB/WB) adversaries on STL10, MNIST, and CIFAR10; with a &#x02113;<sub>2</sub>-radii of 5.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left" rowspan="2"><bold>Evasion attack</bold></th>
<th valign="top" align="left" rowspan="2"><bold>Robust optim</bold>.</th>
<th valign="top" align="left" rowspan="2"><bold>Dataset</bold></th>
<th valign="top" align="center" rowspan="2"><bold>WB</bold></th>
<th valign="top" align="center" colspan="7"><bold>Detection accuracy</bold> &#x021D1;</th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="center"><bold>MS-8</bold></th>
<th valign="top" align="center"><bold>I-Def</bold></th>
<th valign="top" align="center"><bold>LRD (LFA)</bold></th>
<th valign="top" align="center"><bold>PCA</bold></th>
<th valign="top" align="center"><bold>KDE&#x0002B;BU</bold></th>
<th valign="top" align="center"><bold>FSQ</bold></th>
<th valign="top" align="center"><bold>MAH</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="left">STL-10</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">0.520</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.549</bold></td>
<td valign="top" align="center">0.511</td>
<td valign="top" align="center">0.521</td>
<td valign="top" align="center"><bold>0.542</bold></td>
<td valign="top" align="center">0.494</td>
<td valign="top" align="center"><bold>0.546</bold></td>
</tr> <tr>
<td/>
<td/>
<td valign="top" align="left">MNIST</td>
<td/>
<td valign="top" align="center"><bold>0.823</bold></td>
<td valign="top" align="center"><bold>0.814</bold></td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.829</bold></td>
<td valign="top" align="center">0.612</td>
<td valign="top" align="center">0.809</td>
<td valign="top" align="center">0.667</td>
<td valign="top" align="center">0.797</td>
</tr> <tr>
<td valign="top" align="left"><italic>VGG-13:</italic></td>
<td/>
<td valign="top" align="left">CIFAR-10</td>
<td/>
<td valign="top" align="center"><bold>0.808</bold></td>
<td valign="top" align="center">0.569</td>
<td valign="top" align="center">0.690</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.843</bold></td>
<td valign="top" align="center"><bold>0.748</bold></td>
<td valign="top" align="center">0.594</td>
<td valign="top" align="center">0.682</td>
</tr> <tr>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="left">AL</td>
<td valign="top" align="left">STL-10</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">0.485</td>
<td valign="top" align="center"><bold>0.508</bold></td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.517</bold></td>
<td valign="top" align="center"><bold>0.506</bold></td>
<td valign="top" align="center">0.498</td>
<td valign="top" align="center">0.497</td>
<td valign="top" align="center">0.505</td>
</tr> <tr>
<td/>
<td/>
<td valign="top" align="left">MNIST</td>
<td/>
<td valign="top" align="center">0.897</td>
<td valign="top" align="center">0.894</td>
<td valign="top" align="center"><bold>0.928</bold></td>
<td valign="top" align="center">0.614</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.957</bold></td>
<td valign="top" align="center">0.828</td>
<td valign="top" align="center"><bold>0.915</bold></td>
</tr> <tr>
<td valign="top" align="left"><italic>VGG-13:</italic></td>
<td/>
<td valign="top" align="left">CIFAR-10</td>
<td/>
<td valign="top" align="center" style="color:#00ffff"><bold>0.942</bold></td>
<td valign="top" align="center">0.508</td>
<td valign="top" align="center">0.489</td>
<td valign="top" align="center"><bold>0.941</bold></td>
<td valign="top" align="center">0.538</td>
<td valign="top" align="center"><bold>0.676</bold></td>
<td valign="top" align="center">0.571</td>
</tr> <tr>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="left">RCE</td>
<td valign="top" align="left">STL-10</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.737</bold></td>
<td valign="top" align="center">0.567</td>
<td valign="top" align="center">0.528</td>
<td valign="top" align="center">0.531</td>
<td valign="top" align="center">0.571</td>
<td valign="top" align="center"><bold>0.667</bold></td>
<td valign="top" align="center"><bold>0.635</bold></td>
</tr> <tr>
<td/>
<td/>
<td valign="top" align="left">MNIST</td>
<td/>
<td valign="top" align="center"><bold>0.953</bold></td>
<td valign="top" align="center">0.935</td>
<td valign="top" align="center"><bold>0.955</bold></td>
<td valign="top" align="center">0.678</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.975</bold></td>
<td valign="top" align="center">0.915</td>
<td valign="top" align="center">0.659</td>
</tr> <tr>
<td valign="top" align="left"><italic>VGG-13:</italic></td>
<td/>
<td valign="top" align="left">CIFAR-10</td>
<td/>
<td valign="top" align="center" style="color:#00ffff"><bold>0.956</bold></td>
<td valign="top" align="center"><bold>0.606</bold></td>
<td valign="top" align="center">0.570</td>
<td valign="top" align="center"><bold>0.949</bold></td>
<td valign="top" align="center">0.602</td>
<td valign="top" align="center">0.593</td>
<td valign="top" align="center">0.592</td>
</tr> <tr>
<td valign="top" align="left">DeepFool</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="left">STL-10</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center"><bold>0.540</bold></td>
<td valign="top" align="center">0.495</td>
<td valign="top" align="center">0.507</td>
<td valign="top" align="center">0.500</td>
<td valign="top" align="center"><bold>0.542</bold></td>
<td valign="top" align="center">0.481</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.556</bold></td>
</tr> <tr>
<td/>
<td/>
<td valign="top" align="left">MNIST</td>
<td/>
<td valign="top" align="center" style="color:#00ffff"><bold>0.951</bold></td>
<td valign="top" align="center">0.877</td>
<td valign="top" align="center">0.726</td>
<td valign="top" align="center">0.522</td>
<td valign="top" align="center">0.840</td>
<td valign="top" align="center"><bold>0.942</bold></td>
<td valign="top" align="center"><bold>0.938</bold></td>
</tr> <tr>
<td valign="top" align="left"><italic>VGG-13:</italic></td>
<td/>
<td valign="top" align="left">CIFAR-10</td>
<td/>
<td valign="top" align="center"><bold>0.640</bold></td>
<td valign="top" align="center">0.568</td>
<td valign="top" align="center">0.519</td>
<td valign="top" align="center">0.502</td>
<td valign="top" align="center">0.601</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.894</bold></td>
<td valign="top" align="center"><bold>0.756</bold></td>
</tr> <tr>
<td valign="top" align="left"><italic>ResNet-50:</italic></td>
<td/>
<td valign="top" align="left">CIFAR-10</td>
<td/>
<td valign="top" align="center"><bold>0.729</bold></td>
<td valign="top" align="center">0.700</td>
<td valign="top" align="center">0.546</td>
<td valign="top" align="center">0.506</td>
<td valign="top" align="center">0.630</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.900</bold></td>
<td valign="top" align="center"><bold>0.801</bold></td>
</tr> <tr>
<td valign="top" align="left">DeepFool</td>
<td valign="top" align="left">AL</td>
<td valign="top" align="left">STL-10</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.665</bold></td>
<td valign="top" align="center">0.502</td>
<td valign="top" align="center">0.509</td>
<td valign="top" align="center">0.500</td>
<td valign="top" align="center">0.503</td>
<td valign="top" align="center"><bold>0.588</bold></td>
<td valign="top" align="center"><bold>0.510</bold></td>
</tr> <tr>
<td/>
<td/>
<td valign="top" align="left">MNIST</td>
<td/>
<td valign="top" align="center" style="color:#00ffff"><bold>0.941</bold></td>
<td valign="top" align="center">0.880</td>
<td valign="top" align="center">0.744</td>
<td valign="top" align="center">0.645</td>
<td valign="top" align="center">0.829</td>
<td valign="top" align="center"><bold>0.915</bold></td>
<td valign="top" align="center"><bold>0.927</bold></td>
</tr> <tr>
<td valign="top" align="left"><italic>VGG-13:</italic></td>
<td/>
<td valign="top" align="left">CIFAR-10</td>
<td/>
<td valign="top" align="center"><bold>0.686</bold></td>
<td valign="top" align="center">0.552</td>
<td valign="top" align="center">0.501</td>
<td valign="top" align="center">0.560</td>
<td valign="top" align="center">0.579</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.857</bold></td>
<td valign="top" align="center"><bold>0.650</bold></td>
</tr> <tr>
<td valign="top" align="left">DeepFool</td>
<td valign="top" align="left">RCE</td>
<td valign="top" align="left">STL-10</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center"><bold>0.612</bold></td>
<td valign="top" align="center">0.531</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.647</bold></td>
<td valign="top" align="center">0.500</td>
<td valign="top" align="center">0.611</td>
<td valign="top" align="center"><bold>0.634</bold></td>
<td valign="top" align="center">0.524</td>
</tr> <tr>
<td/>
<td/>
<td valign="top" align="left">MNIST</td>
<td/>
<td valign="top" align="center"><bold>0.953</bold></td>
<td valign="top" align="center">0.934</td>
<td valign="top" align="center">0.925</td>
<td valign="top" align="center">0.505</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.960</bold></td>
<td valign="top" align="center"><bold>0.947</bold></td>
<td valign="top" align="center">0.945</td>
</tr> <tr>
<td valign="top" align="left"><italic>VGG-13:</italic></td>
<td/>
<td valign="top" align="left">CIFAR-10</td>
<td/>
<td valign="top" align="center">0.657</td>
<td valign="top" align="center">0.633</td>
<td valign="top" align="center">0.546</td>
<td valign="top" align="center">0.567</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.850</bold></td>
<td valign="top" align="center"><bold>0.759</bold></td>
<td valign="top" align="center"><bold>0.698</bold></td>
</tr> <tr>
<td valign="top" align="left">C&#x00026;W</td>
<td valign="top" align="left">Best perf.</td>
<td valign="top" align="left">STL-10</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">0.517</td>
<td valign="top" align="center">0.518</td>
<td valign="top" align="center">0.512</td>
<td valign="top" align="center">0.500</td>
<td valign="top" align="center"><bold>0.524</bold></td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.586</bold></td>
<td valign="top" align="center"><bold>0.522</bold></td>
</tr> <tr>
<td/>
<td/>
<td valign="top" align="left">MNIST</td>
<td/>
<td valign="top" align="center" style="color:#00ffff"><bold>0.958</bold></td>
<td valign="top" align="center">0.924</td>
<td valign="top" align="center">0.894</td>
<td valign="top" align="center">0.609</td>
<td valign="top" align="center">0.866</td>
<td valign="top" align="center"><bold>0.949</bold></td>
<td valign="top" align="center"><bold>0.947</bold></td>
</tr> <tr>
<td valign="top" align="left"><italic>VGG-13:</italic></td>
<td/>
<td valign="top" align="left">CIFAR-10</td>
<td/>
<td valign="top" align="center"><bold>0.803</bold></td>
<td valign="top" align="center">0.574</td>
<td valign="top" align="center">0.542</td>
<td valign="top" align="center">0.564</td>
<td valign="top" align="center"><bold>0.652</bold></td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.870</bold></td>
<td valign="top" align="center">0.638</td>
</tr>
<tr>
<td valign="top" align="left">C&#x00026;W</td>
<td valign="top" align="left">Best perf.</td>
<td valign="top" align="left">STL-10</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.619</bold></td>
<td valign="top" align="center">0.475</td>
<td valign="top" align="center"><bold>0.507</bold></td>
<td valign="top" align="center">0.456</td>
<td valign="top" align="center">0.498</td>
<td valign="top" align="center"><bold>0.547</bold></td>
<td valign="top" align="center">0.479</td>
</tr> <tr>
<td/>
<td/>
<td valign="top" align="left">MNIST</td>
<td/>
<td valign="top" align="center" style="color:#00ffff"><bold>0.989</bold></td>
<td valign="top" align="center">0.953</td>
<td valign="top" align="center">0.940</td>
<td valign="top" align="center">0.599</td>
<td valign="top" align="center">0.912</td>
<td valign="top" align="center"><bold>0.986</bold></td>
<td valign="top" align="center"><bold>0.988</bold></td>
</tr> <tr>
<td valign="top" align="left"><italic>VGG-13:</italic></td>
<td/>
<td valign="top" align="left">CIFAR-10</td>
<td/>
<td valign="top" align="center" style="color:#00ffff"><bold>0.896</bold></td>
<td valign="top" align="center">0.492</td>
<td valign="top" align="center">0.536</td>
<td valign="top" align="center"><bold>0.581</bold></td>
<td valign="top" align="center">0.521</td>
<td valign="top" align="center"><bold>0.606</bold></td>
<td valign="top" align="center">0.485</td>
</tr> <tr>
<td valign="top" align="left">PerC-AL</td>
<td valign="top" align="left">RCE</td>
<td valign="top" align="left">STL-10</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center"><bold>0.523</bold></td>
<td valign="top" align="center">0.503</td>
<td valign="top" align="center">0.503</td>
<td valign="top" align="center"><bold>0.526</bold></td>
<td valign="top" align="center">0.516</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.533</bold></td>
<td valign="top" align="center">0.505</td>
</tr> <tr>
<td/>
<td/>
<td valign="top" align="left">MNIST</td>
<td/>
<td valign="top" align="center">N/A</td>
<td valign="top" align="center">N/A</td>
<td valign="top" align="center">N/A</td>
<td valign="top" align="center">N/A</td>
<td valign="top" align="center">N/A</td>
<td valign="top" align="center">N/A</td>
<td valign="top" align="center">N/A</td>
</tr> <tr>
<td valign="top" align="left"><italic>VGG-13:</italic></td>
<td/>
<td valign="top" align="left">CIFAR-10</td>
<td/>
<td valign="top" align="center" style="color:#00ffff"><bold>0.815</bold></td>
<td valign="top" align="center">0.745</td>
<td valign="top" align="center">0.771</td>
<td valign="top" align="center"><bold>0.795</bold></td>
<td valign="top" align="center"><bold>0.782</bold></td>
<td valign="top" align="center">0.514</td>
<td valign="top" align="center">0.766</td>
</tr> <tr>
<td valign="top" align="left"><italic>ResNet-50:</italic></td>
<td/>
<td valign="top" align="left">CIFAR-10</td>
<td/>
<td valign="top" align="center">0.553</td>
<td valign="top" align="center">0.529</td>
<td valign="top" align="center"><bold>0.560</bold></td>
<td valign="top" align="center"><bold>0.612</bold></td>
<td valign="top" align="center">0.525</td>
<td valign="top" align="center">0.548</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.667</bold></td>
</tr> <tr>
<td valign="top" align="left">Shadow attack</td>
<td valign="top" align="left">RCE</td>
<td valign="top" align="left">STL-10</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">0.587</td>
<td valign="top" align="center">0.479</td>
<td valign="top" align="center"><bold>0.624</bold></td>
<td valign="top" align="center">0.450</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.683</bold></td>
<td valign="top" align="center"><bold>0.597</bold></td>
<td valign="top" align="center">0.464</td>
</tr> <tr>
<td/>
<td/>
<td valign="top" align="left">MNIST</td>
<td/>
<td valign="top" align="center"><bold>0.927</bold></td>
<td valign="top" align="center"><bold>0.937</bold></td>
<td valign="top" align="center">0.887</td>
<td valign="top" align="center">0.453</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.963</bold></td>
<td valign="top" align="center">0.449</td>
<td valign="top" align="center">0.833</td>
</tr> <tr>
<td valign="top" align="left"><italic>VGG-13:</italic></td>
<td/>
<td valign="top" align="left">CIFAR-10</td>
<td/>
<td valign="top" align="center"><bold>0.665</bold></td>
<td valign="top" align="center">0.477</td>
<td valign="top" align="center">0.511</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.698</bold></td>
<td valign="top" align="center">0.529</td>
<td valign="top" align="center">0.504</td>
<td valign="top" align="center"><bold>0.663</bold></td>
</tr> <tr>
<td valign="top" align="left"><italic>ResNet-50:</italic></td>
<td/>
<td valign="top" align="left">CIFAR-10</td>
<td/>
<td valign="top" align="center"><bold>0.653</bold></td>
<td valign="top" align="center">0.561</td>
<td valign="top" align="center">0.596</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.716</bold></td>
<td valign="top" align="center">0.485</td>
<td valign="top" align="center">0.513</td>
<td valign="top" align="center"><bold>0.619</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Note that PerC-AL requires color datasets so we do not include results for MNIST, a dataset contains gray scale images. The top-3 results are bolded. Top-1 results are highlighted in blue.</p>
</table-wrap-foot>
</table-wrap>
<p>Regarding the other methods, we see that especially PCA excels with accuracies of approximately 80% for CIFAR-10. Surprisingly, PCA was proven as not robust earlier (<xref ref-type="bibr" rid="B7">Carlini and Wagner, 2017a</xref>). LRD is, in addition to PCA, also somewhat resilient against adaptive attacks, although it should be noted that BRISQUE does utilize local non-linear operations to estimate generalized gamma functions (<xref ref-type="bibr" rid="B29">Mittal et al., 2012</xref>), which are not smooth functions. We can therefore only consider LRD robust on the hidden features. Some results are worse than in the gray-box setting, that is possible due to the positive confidence value. Hence, the adversarial example is stimulated to be 5% below the detector&#x00027;s threshold, which improves its transferability on models with feature bags or other uncertainties.</p>
<p>We show the performance of LRD, PCA, and FSQ and their MeetSafe ensemble across datasets in <xref ref-type="table" rid="T2">Table 2</xref>. The p-values of the methods&#x00027; confidence are computed and compared against a threshold. Specifically, we evaluate the p-values for 10 random FGSM samples using a reversed ResNet-50 model trained on a benign dataset. A low p-value is beneficial for the GMM&#x00027;s generalization as this assumes normality. On MNIST, an opposing utility between LRD and whitening techniques becomes clear. Here, LRD has a p-value of near zero (&#x02264;10<sup>&#x02212;99</sup>), while whitening has a value of 6e-6. On the other hand, whitening performs relatively better on CIFAR-10 with a p-value smaller than 1e-80 against 5e-17 for LRD. Whitening and LRD might therefore offset each other&#x00027;s effects against FGSM.</p>
<p>The added value of feature squeezing is apparent for small CIFAR-10 perturbations. <xref ref-type="fig" rid="F2">Figure 2a</xref> shows the performance of the three detectors for DeepFool and FGSM. It shows a noticeably higher AUROC for feature squeezing on DeepFool. Furthermore, feature squeezing has the smallest <italic>p</italic>-value of 0.01, followed by LRD with 0.25. Feature squeezing could thus be helpful in the case when the adversarial sample is near the benign input. Section 5.2 discusses the importance of each components of MeetSafe.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p><bold>(a)</bold> ROC curves of MeetSafe&#x00027;s features on CIFAR-10 and a reverse ResNet-50 as target model. The curves show the performance against DeepFool and FGSM. <bold>(b)</bold> Graph that shows the inference times of MS-8 and a reversed ResNet-50 on MNIST, CIFAR-10, Tiny-ImageNet, and STL-10. The error bar denotes the t-CI across 5 runs.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1631561-g0002.tif">
<alt-text>Graph (a) shows a ROC curve comparing different methods (LRD, Feature Squeezing, Whitening, Combined) using FGSM and DeepFool. The true positive rate rises with false positive rate. Graph (b) compares batch processing speed between Resnet-50 and MeetSafe, showing Resnet-50 is faster across varying input sizes.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<title>4.3.3 GMM-based detection</title>
<p>Our MeetSafe constructs a GMM-based detection with PCA, feature squeezing, and LRD. We test MeetSafe&#x00027;s performance under certain number of Gaussian components: 4, 8, and 16 (<xref ref-type="table" rid="T4">Table 4</xref>). For gray-box perturbations, we see that only 4 components may be useful for FGSM, but this increases for small perturbations. To balance these accuracies, we think that 8 components are desirable. Comparing the performance of MS-8 to that of I-Defender shows similar results on FGSM, but lower accuracies on stronger attacks. Meanwhile, the accuracy for RCE and MeetSafe does not scale well. This is most notable when we compare the efficacy for datasets of higher resolution. Tiny-ImageNet shows accuracies on C&#x00026;W of at most 0.553 for MeetSafe and 0.511 for I-Defender. STL-10 shows similar results (<xref ref-type="table" rid="T4">Table 4</xref>). However, the effect of dimensionality is not a limitation specific to our method but rather a general issue of defenses against AEs (<xref ref-type="bibr" rid="B13">Goodfellow et al., 2015</xref>). MS-8 achieves an improvement of at least 8.1% on adaptive attacks and 10.2% on the worst-case results for each evaluated method by averaging across STL-10, MNIST, and CIFAR-10. MeetSafe may therefore be employed universally while maintaining a considerable detection accuracy.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Accuracies of GMM-based detectors against gray- and white-box (GB/WB) adversaries with a &#x02113;<sub>2</sub>-radii of 5 like in <xref ref-type="table" rid="T2">Table 2</xref>.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left" rowspan="3"><bold>Evasion attack</bold></th>
<th valign="top" align="left" rowspan="3"><bold>Robust optim</bold>.</th>
<th valign="top" align="center" rowspan="3"><bold>Dataset model</bold><break/>&#x02192;<break/><bold>WB</bold> &#x02193;</th>
<th valign="top" align="center" colspan="10"><bold>Detection accuracy</bold> &#x000B1;<bold>[t-CI 95%]</bold> &#x021D1;</th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="center" colspan="4"><bold>CIFAR-10</bold></th>
<th valign="top" align="center" colspan="2"><bold>STL-10</bold></th>
<th valign="top" align="center" colspan="2"><bold>MNIST</bold></th>
<th valign="top" align="center" colspan="2"><bold>VGG-13</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="center"><bold>MS-4</bold></th>
<th valign="top" align="center"><bold>MS-8</bold></th>
<th valign="top" align="center"><bold>MS-16</bold></th>
<th valign="top" align="center"><bold>I-Def</bold></th>
<th valign="top" align="center"><bold>MS-8</bold></th>
<th valign="top" align="center"><bold>I-Def</bold></th>
<th valign="top" align="center"><bold>MS-8</bold></th>
<th valign="top" align="center"><bold>I-Def</bold></th>
<th valign="top" align="center"><bold>MS-8</bold></th>
<th valign="top" align="center"><bold>I-Def</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center"><underline>0.680</underline></td>
<td valign="top" align="center"><underline>0.671</underline></td>
<td valign="top" align="center"><underline>0.672</underline></td>
<td valign="top" align="center">0.627</td>
<td valign="top" align="center">0.520</td>
<td valign="top" align="center">0.549</td>
<td valign="top" align="center" style="color:#00ffff"><underline><bold>0.823</bold></underline></td>
<td valign="top" align="center"><bold>0.814</bold></td>
<td valign="top" align="center"><bold>0.808</bold></td>
<td valign="top" align="center">0.569</td>
</tr> <tr>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="left">AL</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">0.876</td>
<td valign="top" align="center">0.829</td>
<td valign="top" align="center">0.853</td>
<td valign="top" align="center"><bold>0.925</bold></td>
<td valign="top" align="center"><underline>0.485</underline></td>
<td valign="top" align="center">0.508</td>
<td valign="top" align="center"><bold>0.897</bold></td>
<td valign="top" align="center">0.894</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.942</bold></td>
<td valign="top" align="center">0.508</td>
</tr> <tr>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="left">RCE</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.965</bold></td>
<td valign="top" align="center">0.926</td>
<td valign="top" align="center">0.895</td>
<td valign="top" align="center">0.717</td>
<td valign="top" align="center">0.737</td>
<td valign="top" align="center">0.567</td>
<td valign="top" align="center"><bold>0.953</bold></td>
<td valign="top" align="center">0.935</td>
<td valign="top" align="center"><bold>0.956</bold></td>
<td valign="top" align="center">0.606</td>
</tr>
<tr>
<td valign="top" align="left">DeepFool</td>
<td valign="top" align="left">AL</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">0.691</td>
<td valign="top" align="center">0.752</td>
<td valign="top" align="center"><bold>0.762</bold></td>
<td valign="top" align="center">0.520</td>
<td valign="top" align="center">0.665</td>
<td valign="top" align="center">0.502</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.941</bold></td>
<td valign="top" align="center"><bold>0.880</bold></td>
<td valign="top" align="center">0.686</td>
<td valign="top" align="center">0.552</td>
</tr> <tr>
<td valign="top" align="left">DeepFool</td>
<td valign="top" align="left">RCE</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">0.738</td>
<td valign="top" align="center">0.785</td>
<td valign="top" align="center"><bold>0.745</bold></td>
<td valign="top" align="center">0.571</td>
<td valign="top" align="center">0.612</td>
<td valign="top" align="center">0.531</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.953</bold></td>
<td valign="top" align="center"><bold>0.934</bold></td>
<td valign="top" align="center"><underline>0.657</underline></td>
<td valign="top" align="center">0.633</td>
</tr> <tr>
<td valign="top" align="left">C&#x00026;W</td>
<td valign="top" align="left">AL</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">0.746</td>
<td valign="top" align="center"><bold>0.804</bold></td>
<td valign="top" align="center"><bold>0.815</bold></td>
<td valign="top" align="center">0.518</td>
<td valign="top" align="center">0.517</td>
<td valign="top" align="center">0.498</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.958</bold></td>
<td valign="top" align="center"><underline>0.696</underline></td>
<td valign="top" align="center">0.803</td>
<td valign="top" align="center">0.509</td>
</tr> <tr>
<td valign="top" align="left">C&#x00026;W</td>
<td valign="top" align="left">RCE</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">0.810</td>
<td valign="top" align="center">0.818</td>
<td valign="top" align="center"><bold>0.814</bold></td>
<td valign="top" align="center">0.557</td>
<td valign="top" align="center">0.574</td>
<td valign="top" align="center">0.518</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.958</bold></td>
<td valign="top" align="center"><bold>0.924</bold></td>
<td valign="top" align="center">0.689</td>
<td valign="top" align="center">0.574</td>
</tr> <tr>
<td valign="top" align="left">C&#x00026;W</td>
<td valign="top" align="left">RCE</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">0.762</td>
<td valign="top" align="center">0.745 &#x000B1; 0.04</td>
<td valign="top" align="center">0.689</td>
<td valign="top" align="center"><underline>0.469</underline></td>
<td valign="top" align="center">0.544</td>
<td valign="top" align="center"><underline>0.475</underline></td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.989</bold></td>
<td valign="top" align="center"><bold>0.953</bold></td>
<td valign="top" align="center"><bold>0.896</bold></td>
<td valign="top" align="center"><underline>0.492</underline></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>We show experimental results on additional datasets and model architectures in <xref ref-type="table" rid="T5">Table 5</xref>. The t-CI is based on 5 runs. Top-3 results are bolded, and the worst-case of a detection algorithm is underlined. Top-1 results are highlighted in blue.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Accuracies of GMM-based detectors against gray-box adversaries on Tiny-ImageNet, STL-10, and CIFAR-10, with a &#x02113;<sub>2</sub>-radii of 5.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left" rowspan="3"><bold>Evasion attack</bold></th>
<th valign="top" align="left" rowspan="3"><bold>Model</bold></th>
<th valign="top" align="left" rowspan="3"><bold>Dataset</bold></th>
<th valign="top" align="center" rowspan="3"><break/><bold>Robust optim</bold>. &#x02192;<break/><bold>WB</bold> &#x02193;</th>
<th valign="top" align="center" colspan="6"><bold>Detection accuracy</bold> &#x021D1;</th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="center" colspan="2"><bold>Plain model</bold></th>
<th valign="top" align="center" colspan="2"><bold>AL model</bold></th>
<th valign="top" align="center" colspan="2"><bold>RCE model</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="center"><bold>MS-8</bold></th>
<th valign="top" align="center"><bold>I-Def</bold></th>
<th valign="top" align="center"><bold>MS-8</bold></th>
<th valign="top" align="center"><bold>I-Def</bold></th>
<th valign="top" align="center"><bold>MS-8</bold></th>
<th valign="top" align="center"><bold>I-Def</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">FGSM</td>
<td valign="top" align="left">ResNet-50</td>
<td valign="top" align="left">Tiny-ImageNet</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">0.954</td>
<td valign="top" align="center">0.532</td>
<td valign="top" align="center">0.939</td>
<td valign="top" align="center">0.533</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.976</bold></td>
<td valign="top" align="center">0.503</td>
</tr> <tr>
<td valign="top" align="left">C&#x00026;W</td>
<td valign="top" align="left">VGG-13</td>
<td valign="top" align="left">CIFAR-10</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">0.790</td>
<td valign="top" align="center">0.593</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.803</bold></td>
<td valign="top" align="center">0.509</td>
<td valign="top" align="center">0.689</td>
<td valign="top" align="center">0.574</td>
</tr> <tr>
<td valign="top" align="left">C&#x00026;W</td>
<td valign="top" align="left">ResNet-50</td>
<td valign="top" align="left">STL-10</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">0.531</td>
<td valign="top" align="center">0.499</td>
<td valign="top" align="center">0.517</td>
<td valign="top" align="center">0.498</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.574</bold></td>
<td valign="top" align="center">0.518</td>
</tr> <tr>
<td valign="top" align="left">C&#x00026;W</td>
<td valign="top" align="left">ResNet-50</td>
<td valign="top" align="left">Tiny-ImageNet</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center" style="color:#00ffff"><bold>0.553</bold></td>
<td valign="top" align="center">0.494</td>
<td valign="top" align="center">0.523</td>
<td valign="top" align="center">0.497</td>
<td valign="top" align="center">0.502</td>
<td valign="top" align="center">0.511</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Demonstrating the efficacy of small-scale models in comparison with datasets containing images of higher resolution. DeepFool&#x00027;s and C&#x00026;W&#x00027;s results are dependent on the error rate <inline-formula><mml:math id="M73"><mml:mrow><mml:mi mathvariant="script">R</mml:mi></mml:mrow></mml:math></inline-formula><sub><italic>f</italic></sub> of the models, like in <xref ref-type="table" rid="T2">Table 2</xref>. Best results are bolded. Top-1 results are highlighted in blue.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>4.3.4 Inference time</title>
<p><xref ref-type="fig" rid="F2">Figure 2b</xref> shows the inference time of MeetSafe and a ResNet-50 target model on four datasets: MNIST, CIFAR-10, Tiny-ImageNet, and STL-10. Plotted according to the input size of one sample: 784, 3, 072, 12, 288, and 27, 648, respectively. From the figure, we can observe that the discrepancy of ResNet-50 and MeetSafe converges to a factor of approximately 2.3 when the input size gets larger. We increased the feature size from 10 to 17 and the pool size from 500 to 850. We anticipated that this would lead to increased computation, especially for LRD. The results on CIFAR-10 demonstrate that this increase led to a slower processing rate, with the model running 0.28 batches per second slower than before. We consider this change to be limited, indicating that the computational overhead is also manageable given the increased feature and pool sizes.</p>
</sec>
</sec>
</sec>
<sec id="s5">
<title>5 Ablation study</title>
<sec>
<title>5.1 Sensitivity and utility of the hidden layers</title>
<p><xref ref-type="fig" rid="F3">Figure 3</xref> explores the <inline-formula><mml:math id="M74"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:math></inline-formula> utility in the study. The barplots show the utility at the outputs of that layer, which tends to increase when the perturbed sample traverses deeper layers, but this may come with fluctuations. For instance, the values for the STL-10 models seem to decrease in the last layers. The adversarial learned CIFAR-10 model also decreases in utility after the first bottleneck. Still, the final bottleneck remains in the Top-3, suggesting its pivotal role in model performance and feature extraction. Conversely, the initial convolution, denoted as &#x0201C;conv1&#x0201D;, frequently exhibits the least utility.</p>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>Average of normalized Euclidean distance between FGSM AEs (&#x02113;<sub>2</sub> of 5) and benign samples under <inline-formula><mml:math id="M75"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:math></inline-formula> at each ResNet-50 layer. The error bars denote the 95% t-CI across 5 runs. <bold>(a)</bold> CIFAR-10 dataset, <bold>(b)</bold> MNIST dataset, and <bold>(c)</bold> STL-10 dataset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1631561-g0003.tif">
<alt-text>Bar charts comparing &#x0201D;RCE Model&#x0201D; and &#x0201D;AL Model&#x0201D; for three datasets: CIFAR-10, MNIST, and STL-10. Each chart shows layer-wise utility values with RCE Model in blue and AL Model in orange. The CIFAR-10 chart highlights layers from conv1 to avg_pool, the MNIST chart from conv1 to avg_pool, and the STL-10 chart from conv1 to avg_pool. Each dataset chart presents varying utility values, indicating different model performances across layers.</alt-text>
</graphic>
</fig>
<p>We also see a major difference in the CI&#x00027;s critical region of the CIFAR-10 models. Specifically, the adversarial learned model displays notable variability on each layer, especially when compared to the RCE model. That is in line with the narrow activations in the scatterplots of <xref ref-type="fig" rid="F4">Figure 4</xref>. The RCE models exhibit greater utilities in its deeper layers. Consequently, we anticipate enhanced performance when these hidden units are utilized for detection purposes. Otherwise, adversarial learning would become more intriguing due to the improved model accuracy (see <xref ref-type="table" rid="T1">Table 1</xref>).</p>
<fig position="float" id="F4">
<label>Figure 4</label>
<caption><p>Activations&#x00027; Z-Score of three hidden features before and after an FGSM perturbation, with a &#x02113;<sub>2</sub>-radii of 5. The activations are sampled from a random pool of 500 hidden features, directly after convolution. We show the activation of hidden features with the highest and lowest Z-score in <bold>(a, b)</bold> for an adversarially trained ResNet50 on CIFAR10 and non-adversarially trained results in <bold>(c, d)</bold>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1631561-g0004.tif">
<alt-text>Four 3D scatter plots (a, b, c, d) display data points with different axes labeled as various features. Points in blue represent the original image data, while red points represent adversarial image data. Each subplot highlights differences in feature distributions, with varying overlaps and separations between the original and adversarial data in three-dimensional space.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<title>5.2 On the impact of MeetSafe components</title>
<p>We check the detection performance by removing each component of MeetSafe (MS-8) to establish its importance within the ensemble. For the ablation, we trained 3 GMMs all with two components (LRD, PCA, FSQ) on a reversed ResNet-50 and CIFAR-10. First, when we remove LRD the accuracy decreases by 0.054 for C&#x00026;W, &#x02212;0.027 for FGSM, and 0.117 for DeepFool. For FGSM perturbations, the results do show that LRD does not add much for on CIFAR-10, as its ablation leads to equivalent accuracies of the GMM. Overall, LRD increases the effectiveness of MeetSafe across various scenarios, aligning with the findings presented in Section 4.3. Second, when we remove PCA, the accuracy decreases by 0.191 for C&#x00026;W, 0.234 for FGSM, and 0.145 for DeepFool. Whitening is thus an important component on CIFAR-10. Third, when we remove FSQ, the accuracy decreases by 0.197 for C&#x00026;W, &#x02212;0.048 for FGSM, and 0.137 for DeepFool, which is also in line with the results in Section 4.3.</p>
</sec>
<sec>
<title>5.3 On the impact of <italic>k</italic> for kNNs</title>
<p>To determine the optimal <italic>k</italic>, we evaluated the accuracy of LOF and LRD using features from a random pool of 500 hidden features, as shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. The left axis represents LRD accuracy, while the right axis denotes LOF accuracy. The results are plotted separately to highlight their distinct trends, each spanning an accuracy range of 0.06. The elbow points indicate an optimal <italic>k</italic> &#x0003D; 8 for both methods, which we adopt for all <italic>k</italic>-NN-based approaches. Notably, LRD achieves significantly higher accuracy than LOF, and <italic>k</italic> impacts LOF more than LRD.</p>
<fig position="float" id="F5">
<label>Figure 5</label>
<caption><p>Performance of LOF and LRD under different <italic>k</italic> with the amount of feature bags <italic>t</italic>=30 under CIFAR-10 and adversarial learned ResNet-50.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1631561-g0005.tif">
<alt-text>Line chart titled &#x0201D;Performance of LRD and LOF&#x0201D; showing accuracy against k values from 4 to 16. The LOF curve in orange rises from 0.91 to 0.94 as k increases, while LRD in pink shows slight growth from 0.88 to 0.90. Accuracy values for LOF range from 0.58 to 0.63 on the right axis.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="s6">
<title>6 Conclusion and limitation</title>
<p>This study present MeetSafe, a scalable and effective framework to detect white-box AEs. By leveraging insights from feature distribution irregularities, MeetSafe integrates utility-based feature selection with feature squeezing, whitening, and feature squeezing to achieve high defense effectiveness and scalability against model size with the increase of high-dimensional feature spaces. Experimental results demonstrate an high detection accuracy of MeetSafe across adaptive and classic adversarial attacks, as well as robust whitening under white-box scenarios.</p>
<sec>
<title>6.1 Limitations</title>
<p>Due to resource constraints, we considered I-Defender and (<xref ref-type="bibr" rid="B11">Feinman et al. 2017</xref>) KDE not practical in certain situations. For models like ResNet-50, it would cost at least 372.53 <italic>GiB</italic> to evaluate I-Defender. On the other hand, KDE&#x0002B;BU required a large computational graph in Pytorch during white box testing. That was because of the tens of forwards for dropout. For the same reason, we could only utilize <inline-formula><mml:math id="M76"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:math></inline-formula> for the extremal value. Other methods (LRD, LID, and KDE&#x0002B;BU) require instance-based learning and more memory.</p>
</sec>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://docs.pytorch.org/vision/stable/index.html">https://docs.pytorch.org/vision/stable/index.html</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>RS: Conceptualization, Data curation, Investigation, Methodology, Software, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. DL: Conceptualization, Investigation, Methodology, Supervision, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. YQ: Methodology, Validation, Writing &#x02013; review &#x00026; editing. MC: Validation, Writing &#x02013; review &#x00026; editing. MP: Validation, Writing &#x02013; review &#x00026; editing. KL: Funding acquisition, Project administration, Resources, Validation, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported by the EU Horizon Europe Research and Innovation Program under grant agreements 101073920 (TENSOR), 101070052 (TANGO), 101070627 (REWIRE) and 101092912 (MLSysOps).</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that Gen AI was used in the creation of this manuscript. Gen AI was used for: (1) Generate LaTeX code for tables and figures to ensure a good layout. Note that all figures are drawn by the author(s) and all data is obtained by the author(s) by running experiments in human. (2) Refine the language. (3) Address LaTeX compilation issues when errors are encountered in Overleaf.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Akhtar</surname> <given-names>Z.</given-names></name> <name><surname>Monteiro</surname> <given-names>J.</given-names></name> <name><surname>Falk</surname> <given-names>T. H.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Adversarial examples detection using no-reference image quality features,&#x0201D;</article-title> in <source>International Carnahan Conference on Security Technology</source> (<publisher-loc>Montreal, QC</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>5</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Aldahdooh</surname> <given-names>A.</given-names></name> <name><surname>Hamidouche</surname> <given-names>W.</given-names></name> <name><surname>Fezza</surname> <given-names>S. A.</given-names></name> <name><surname>D&#x000E9;forges</surname> <given-names>O.</given-names></name></person-group> (<year>2022</year>). <article-title>Adversarial example detection for dnn models: a review and experimental comparison</article-title>. <source>Artif. Intellig. Rev</source>. <volume>55</volume>, <fpage>4403</fpage>&#x02013;<lpage>4462</lpage>. <pub-id pub-id-type="doi">10.1007/s10462-021-10125-w</pub-id></citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Athalye</surname> <given-names>A.</given-names></name> <name><surname>Carlini</surname> <given-names>N.</given-names></name></person-group> (<year>2018</year>). <article-title>On the robustness of the cvpr 2018 white-box adversarial example defenses</article-title>. <source>arXiv</source> preprint arXiv:1804.03286. <pub-id pub-id-type="doi">10.48550/arXiv.1804.03286</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Athalye</surname> <given-names>A.</given-names></name> <name><surname>Carlini</surname> <given-names>N.</given-names></name> <name><surname>Wagner</surname> <given-names>D. A.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Obfuscated gradients give a false sense of security: Circumventing defenses to adversarial examples,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>Stockholm</publisher-loc>: <publisher-name>ICML.cc</publisher-name>), <fpage>274</fpage>&#x02013;<lpage>283</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Mesnil</surname> <given-names>G.</given-names></name> <name><surname>Dauphin</surname> <given-names>Y.</given-names></name> <name><surname>Rifai</surname> <given-names>S.</given-names></name></person-group> (<year>2013</year>). <article-title>&#x0201C;Better mixing via deep representations,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>Atlanta</publisher-loc>: <publisher-name>ICML.cc</publisher-name>), <fpage>552</fpage>&#x02013;<lpage>560</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Breunig</surname> <given-names>M. M.</given-names></name> <name><surname>Kriegel</surname> <given-names>H.</given-names></name> <name><surname>Ng</surname> <given-names>R. T.</given-names></name> <name><surname>Sander</surname> <given-names>J.</given-names></name></person-group> (<year>2000</year>). <article-title>&#x0201C;LOF: identifying density-based local outliers,&#x0201D;</article-title> in <source>ACM International Conference on Management of Data</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>ACM</publisher-name>), <fpage>93</fpage>&#x02013;<lpage>104</lpage>.</citation>
</ref>
<ref id="B7">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Carlini</surname> <given-names>N.</given-names></name> <name><surname>Wagner</surname> <given-names>D. A.</given-names></name></person-group> (<year>2017a</year>). <article-title>&#x0201C;Adversarial examples are not easily detected: bypassing ten detection methods,&#x0201D;</article-title> in <source>ACM Workshop on Artificial Intelligence and Security</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>ACM</publisher-name>) 3&#x02013;14. <pub-id pub-id-type="doi">10.1145/3128572.3140444</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Carlini</surname> <given-names>N.</given-names></name> <name><surname>Wagner</surname> <given-names>D. A.</given-names></name></person-group> (<year>2017b</year>). <article-title>&#x0201C;Towards evaluating the robustness of neural networks,&#x0201D;</article-title> in <source>IEEE Symposium on Security and Privacy</source> (<publisher-loc>SAN JOSE, CA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>39</fpage>&#x02013;<lpage>57</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Coates</surname> <given-names>A.</given-names></name> <name><surname>Ng</surname> <given-names>A. Y.</given-names></name> <name><surname>Lee</surname> <given-names>H.</given-names></name></person-group> (<year>2011</year>). <article-title>&#x0201C;An analysis of single-layer networks in unsupervised feature learning,&#x0201D;</article-title> in <source>International Conference on Artificial Intelligence and Statistics</source> (<publisher-loc>JMLR</publisher-loc>), <fpage>215</fpage>&#x02013;<lpage>223</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Duan</surname> <given-names>R.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Niu</surname> <given-names>D.</given-names></name> <name><surname>Yang</surname> <given-names>Y.</given-names></name> <name><surname>Qin</surname> <given-names>A. K.</given-names></name> <name><surname>He</surname> <given-names>Y.</given-names></name></person-group> (<year>2021</year>). <article-title>Advdrop: Adversarial attack to dnns by dropping information</article-title>. <source>In IEEE/CVF International Conference on Computer Vision</source> (<publisher-loc>Montreal, BC</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>7506</fpage>&#x02013;<lpage>7515</lpage></citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Feinman</surname> <given-names>R.</given-names></name> <name><surname>Curtin</surname> <given-names>R. R.</given-names></name> <name><surname>Shintre</surname> <given-names>S.</given-names></name> <name><surname>Gardner</surname> <given-names>A. B.</given-names></name></person-group> (<year>2017</year>). <article-title>Detecting adversarial samples from artifacts</article-title>. <source>arXiv</source> preprint arXiv:1703.00410. <pub-id pub-id-type="doi">10.48550/arXiv.1703.00410</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ghiasi</surname> <given-names>A.</given-names></name> <name><surname>Shafahi</surname> <given-names>A.</given-names></name> <name><surname>Goldstein</surname> <given-names>T.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Breaking certified defenses: Semantic adversarial examples with spoofed robustness certificates,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source> (<publisher-loc>ICLR.cc</publisher-loc>).</citation>
</ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Goodfellow</surname> <given-names>I. J.</given-names></name> <name><surname>Shlens</surname> <given-names>J.</given-names></name> <name><surname>Szegedy</surname> <given-names>C.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Explaining and harnessing adversarial examples,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source> (<publisher-loc>San Diego, CA</publisher-loc>: <publisher-name>ICLR.cc</publisher-name>).</citation>
</ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Deep residual learning for image recognition,&#x0201D;</article-title> in <source>IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>770</fpage>&#x02013;<lpage>778</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hendrycks</surname> <given-names>D.</given-names></name> <name><surname>Gimpel</surname> <given-names>K.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Early methods for detecting adversarial images,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source> (<publisher-loc>Toulon</publisher-loc>: <publisher-name>ICLR.cc</publisher-name>).</citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>S.</given-names></name> <name><surname>Yu</surname> <given-names>T.</given-names></name> <name><surname>Guo</surname> <given-names>C.</given-names></name> <name><surname>Chao</surname> <given-names>W.-L.</given-names></name> <name><surname>Weinberger</surname> <given-names>K. Q.</given-names></name></person-group> (<year>2019</year>). <article-title>A new defense against adversarial images: Turning a weakness into a strength</article-title>. <source>arXiv</source> [preprint] arXiv:1910.07629. <pub-id pub-id-type="doi">10.48550/arXiv.1910.07629</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kherchouche</surname> <given-names>A.</given-names></name> <name><surname>Fezza</surname> <given-names>S. A.</given-names></name> <name><surname>Hamidouche</surname> <given-names>W.</given-names></name> <name><surname>D&#x000E9;forges</surname> <given-names>O.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Detection of adversarial examples in deep neural networks with natural scene statistics,&#x0201D;</article-title> in <source>International Joint Conference on Neural Networks</source> (<publisher-loc>Glasgow</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>7</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="thesis"><person-group person-group-type="author"><name><surname>Krizhevsky</surname> <given-names>A.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2009</year>). <source>Learning Multiple Layers of Features from Tiny Images</source> (Master&#x00027;s thesis). <publisher-name>University of Toronto</publisher-name>, <publisher-loc>Toronto, ON, Canada</publisher-loc>.</citation>
</ref>
<ref id="B19">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lazarevic</surname> <given-names>A.</given-names></name> <name><surname>Kumar</surname> <given-names>V.</given-names></name></person-group> (<year>2005</year>). <article-title>&#x0201C;Feature bagging for outlier detection,&#x0201D;</article-title> in <source>ACM SIGKDD International Conference on Knowledge Discovery in Data Mining</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>ACM</publisher-name>), <fpage>157</fpage>&#x02013;<lpage>166</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Le</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>X.</given-names></name></person-group> (<year>2015</year>). <source>Tiny Imagenet Visual Recognition Challenge</source>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>LeCun</surname> <given-names>Y.</given-names></name></person-group> (<year>1998</year>). <source>The MNIST Database of Handwritten Digits</source>.</citation>
</ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Lee</surname> <given-names>H.</given-names></name> <name><surname>Shin</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;A simple unified framework for detecting out-of-distribution samples and adversarial attacks,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source> (<publisher-loc>Montreal, QC</publisher-loc>: <publisher-name>neurips.cc</publisher-name>), <fpage>31</fpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>F.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Adversarial examples detection in deep networks with convolutional filter statistics,&#x0201D;</article-title> in <source>International Conference on Computer Vision</source> (<publisher-loc>Venice</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>5775</fpage>&#x02013;<lpage>5783</lpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Bradshaw</surname> <given-names>J.</given-names></name> <name><surname>Sharma</surname> <given-names>Y.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Are generative classifiers more robust to adversarial attacks?,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>Long Beach, CA</publisher-loc>: <publisher-name>icml.cc</publisher-name>), <fpage>3804</fpage>&#x02013;<lpage>3814</lpage>.</citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liang</surname> <given-names>B.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Su</surname> <given-names>M.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Shi</surname> <given-names>W.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name></person-group> (<year>2018</year>). <article-title>Detecting adversarial image examples in deep neural networks with adaptive noise reduction</article-title>. <source>IEEE Trans. Depend. Secure Comp</source>. <volume>18</volume>, <fpage>72</fpage>&#x02013;<lpage>85</lpage>. <pub-id pub-id-type="doi">10.1109/TDSC.2018.2874243</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Litjens</surname> <given-names>G.</given-names></name> <name><surname>Kooi</surname> <given-names>T.</given-names></name> <name><surname>Bejnordi</surname> <given-names>B. E.</given-names></name> <name><surname>Setio</surname> <given-names>A. A. A.</given-names></name> <name><surname>Ciompi</surname> <given-names>F.</given-names></name> <name><surname>Ghafoorian</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>A survey on deep learning in medical image analysis</article-title>. <source>Med. Image Analy</source>.<volume>42</volume>:<fpage>60</fpage>&#x02013;<lpage>88</lpage>. <pub-id pub-id-type="doi">10.1016/j.media.2017.07.005</pub-id><pub-id pub-id-type="pmid">28778026</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>C.</given-names></name> <name><surname>Lin</surname> <given-names>Q.</given-names></name> <name><surname>Xie</surname> <given-names>W.</given-names></name> <name><surname>Wu</surname> <given-names>B.</given-names></name> <name><surname>Xie</surname> <given-names>J.</given-names></name> <name><surname>Shen</surname> <given-names>L.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Frequency-driven imperceptible adversarial attack on semantic similarity,&#x0201D;</article-title> in <source>IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>New Orleans, LA</publisher-loc>: <publisher-name>IEEE</publisher-name>) <fpage>15315</fpage>&#x02013;<lpage>15324</lpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>B.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Erfani</surname> <given-names>S. M.</given-names></name> <name><surname>Wijewickrema</surname> <given-names>S. N. R.</given-names></name> <name><surname>Schoenebeck</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>&#x0201C;Characterizing adversarial subspaces using local intrinsic dimensionality,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source> (<publisher-loc>Vancouver, BC</publisher-loc>: <publisher-name>iclr.cc</publisher-name>).</citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mittal</surname> <given-names>A.</given-names></name> <name><surname>Moorthy</surname> <given-names>A. K.</given-names></name> <name><surname>Bovik</surname> <given-names>A. C.</given-names></name></person-group> (<year>2012</year>). <article-title>No-reference image quality assessment in the spatial domain</article-title>. <source>IEEE Trans. Image Proc</source>. <volume>21</volume>, <fpage>4695</fpage>&#x02013;<lpage>4708</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2012.2214050</pub-id><pub-id pub-id-type="pmid">22910118</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Moosavi-Dezfooli</surname> <given-names>S.-M.</given-names></name> <name><surname>Fawzi</surname> <given-names>A.</given-names></name> <name><surname>Frossard</surname> <given-names>P.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Deepfool: a simple and accurate method to fool deep neural networks,&#x0201D;</article-title> in <source>IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>2574</fpage>&#x02013;<lpage>2582</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pang</surname> <given-names>T.</given-names></name> <name><surname>Du</surname> <given-names>C.</given-names></name> <name><surname>Dong</surname> <given-names>Y.</given-names></name> <name><surname>Zhu</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). T&#x0201C;owards robust detection of adversarial examples,&#x0201D; in <italic>Advances in Neural Information Processing Systems</italic> (Montreal, QC: nips.cc), <volume>31</volume>, <fpage>4579</fpage>&#x02013;<lpage>4589</lpage>.</citation>
</ref>
<ref id="B32">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Raghunathan</surname> <given-names>A.</given-names></name> <name><surname>Steinhardt</surname> <given-names>J.</given-names></name> <name><surname>Liang</surname> <given-names>P.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Certified defenses against adversarial examples,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source> (<publisher-loc>ICLR.cc</publisher-loc>).</citation>
</ref>
<ref id="B33">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Raghuram</surname> <given-names>J.</given-names></name> <name><surname>Chandrasekaran</surname> <given-names>V.</given-names></name> <name><surname>Jha</surname> <given-names>S.</given-names></name> <name><surname>Banerjee</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;A general framework for detecting anomalous inputs to dnn classifiers,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>PMLR</publisher-loc>), <fpage>8764</fpage>&#x02013;<lpage>8775</lpage>.</citation>
</ref>
<ref id="B34">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Simonyan</surname> <given-names>K.</given-names></name> <name><surname>Zisserman</surname> <given-names>A.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Very deep convolutional networks for large-scale image recognition,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source> (<publisher-loc>San Diego, CA</publisher-loc>: <publisher-name>iclr.cc</publisher-name>).</citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Song</surname> <given-names>Y.</given-names></name> <name><surname>Kim</surname> <given-names>T.</given-names></name> <name><surname>Nowozin</surname> <given-names>S.</given-names></name> <name><surname>Ermon</surname> <given-names>S.</given-names></name> <name><surname>Kushman</surname> <given-names>N.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Pixeldefend: Leveraging generative models to understand and defend against adversarial examples,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source>.</citation>
</ref>
<ref id="B36">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Szegedy</surname> <given-names>C.</given-names></name> <name><surname>Zaremba</surname> <given-names>W.</given-names></name> <name><surname>Sutskever</surname> <given-names>I.</given-names></name> <name><surname>Bruna</surname> <given-names>J.</given-names></name> <name><surname>Erhan</surname> <given-names>D.</given-names></name> <name><surname>Goodfellow</surname> <given-names>I. J.</given-names></name> <name><surname>Fergus</surname> <given-names>R.</given-names></name></person-group> (<year>2014</year>). <article-title>&#x0201C;Intriguing properties of neural networks,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source> (<publisher-loc>iclr.cc</publisher-loc>).</citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tanay</surname> <given-names>T.</given-names></name> <name><surname>Griffin</surname> <given-names>L.</given-names></name></person-group> (<year>2016</year>). <article-title>A boundary tilting persepective on the phenomenon of adversarial examples</article-title>. <source>arXiv</source> [preprint] arXiv:1608.07690. <pub-id pub-id-type="doi">10.48550/arXiv.1608.07690</pub-id></citation>
</ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tian</surname> <given-names>J.</given-names></name> <name><surname>Zhou</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Duan</surname> <given-names>J.</given-names></name></person-group> (<year>2021</year>). <article-title>Detecting adversarial examples from sensitivity inconsistency of spatial-transform domain</article-title>. <source>AAAI Conf. Artif. Intellig</source>. <volume>35</volume>, <fpage>9877</fpage>&#x02013;<lpage>9885</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v35i11.17187</pub-id></citation>
</ref>
<ref id="B39">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Tramer</surname> <given-names>F.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Detecting adversarial examples is (nearly) as hard as classifying them,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>Baltimore, MD</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>21692</fpage>&#x02013;<lpage>21702</lpage>.</citation>
</ref>
<ref id="B40">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Tramer</surname> <given-names>F.</given-names></name> <name><surname>Carlini</surname> <given-names>N.</given-names></name> <name><surname>Brendel</surname> <given-names>W.</given-names></name> <name><surname>Madry</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;On adaptive attacks to adversarial example defenses,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems (NeurIPS)</source> (<publisher-loc>neurips.cc</publisher-loc>), <fpage>1633</fpage>&#x02013;<lpage>1645</lpage>.</citation>
</ref>
<ref id="B41">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Weng</surname> <given-names>T.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Chen</surname> <given-names>P.</given-names></name> <name><surname>Yi</surname> <given-names>J.</given-names></name> <name><surname>Su</surname> <given-names>D.</given-names></name> <name><surname>Gao</surname> <given-names>Y.</given-names></name> <name><surname>Hsieh</surname> <given-names>C.</given-names></name> <name><surname>Daniel</surname> <given-names>L.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Evaluating the robustness of neural networks: An extreme value theory approach,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source> (<publisher-loc>Vancouver, BC</publisher-loc>: <publisher-name>iclr.cc</publisher-name>).</citation>
</ref>
<ref id="B42">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>W.</given-names></name> <name><surname>Evans</surname> <given-names>D.</given-names></name> <name><surname>Qi</surname> <given-names>Y.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Feature squeezing: Detecting adversarial examples in deep neural networks,&#x0201D;</article-title> in <source>Network and Distributed System Security Symposium</source> (<publisher-loc>San Diego, CA</publisher-loc>: <publisher-name>NDSS-Symposium.org</publisher-name>).</citation>
</ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yin</surname> <given-names>X.</given-names></name> <name><surname>Kolouri</surname> <given-names>S.</given-names></name> <name><surname>Rohde</surname> <given-names>G. K.</given-names></name></person-group> (<year>2019</year>). <article-title>GAT: Generative adversarial training for adversarial example detection and robust classification</article-title>. <source>arXiv</source> [preprint] arXiv:1905.11475. <pub-id pub-id-type="doi">10.48550/arXiv.1905.11475</pub-id></citation>
</ref>
<ref id="B44">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>W.</given-names></name> <name><surname>Chellappa</surname> <given-names>R.</given-names></name> <name><surname>Phillips</surname> <given-names>P. J.</given-names></name> <name><surname>Rosenfeld</surname> <given-names>A.</given-names></name></person-group> (<year>2003</year>). <article-title>&#x0201C;Face recognition: a literature survey,&#x0201D;</article-title> in <source>ACM Computing Surveys</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>ACM</publisher-name>), <fpage>399</fpage>&#x02013;<lpage>458</lpage>.</citation>
</ref>
<ref id="B45">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>Z.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Larson</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Towards large yet imperceptible adversarial image perturbations with perceptual color distance,&#x0201D;</article-title> in <source>IEEE/CVF conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Seattle, WA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1039</fpage>&#x02013;<lpage>1048</lpage>.</citation>
</ref>
<ref id="B46">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zheng</surname> <given-names>Z.</given-names></name> <name><surname>Hong</surname> <given-names>P.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Robust detection of adversarial attacks by modeling the intrinsic properties of deep neural networks,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source> (<publisher-loc>Montreal, CA</publisher-loc>: <publisher-name>neurips.cc</publisher-name>), <fpage>31</fpage>.</citation>
</ref>
</ref-list>
</back>
</article>