<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurosci.</journal-id>
<journal-title>Frontiers in Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-453X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnins.2024.1362286</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Algorithm of face anti-spoofing based on pseudo-negative features generation</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Ma</surname> <given-names>Yukun</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2623796/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Lyu</surname> <given-names>Chengzhen</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2610925/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Liangliang</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2633991/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wei</surname> <given-names>Yajun</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2670964/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Xu</surname> <given-names>Yaowen</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>School of Software, Henan Institute of Science and Technology</institution>, <addr-line>Xinxiang</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>School of Information Engineering, Henan Institute of Science and Technology</institution>, <addr-line>Xinxiang</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>School of Information and Electronics, Beijing Institute of Technology</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Data and AI Technology Company, China Telecom Corporation Ltd.</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<author-notes>
<fn id="fn0001" fn-type="edited-by"><p>Edited by: Lu Tang, Xuzhou Medical University, China</p></fn>
<fn id="fn0002" fn-type="edited-by"><p>Reviewed by: Chhavi Dhiman, Delhi Technological University, India</p>
<p>Qianyu Zhou, Shanghai Jiao Tong University, China</p></fn>
<corresp id="c001">&#x002A;Correspondence: Yukun Ma, <email>yukuner@126.com</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>12</day>
<month>04</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>18</volume>
<elocation-id>1362286</elocation-id>
<history>
<date date-type="received">
<day>28</day>
<month>12</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>03</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2024 Ma, Lyu, Li, Wei and Xu.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Ma, Lyu, Li, Wei and Xu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Despite advancements in face anti-spoofing technology, attackers continue to pose challenges with their evolving deceptive methods. This is primarily due to the increased complexity of their attacks, coupled with a diversity in presentation modes, acquisition devices, and prosthetic materials. Furthermore, the scarcity of negative sample data exacerbates the situation by causing domain shift issues and impeding robust generalization. Hence, there is a pressing need for more effective cross-domain approaches to bolster the model&#x2019;s capability to generalize across different scenarios.</p>
</sec>
<sec>
<title>Methods</title>
<p>This method improves the effectiveness of face anti-spoofing systems by analyzing pseudo-negative sample features, expanding the training dataset, and boosting cross-domain generalization. By generating pseudo-negative features with a new algorithm and aligning these features with the use of KL divergence loss, we enrich the negative sample dataset, aiding the training of a more robust feature classifier and broadening the range of attacks that the system can defend against.</p>
</sec>
<sec>
<title>Results</title>
<p>Through experiments on four public datasets (MSU-MFSD, OULU-NPU, Replay-Attack, and CASIA-FASD), we assess the model&#x2019;s performance within and across datasets by controlling variables. Our method delivers positive results in multiple experiments, including those conducted on smaller datasets.</p>
</sec>
<sec>
<title>Discussion</title>
<p>Through controlled experiments, we demonstrate the effectiveness of our method. Furthermore, our approach consistently yields favorable results in both intra-dataset and cross-dataset evaluations, thereby highlighting its excellent generalization capabilities. The superior performance on small datasets further underscores our method&#x2019;s remarkable ability to handle unseen data beyond the training set.</p>
</sec>
</abstract>
<kwd-group>
<kwd>face anti-spoofing</kwd>
<kwd>pseudo-negative feature</kwd>
<kwd>features generation</kwd>
<kwd>feature analysis</kwd>
<kwd>cross-domain</kwd>
</kwd-group>
<counts>
<fig-count count="10"/>
<table-count count="7"/>
<equation-count count="7"/>
<ref-count count="56"/>
<page-count count="13"/>
<word-count count="8220"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Visual Neuroscience</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>With the continuous development of computer technology, identity authentication based on face information has been widely used. However, most existing face recognition methods are very vulnerable to face prosthesis attacks. Face spoofing attack refers to illegal users attempting to cheat the face authentication system and the face detection system through some prosthesis methods, such as print attacks, replay attacks, and mask attacks. Face anti-spoofing is developed to detect illegal facial spoofing attacks, thereby improving the security of face authentication systems (<xref ref-type="bibr" rid="ref50">Yu et al., 2022</xref>).</p>
<p>Though facial recognition technology has been widely used in biometric authentication, it is susceptible to presentation attacks (commonly referred to as &#x201C;spoofing attacks&#x201D;), which have attracted much attention in secure scenarios. These attack forms include using synthesized or fake facial images or information to mimic the facial features of legitimate users, thereby bypassing facial recognition systems. Examples of such attacks include printed photos, facial digital images on electronic screens, 3D masks, and other innovative methods. There are special material attacks, where facial models made from special materials attempt to evade traditional facial recognition systems; meanwhile, virtual generation attacks utilize computer graphics and generative adversarial networks (GANs) to produce realistic synthetic faces and bypass facial recognition systems; additionally, lighting manipulation attacks use lighting effects, special lights, or reflective materials to change facial appearance, making it challenging for systems to accurately identify faces. Though various methods have been proposed to defend against these attacks, existing defense methods often lack sufficient generalization ability when confronted with unknown attacks types (<xref ref-type="bibr" rid="ref9">de Freitas Pereira et al., 2013</xref>). In practical scenarios, training facial anti-spoofing models to predict all types of attacks is a challenging task.</p>
<p>Face anti-spoofing technology, designed to detect and prevent fraud in facial recognition, has significantly advanced in recent years, yielding promising results. However, a major challenge for current methods is their limited ability to generalize to previously unseen or novel attack types. In the real world, it&#x2019;s nearly impossible to anticipate and incorporate all potential attack scenarios into the training phase, which makes maintaining effectiveness difficult.</p>
<p>As technology evolves and face anti-spoofing techniques become more sophisticated, attackers are also adapting their deceptive methods, leading to new and more complex attack forms. The vast and diverse data space associated with prosthetic attacks, involving high-quality masks or other facial replicas, poses a significant challenge for cross-domain face anti-spoofing. This diversity in attack methods, coupled with variations in presentation, acquisition devices, and prosthetic materials, complicates the task of developing robust and generalizable solutions.</p>
<p>In cross-domain scenarios, where data from multiple sources or domains are involved, existing methods often face significant challenges in training and testing across various devices and materials. These introduce distinct characteristics and variations that can greatly impact model performance and reliability. The fundamental issue is the inadequacy of negative sample data when faced with diverse attacks or perturbations. This scarcity prevents models from adequately learning and generalizing to new, unseen domains, leading to domain shift issues during learning. There&#x2019;s an urgent need for more robust and effective approaches to address these issues and enhance cross-domain performance.</p>
<p>The contributions of this paper are numerous and significant. Firstly, we introduce an innovative algorithm capable of generating pseudo-negative features by collecting and analyzing features from existing datasets. Secondly, we employ the Kullback&#x2013;Leibler (KL) divergence loss function to effectively guide the distribution of the generated virtual features, ensuring their alignment with the desired characteristics and further optimizing the system&#x2019;s accuracy. Finally, our approach has achieved promising results across multiple cross-domain tests, demonstrating robust performance. Overall, our contributions advance the state-of-the-art in face anti-spoofing technology.</p>
</sec>
<sec id="sec2">
<label>2</label>
<title>Related work</title>
<p>At the initial stage, manually annotated features were used to construct face anti-spoofing. <xref ref-type="bibr" rid="ref30">M&#x00E4;&#x00E4;tt&#x00E4; et al. (2011)</xref> developed a method based on the analysis of facial textures to determine whether there is a living person or facial imprint in front of the camera. <xref ref-type="bibr" rid="ref10">de Freitas Pereira et al. (2014)</xref> extracted local binary patterns (LBP) features in three orthogonal planes of spatiotemporal space for face fraud detection. Similarly, most of the histogram-based 2D features can be generalized to their corresponding 3D forms. In recent years, face anti-spoofing based on deep learning has attracted much attention. Compared with traditional hand-crafted features, deep features learned by the neural network have a more robust representation ability, and the accuracy of the trained model is also greatly enhanced. <xref ref-type="bibr" rid="ref48">Yang et al. (2014)</xref> first applied the Convolutional Neural Network (CNN) to face anti-spoofing by using the AlexNet network model as a feature extractor to extract the features of the original image and using the Support Vector Machine (SVM) for classification. <xref ref-type="bibr" rid="ref31">Menotti et al. (2015)</xref> employed the hyperparameter search method to find a suitable CNN network structure for face fraud detection. To narrow the search range of hyperparameters, the searched CNN contained at most three convolutional layers. <xref ref-type="bibr" rid="ref34">Rehman et al. (2017)</xref> trained an 11-layer VGG network and two variant networks in an end-to-end manner for face fraud detection. <xref ref-type="bibr" rid="ref32">Nagpal and Dubey (2019)</xref> investigated deeper face fraud detection based on ResNet and GoogLeNet. <xref ref-type="bibr" rid="ref18">Li et al. (2016)</xref> used transfer learning to extract features after fine-tuning the pre-trained VGG face model, which mitigated overfitting in the model. Some researchers replaced the original hand-crafted features with features learned by the network (<xref ref-type="bibr" rid="ref6">Cai et al., 2022</xref>). Additionally, the optical flow feature provides an effective method for extracting motion information from videos (<xref ref-type="bibr" rid="ref37">Simonyan and Zisserman, 2014</xref>; <xref ref-type="bibr" rid="ref40">Sun et al., 2016</xref>, <xref ref-type="bibr" rid="ref41">2019</xref>). <xref ref-type="bibr" rid="ref49">Yin et al. (2016)</xref> found motion cues of face fraud based on optical flow features. <xref ref-type="bibr" rid="ref33">Pinto et al. (2015)</xref> proposed a feature based on low-level motion features and mid-level visual encoding for face fraud detection. <xref ref-type="bibr" rid="ref11">De Marsico et al. (2012)</xref> extracted geometrically invariant features around facial feature points to detect cues in video replay. Moreover, some studies used temporal features between consecutive frames for face anti-spoofing (<xref ref-type="bibr" rid="ref43">Wang et al., 2022a</xref>).</p>
<p>In the early stage, the deep learning-based detection algorithm employed the softmax loss function for face authenticity classifications. Although these methods improved the detection performance on a single database, their generalization ability remained challenging when tested across data sets. Different from the previous binary classification approach, <xref ref-type="bibr" rid="ref25">Liu et al. (2018)</xref> proposed training networks using auxiliary information. This method combined face depth information and rPPG (remote photoplethysmography) as an auxiliary supervised guidance model to learn essential features, and it achieved a good detection effect. <xref ref-type="bibr" rid="ref16">Kim et al. (2019)</xref> introduced reflection-based supervision based on depth graph supervision, which further improved the network&#x2019;s detection performance. Moreover, <xref ref-type="bibr" rid="ref19">Li et al. (2020)</xref> and <xref ref-type="bibr" rid="ref51">Yu et al. (2020)</xref> proposed new convolution operators and loss functions for live face detection, respectively. To better resist various unknown attacks and improve the generalization ability of deep models across data sets, researchers also used zero-shot learning (<xref ref-type="bibr" rid="ref26">Liu et al., 2019</xref>), domain adaptation, and domain generalization to enhance the model&#x2019;s generalization ability (<xref ref-type="bibr" rid="ref35">Saha et al., 2020</xref>; <xref ref-type="bibr" rid="ref45">Wang et al., 2021</xref>). To obtain better domain generalization approaches, <xref ref-type="bibr" rid="ref15">Jia et al. (2020)</xref> proposed an end-to-end single-side domain generalization framework (SSDG) to improve the generalization ability of face anti-spoofing. Furthermore, <xref ref-type="bibr" rid="ref12">Dong et al. (2021)</xref> proposed an end-to-end open-set face anti-spoofing (OSFA) approach for recognizing unseen attacks. However, the accuracy and generalization ability of classification models are still areas of active research.</p>
<p>In recent years, the application of transformers in the visual domain has led to numerous advancements in addressing domain generalization issues. Specifically, approaches like the Domain-invariant Vision Transformer (DiVT) have effectively leveraged transformers to enhance the generalization capabilities of face anti-spoofing tasks (<xref ref-type="bibr" rid="ref20">Liao et al., 2023</xref>). Additionally, initializing Vision Transformers (ViT) with pre-trained weights from multimodal models such as CLIP has been shown to improve the generalization of FAS tasks (<xref ref-type="bibr" rid="ref38">Srivatsan et al., 2023</xref>). Furthermore, adaptive ViT models have been introduced for robust cross-domain face anti-spoofing (<xref ref-type="bibr" rid="ref14">Huang et al., 2022</xref>). By employing overlapping patches and parameter sharing within the ViT network, these approaches efficiently utilize multiple modalities, resulting in computationally efficient face anti-spoofing solutions (<xref ref-type="bibr" rid="ref3">Antil and Dhiman, 2024</xref>).</p>
<p>To further enhance domain generalization, unsupervised or self-supervised methods have been employed during model construction and training. One such approach involves stylizing target data to match the source domain style using image translation techniques and then classifying the stylized data using a well-trained source model (<xref ref-type="bibr" rid="ref56">Zhou et al., 2022a</xref>). Additionally, novel frameworks such as Source-free Domain Adaptation for Face Anti-Spoofing (SDAFAS; <xref ref-type="bibr" rid="ref22">Liu et al., 2022a</xref>) and a source data-free domain adaptive face anti-spoofing framework (<xref ref-type="bibr" rid="ref29">Lv et al., 2021</xref>) have been proposed to tackle issues related to source knowledge adaptation and target data exploration in a source-free setting. These frameworks aim to optimize the network in the target domain without relying on labeled source data by treating it as a problem of learning with noisy labels.</p>
<p>Moreover, a new perspective for domain generalization in face anti-spoofing has been introduced that focuses on aligning features at the instance level without requiring domain labels (<xref ref-type="bibr" rid="ref54">Zhou et al., 2023</xref>). Frameworks like the Unsupervised Domain Generalization for Face Anti-Spoofing (UDGFAS) exploit large amounts of easily accessible unlabeled data to learn generalizable features (<xref ref-type="bibr" rid="ref24">Liu et al., 2023</xref>), thereby enhancing the performance of FAS in low-data regimes. These approaches explore the relationship between source domains and unseen domains to achieve effective domain generalization.</p>
<p>Additionally, a self-domain adaptation framework has been proposed that leverages unlabeled test domain data during inference time (<xref ref-type="bibr" rid="ref45">Wang et al., 2021</xref>). Another approach involves encouraging domain separability while aligning the live-to-spoof transition (i.e., the trajectory from live to spoof) to be consistent across all domains (<xref ref-type="bibr" rid="ref39">Sun et al., 2023</xref>). The Adaptive Mixture of Experts Learning (AMEL) framework (<xref ref-type="bibr" rid="ref55">Zhou et al., 2022b</xref>) exploits domain-specific information to adaptively establish links among seen source domains and unseen target domains, further improving generalization. A generalizable Face Anti-Spoofing approach based on causal intervention is proposed, aiming to enhance the model&#x2019;s generalization ability in unseen scenarios by identifying and adjusting domain-related confounding factors (<xref ref-type="bibr" rid="ref23">Liu et al., 2022b</xref>).</p>
<p>Studying the local features of images has also proven beneficial for achieving good domain generalization. For instance, PatchNet reformulates face anti-spoofing as a fine-grained patch-type recognition problem, recognizing combinations of capturing devices and presentation materials based on patches cropped from non-distorted face images (<xref ref-type="bibr" rid="ref42">Wang C. Y. et al., 2022</xref>). Furthermore, a novel Selective Domain-invariant Feature Alignment Network (SDFANet) has been proposed for cross-domain face anti-spoofing. This network aims to seek common feature representations by fully exploring the generalization capabilities of different regions within images (<xref ref-type="bibr" rid="ref53">Zhou et al., 2021</xref>).</p>
<p>The current limited cross-domain performance of facial liveness detection methods is due to the incomplete nature of negative sample data under diverse attacks. Based on the above research, considering that the existing feature information is not complete while disregarding the relationship between features, this paper proposes a new face anti-spoofing method based on CNN to generate pseudo-negative feature data of the training sample, and then calculate the feature distribution, and control the generation of the virtual feature distribution by using the KL divergence loss function. Additionally, based on the generated new pseudo data, the proposed method employs a collaborative training algorithm with the original features to improve the generalization performance of face anti-spoofing systems.</p>
</sec>
<sec id="sec3">
<label>3</label>
<title>Proposed method</title>
<p>Face anti-spoofing is a binary classification task (real/fake). Unlike typical coarse-grained binary classification tasks, the liveness detection task exhibits a property that is inconsistent with human visual distance, as illustrated in <xref ref-type="fig" rid="fig1">Figure 1</xref>.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p><bold>(A)</bold> True and false samples of different people in human vision; <bold>(B)</bold> True and false samples of different people in the living body detection classifier.</p>
</caption>
<graphic xlink:href="fnins-18-1362286-g001.tif"/>
</fig>
<p>Currently, most of the studies on face anti-spoofing systems focus on increasing the type and number of attack samples to enhance the stability and generalization of face anti-spoofing systems. However, due to the unseen data in the training stage, the original method has some limitations in dealing with unknown attack methods.</p>
<p>By analyzing existing face anti-spoofing methods, it is observed that the incompleteness of negative samples is the primary factor limiting the algorithm&#x2019;s cross-domain performance. Therefore, this method aims to research pseudo-negative sample features, expand the training dataset, and improve the cross-domain generalization of face anti-spoofing methods. First, to address the issue of incomplete negative samples, this study generates pseudo-negative features based on the distribution of <italic>bona fide</italic> and attack features. These features complement existing negative class data, enhancing the diversity and completeness of the negative sample dataset. Then, this study uses pseudo-negative features together with existing negative class data to assist in training a feature classifier for real faces, further adjusting the parameters of the feature extractor. The generation of pseudo-negative features leads to more comprehensive negative sample features during training, making the system cover attack data in a broader range of scenarios and thus improving the generalization of the detection method.</p>
<p>In the context of prosthetic attacks, there exists a certain level of feature dispersion across various attack scenarios, suggesting a wider intra-class variation. Due to this, cross-scenario liveness detection poses a certain challenge, and collecting all types of attack data during the training process can be challenging. The differences in intra-class distribution between seen and unseen attack types often lead to domain shift issues. To tackle these challenges, this study employs a technique for generating pseudo-negative class features, aiming to directly learn the mapping between the visual space of images and the semantic space of features. This method can avoid information loss. Finally, this study develops an end-to-end training model applicable to cross-domain face liveness detection.</p>
<p>The method proposed in this paper comprises of feature analysis, feature generation, and collaborative training. As illustrated in <xref ref-type="fig" rid="fig2">Figure 2</xref>, the general workflow of the method is as follows: First, after images are inputted, the CNN generates multi-dimensional feature tensor data from the training samples. Then, the tensor data is analyzed to generate new feature data based on their feature distribution and KL divergence value. Meanwhile, attack types and unseen data from the training stage are incorporated to augment the original set of negative features. Finally, the model is trained using both virtual and existing sample features, allowing us to gather the feature distribution of <italic>bona fide</italic> samples and subsequently improve the accuracy and robustness of live face detection.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>The structure diagram of generating pseudo-negative features for face anti-spoofing. The real and attack images are input into the CNN to extract the <italic>bona fide</italic> features and attack features. Then, the distribution of the attack features and the distribution of the <italic>bona fide</italic> features are obtained. These two feature distribution data are fed into the pseudo-negative feature generator to generate the distribution of pseudo-negative features. Finally, the classification task is completed by going through the Fc and the softmax layers. Facial images reproduced with permission from OULU-NPU dataset (<xref ref-type="bibr" rid="ref5">Boulkenafet et al., 2017a</xref>).</p>
</caption>
<graphic xlink:href="fnins-18-1362286-g002.tif"/>
</fig>
<p>During the feature generation process, the corresponding feature distributions are computed by leveraging the extracted features from both attack and <italic>bona fide</italic> images. Then, the distribution data is fed into the data generator <inline-formula><mml:math id="M1"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi>p</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, which uses a random data generator based on these distributions to generate a pseudo-negative feature distribution <inline-formula><mml:math id="M2"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>P</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> that fits the attack feature distribution. The structure of the data generator <inline-formula><mml:math id="M3"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi>p</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is presented in <xref ref-type="fig" rid="fig3">Figure 3</xref>.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>The pseudo-negative feature generator <inline-formula><mml:math id="M4"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi>p</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. The <inline-formula><mml:math id="M5"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> of attack features and the <inline-formula><mml:math id="M6"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>R</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> of <italic>bona fide</italic> features are input into <inline-formula><mml:math id="M7"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi>p</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. Firstly, according to <inline-formula><mml:math id="M8"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, the random generator is used to generate the <inline-formula><mml:math id="M9"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>M</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> that fits the distribution of <inline-formula><mml:math id="M10"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, and the loss function <inline-formula><mml:math id="M11"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mi>p</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is designed to optimize the distribution <inline-formula><mml:math id="M12"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>M</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> of generated pseudo-negative features. To prevent overfitting of the data, a random noise <inline-formula><mml:math id="M13"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>&#x03B8;</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is generated according to <inline-formula><mml:math id="M14"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, and the final virtual feature distribution <inline-formula><mml:math id="M15"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>P</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is obtained by combining <inline-formula><mml:math id="M16"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>&#x03B8;</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> with <inline-formula><mml:math id="M17"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>M</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>.</p>
</caption>
<graphic xlink:href="fnins-18-1362286-g003.tif"/>
</fig>
<p>This section introduces the proposed method from three aspects: feature analysis, feature generation, and loss function.</p>
<sec id="sec4">
<label>3.1</label>
<title>Feature analysis</title>
<p>In this paper, we utilize Android and laptop camera devices to acquire face images and subsequently calculate their feature distributions, aiming to analyze the disparities between real and attack face images. As depicted in <xref ref-type="fig" rid="fig4">Figure 4A</xref>, it is evident that regardless of the capturing device used, the features of <italic>bona fide</italic> face images conform to a normal distribution, resulting in a relatively clustered pattern. <xref ref-type="fig" rid="fig4">Figure 4B</xref> illustrates the image features of attack faces across three distinct display media: three variations of iPad replay video attacks, iPhone replay video attacks, and photo print attacks. Notably, the feature distribution of attack face images employing different display media appears scattered, highlighting the variations in feature distribution among diverse attack methodologies.</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>The distribution of the feature tensors of the statistical images. <bold>(A)</bold> The statistical tensor distribution of <italic>bona fide</italic> images of different types, and <bold>(B)</bold> the tensor of all types of attack images in the statistical dataset.</p>
</caption>
<graphic xlink:href="fnins-18-1362286-g004.tif"/>
</fig>
<p>In light of the characteristics of normal distribution, we aim to generate pseudo-negative feature data from the original sample feature data in order to enhance network performance. Toward this objective, our paper proposes a methodological framework. Initially, we examine the extracted feature data from the training samples obtained via Convolutional Neural Networks (CNNs). Subsequently, we synthesize pseudo-negative feature data that closely resembles the original sample feature data, ensuring alignment with the inherent distributional properties. Finally, we incorporate this pseudo-negative feature data into the classifier training process, with the ultimate goal of bolstering the accuracy and generalization capabilities of the face anti-spoofing system.</p>
<p>In face anti-spoofing systems, <italic>bona fide</italic> sample data are typically acquired through equipment-based face data collection. Conversely, attack samples, encompassing image-based and video replay assaults, primarily initiate with frontal face information gathering followed by secondary imaging involving facial prostheses via shooting equipment. Notably, while the <italic>bona fide</italic> sample collection method remains consistent across various data sets, attack samples may exhibit a more scattered distribution due to disparities in devices and attack methodologies (<xref ref-type="bibr" rid="ref15">Jia et al., 2020</xref>). This difference makes the real face features of different data sets more likely to gather than the attack face features. In the practical application of the face anti-spoofing system, the classification boundary trained based on existing datasets may lead to overlapping characteristics between <italic>bona fide</italic> and novel attack sample data in certain domains, thereby impeding accurate classification. As illustrated in <xref ref-type="fig" rid="fig5">Figure 5A</xref>, the classification boundary delineates the feature space into <italic>bona fide</italic> and attack regions. To enhance system performance and ensure robust responsiveness to emerging attacks encountered in real-world scenarios, this study introduces the generation of pseudo-negative feature data (depicted in <xref ref-type="fig" rid="fig5">Figure 5B</xref>). This approach serves to augment the feature representation of samples, facilitating the clustering of <italic>bona fide</italic> data and optimizing classification outcomes. Consequently, the accuracy and generalization capabilities of face anti-spoofing systems are substantially improved.</p>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>The goal of the proposed method. <bold>(A)</bold> The classification boundary without adding pseudo-negative features, and <bold>(B)</bold> the classification boundary after adding pseudo-negative features.</p>
</caption>
<graphic xlink:href="fnins-18-1362286-g005.tif"/>
</fig>
</sec>
<sec id="sec5">
<label>3.2</label>
<title>Feature generation</title>
<p>In terms of current technology, the collection method for real face data across various datasets is relatively straightforward, as the equipment gathers facial data information directly. Consequently, the feature information of attack face samples tends to be more scattered compared to <italic>bona fide</italic> faces. Additionally, in practical applications, numerous unseen novel attack methods will arise. Therefore, the feature generation module performs feature generation and completes the new attack features in the unknown domain.</p>
<p>According to the analysis presented in section 3.1, the proposed image features follow a normal distribution, and the mean value and standard deviation can be calculated. In this study, a feature sequence that matches the mean and standard deviation of the original feature is randomly generated. Assuming <inline-formula><mml:math id="M18"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>R</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is the distribution of the <italic>bona fide</italic> sample data, <inline-formula><mml:math id="M19"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is the distribution of the attack sample data, and <inline-formula><mml:math id="M20"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>P</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is the distribution of the generated features. To make the model achieve better performance, relative entropy, also known as Kullback&#x2013;Leibler divergence, is used as the loss function of the feature-generating module. In the initialization process, <inline-formula><mml:math id="M21"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, i.e., the generated features and the attack sample features remain in the same distribution. At this time, the <inline-formula><mml:math id="M222"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>K</mml:mi><mml:mi>L</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo>&#x2225;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mi>R</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> has the minimum value, and the classification problem is relatively simple. In the optimization process, the distribution of pseudo-negative features approaches the <italic>bona fide</italic> sample gradually, which increases the multiformity of the attack sample, promotes the gathering of <italic>bona fide</italic> features, improves the classification accuracy of the face anti-spoofing system, and enhances the generalization of invisible new attacks. The loss function of feature generation is shown in the following <xref ref-type="disp-formula" rid="EQ1">Equation (1)</xref>.</p>
<disp-formula id="EQ1"><label>(1)</label><mml:math id="M23"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mtext mathvariant="italic">Pseudo</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>K</mml:mi><mml:mi>L</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo>&#x2225;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mi>R</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>K</mml:mi><mml:mi>L</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo>&#x2225;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>K</mml:mi><mml:mi>L</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo>&#x2225;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mi>R</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula>
<p>As shown in <xref ref-type="disp-formula" rid="EQ2">Equation (2)</xref>, where <inline-formula><mml:math id="M24"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> represents the tensor data of the attack sample extracted by the feature extractor, <inline-formula><mml:math id="M25"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> randomly generates the data according to the mean and variance of the attack and the <italic>bona fide</italic> sample tensor, and <inline-formula><mml:math id="M26"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mi>&#x03B8;</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> represents the random noise generated according to the <inline-formula><mml:math id="M27"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi>p</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>.</p>
<disp-formula id="EQ2"><label>(2)</label><mml:math id="M28"><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi>p</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover></mml:mstyle><mml:munder><mml:mrow><mml:mi>min</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:munder><mml:mo>&#x2225;</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mi>A</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:msup><mml:mo>&#x2225;</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>+</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mi>&#x03B8;</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable><mml:mspace width="thickmathspace"/></mml:mrow></mml:math></disp-formula>
</sec>
<sec id="sec6">
<label>3.3</label>
<title>Loss function</title>
<p>After generating the pseudo-negative feature data, it should be integrated into the face anti-spoofing system to enhance its performance. The cross-entropy loss function can be employed in neural networks as a metric to assess the similarity between the distribution of <italic>bona fide</italic> markers and the distribution predicted by the trained model. In this study, both the original feature data and the generated pseudo-negative feature data are concurrently fed into the loss function, aiming to enhance the generalizability and stability of the face anti-spoofing system in real-world applications. The overall network loss is defined as <xref ref-type="disp-formula" rid="EQ3">Equation (3)</xref>:</p>
<disp-formula id="EQ3"><label>(3)</label><mml:math id="M29"><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mtext mathvariant="italic">Whole</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>&#x03D1;</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03D1;</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mtext mathvariant="italic">Pseudo</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M30"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mtext mathvariant="italic">Whole</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> represents the overall loss function of the network, <inline-formula><mml:math id="M31"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> represents the loss function of the original features, <inline-formula><mml:math id="M32"><mml:mrow><mml:msub><mml:mi>&#x03D1;</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:math></inline-formula> denotes the weight parameter of the original features, <inline-formula><mml:math id="M33"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mtext mathvariant="italic">Pseudo</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> is the loss function of the newly generated features, and <inline-formula><mml:math id="M34"><mml:mrow><mml:msub><mml:mi>&#x03D1;</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow></mml:math></inline-formula> denotes the weight parameter of the newly generated features. The visual representation of the roles played by <inline-formula><mml:math id="M35"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M36"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mtext mathvariant="italic">Pseudo</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> in the processes of feature generation and classifier boundary training is depicted in <xref ref-type="fig" rid="fig6">Figure 6</xref>.</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>The visual representation of the roles played by <inline-formula><mml:math id="M37"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M38"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mtext mathvariant="italic">Pseudo</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> in the processes of feature generation and classifier boundary training.</p>
</caption>
<graphic xlink:href="fnins-18-1362286-g006.tif"/>
</fig>
</sec>
</sec>
<sec id="sec7">
<label>4</label>
<title>Experimental setup</title>
<sec id="sec8">
<label>4.1</label>
<title>Databases</title>
<p>To evaluate the effectiveness of the proposed algorithm, it was tested on three publicly available face datasets, including MSU-MFSD (<xref ref-type="bibr" rid="ref46">Wen et al., 2015</xref>), OULU-NPU (<xref ref-type="bibr" rid="ref5">Boulkenafet et al., 2017a</xref>), and Replay-Attack (<xref ref-type="bibr" rid="ref8">Chingovska et al., 2012</xref>).</p>
<p>The MSU-MFSD dataset (shown in <xref ref-type="fig" rid="fig7">Figure 7</xref>) was released by Michigan State University in 2015. Currently, it consists of 280 videos, publicly available and featuring 35 individuals. The dataset consists of three attack types: iPad air video replay attack, iphone5S video replay attack, and A3 paper printed photo attack.</p>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>Some samples of the subjects recorded in the MSU-MFSD dataset. Images reproduced with permission from MSU-MFSD dataset (<xref ref-type="bibr" rid="ref46">Wen et al., 2015</xref>).</p>
</caption>
<graphic xlink:href="fnins-18-1362286-g007.tif"/>
</fig>
<p>The OULU-NPU dataset (shown in <xref ref-type="fig" rid="fig8">Figure 8</xref>) was released by the University of Oulu in Finland in 2017. It consists of 4,950 video clips, captured from 55 participants with 90 videos collected per participant. The dataset consists of four types of attacks: photo attacks printed by two different printers, and video replay attacks displayed by two different display devices.</p>
<fig position="float" id="fig8">
<label>Figure 8</label>
<caption>
<p>Some samples of the subjects recorded in the OULU-NPU dataset. Images reproduced with permission from OULU-NPU dataset (<xref ref-type="bibr" rid="ref5">Boulkenafet et al., 2017a</xref>).</p>
</caption>
<graphic xlink:href="fnins-18-1362286-g008.tif"/>
</fig>
<p>The Replay-Attack dataset (shown in <xref ref-type="fig" rid="fig9">Figure 9</xref>) was released in 2017 and is comprised of 1,200 video clips. These videos feature 50 clients and showcase attack attempts under varying lighting conditions.</p>
<fig position="float" id="fig9">
<label>Figure 9</label>
<caption>
<p>Some samples of the subjects recorded in the Replay-Attack dataset. Images reproduced with permission from Replay-Attack dataset (<xref ref-type="bibr" rid="ref8">Chingovska et al., 2012</xref>).</p>
</caption>
<graphic xlink:href="fnins-18-1362286-g009.tif"/>
</fig>
<p>Since the dataset comprises entirely of video files, all videos and images were extracted frame-by-frame, and all images have undergone normalization. In these datasets, there are more attack samples than <italic>bona fide</italic> samples, with a large difference in number. During the training process, the quantity of attack and <italic>bona fide</italic> samples was carefully balanced to maintain a similar range, aiming to minimize both data quantity and the chance of overfitting. During data set division, owing to the varied nature of attack samples, the quantity of data samples gathered within identical environmental conditions was two to four times higher compared to <italic>bona fide</italic> samples. Therefore, the attack sample takes the image by the proportion of the <italic>bona fide</italic> sample. In contrast, the attack sample is often intercepted to maintain the amount of the two data in a similar range.</p>
</sec>
<sec id="sec9">
<label>4.2</label>
<title>Experimental metrics</title>
<p>In face anti-spoofing, there are four types of prediction results: True Positives (<italic>TP</italic>), where positive samples are predicted by the model as positive classes; True Negatives (<italic>TN</italic>), where negative samples are predicted by the model as negative classes; False Positives (<italic>FP</italic>), where negative samples are predicted by the model as positive classes; False Negatives (<italic>FN</italic>), where positive samples are predicted by the model as negative classes.</p>
<p>Performance evaluation indicators include Attack Presentation Classification Error Rate (<italic>APCER</italic>), <italic>Bona Fide</italic> Presentation Classification Error Rate (<italic>BPCER</italic>), Average Classification Error Rate (<italic>ACER</italic>), Half Total Error Rate (<italic>HTER</italic>), and Area Under the ROC Curve (<italic>AUC</italic>). These performance indicators are calculated as follows <xref ref-type="disp-formula" rid="EQ4">Equations (4&#x2013;7)</xref>:</p>
<disp-formula id="EQ4"><label>(4)</label><mml:math id="M39"><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:mtext mathvariant="italic">APCER</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr></mml:mtable><mml:mspace width="thickmathspace"/></mml:mrow></mml:math></disp-formula>
<disp-formula id="EQ5"><label>(5)</label><mml:math id="M40"><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:mtext mathvariant="italic">BPCER</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr></mml:mtable><mml:mspace width="thickmathspace"/></mml:mrow></mml:math></disp-formula>
<disp-formula id="EQ6"><label>(6)</label><mml:math id="M41"><mml:mrow><mml:mtext mathvariant="italic">ACER</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mtext mathvariant="italic">APCER</mml:mtext><mml:mo>+</mml:mo><mml:mtext mathvariant="italic">BPCER</mml:mtext></mml:mrow><mml:mrow><mml:mn>2.0</mml:mn></mml:mrow></mml:mfrac><mml:mspace width="thickmathspace"/></mml:mrow></mml:math></disp-formula>
<disp-formula id="EQ7"><label>(7)</label><mml:math id="M42"><mml:mrow><mml:mtext mathvariant="italic">HTER</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mtext mathvariant="italic">FAR</mml:mtext><mml:mo>+</mml:mo><mml:mtext mathvariant="italic">FRR</mml:mtext></mml:mrow><mml:mrow><mml:mn>2.0</mml:mn></mml:mrow></mml:mfrac><mml:mspace width="thickmathspace"/></mml:mrow></mml:math></disp-formula>
<p>where <italic>FAR</italic> represents the false acceptance rate, and it is calculated as <inline-formula><mml:math id="M43"><mml:mrow><mml:mtext mathvariant="italic">FAR</mml:mtext><mml:mo>=</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>/</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, and <italic>FRR</italic> represents the false rejection rate, and it is calculated as <inline-formula><mml:math id="M44"><mml:mrow><mml:mtext mathvariant="italic">FRR</mml:mtext><mml:mo>=</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi><mml:mo>/</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>F</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>.</p>
</sec>
<sec id="sec10">
<label>4.3</label>
<title>Experimental environment</title>
<p>The experiment was conducted on a computer equipped with an AMD Ryzen 75,800&#x00D7; 8-Core CPU, 32&#x2009;GB memory, and Nvidia GTX 3060 GPU (12&#x2009;GB video memory), and the computer runs the Windows 10 operating system. The proposed algorithm was implemented based on the PyTorch framework. The Adam optimizer was adopted for model optimization with a learning rate of 2.00e-4 and a batch size of 32.</p>
</sec>
</sec>
<sec id="sec11">
<label>5</label>
<title>Experimental results</title>
<sec id="sec12">
<label>5.1</label>
<title>Control experiment</title>
<p>In this paper, as a control group, the deep learning network AlexNet was trained and tested on the OULU-NPU dataset and MSU-MFSD dataset (<xref ref-type="bibr" rid="ref17">Krizhevsky et al., 2012</xref>). Based on the native AlexNet, a pseudo-negative feature generation module was added, and then the model was trained and tested on two datasets. The performance of the two models on the OULU-NPU and MSU-FASD datasets is presented in <xref ref-type="table" rid="tab1">Tables 1</xref>, <xref ref-type="table" rid="tab2">2</xref>, respectively. The results in the two tables show that in the model with the pseudo-negative feature generation module, APCER significantly decreased; in most protocols, BPCER reduced correspondingly, and the overall ACER was diminished.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>The performance on the OULU-NPU dataset.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Protocol</th>
<th align="left" valign="top">Model</th>
<th align="center" valign="top">APCER (%)</th>
<th align="center" valign="top">BPCER (%)</th>
<th align="center" valign="top">ACER (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" rowspan="2">I</td>
<td align="left" valign="top">AlexNet</td>
<td align="center" valign="middle">0.94</td>
<td align="center" valign="middle">79.90</td>
<td align="center" valign="middle">40.42</td>
</tr>
<tr>
<td align="left" valign="top">AlexNet+our</td>
<td align="center" valign="middle">0.01</td>
<td align="center" valign="middle">63.19</td>
<td align="center" valign="middle">31.60</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="2">II</td>
<td align="left" valign="top">AlexNet</td>
<td align="center" valign="middle">14.46</td>
<td align="center" valign="middle">6.78</td>
<td align="center" valign="middle">10.62</td>
</tr>
<tr>
<td align="left" valign="top">AlexNet+our</td>
<td align="center" valign="middle">5.06</td>
<td align="center" valign="middle">10.46</td>
<td align="center" valign="middle">7.76</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="2">III</td>
<td align="left" valign="top">AlexNet</td>
<td align="center" valign="middle">3.40&#x2009;&#x00B1;&#x2009;2.98</td>
<td align="center" valign="middle">11.56&#x2009;&#x00B1;&#x2009;7.58</td>
<td align="center" valign="middle">7.17&#x2009;&#x00B1;&#x2009;3.72</td>
</tr>
<tr>
<td align="left" valign="top">AlexNet+our</td>
<td align="center" valign="middle">2.33&#x2009;&#x00B1;&#x2009;2.33</td>
<td align="center" valign="middle">9.75&#x2009;&#x00B1;&#x2009;5.25</td>
<td align="center" valign="middle">6.04&#x2009;&#x00B1;&#x2009;1.45</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="2">IV</td>
<td align="left" valign="top">AlexNet</td>
<td align="center" valign="middle">9.07&#x2009;&#x00B1;&#x2009;9.07</td>
<td align="center" valign="middle">58.87&#x2009;&#x00B1;&#x2009;33.87</td>
<td align="center" valign="middle">32.84&#x2009;&#x00B1;&#x2009;16.00</td>
</tr>
<tr>
<td align="left" valign="top">AlexNet+our</td>
<td align="center" valign="middle">3.53&#x2009;&#x00B1;&#x2009;3.53</td>
<td align="center" valign="middle">55.88&#x2009;&#x00B1;&#x2009;25.88</td>
<td align="center" valign="middle">29.71&#x2009;&#x00B1;&#x2009;11.17</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>The performance on the MSU-FASD dataset.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Model</th>
<th align="center" valign="top">APCER (%)</th>
<th align="center" valign="top">BPCER (%)</th>
<th align="center" valign="top">ACER (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">AlexNet</td>
<td align="center" valign="top">1.47</td>
<td align="center" valign="top">5.27</td>
<td align="center" valign="top">3.37</td>
</tr>
<tr>
<td align="left" valign="top">AlexNet+our</td>
<td align="center" valign="top">1.39</td>
<td align="center" valign="top">3.99</td>
<td align="center" valign="top">2.69</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec13">
<label>5.2</label>
<title>Experimental discussion</title>
<p>The experiment evaluated the performance of the intra-test and inter-test. Specifically, the training and testing were performed on the same dataset, which can reflect the performance of the algorithm; cross-datasets indicate that the training set and test set are from different data sets, and the test on these datasets can usually reflect the generalization ability of the algorithm.</p>
<p>The experiments first compared the results of fusing different features on two datasets, followed by comparing the results of different fusion methods on two datasets, then compared the proposed method with some popular methods, and finally evaluated performance across databases on two datasets. The experimental results demonstrated the effectiveness of the proposed face detection method in face anti-spoofing.</p>
<p>The following four experiments were set for comparison in <xref ref-type="table" rid="tab3">Table 3</xref>. Since there are four protocols in the OULU-NPU dataset, protocol 2 was selected based on the features of the MSU-MFSD dataset.</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Comparison of the experimental results.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top" rowspan="2">Experiment</th>
<th align="center" valign="top" colspan="3">MSU-MFSD</th>
<th align="center" valign="top" colspan="3">OULU-NPU</th>
</tr>
<tr>
<th align="center" valign="top">APCER (%)</th>
<th align="center" valign="top">BPCER (%)</th>
<th align="center" valign="top">ACER (%)</th>
<th align="center" valign="top">APCER (%)</th>
<th align="center" valign="top">BPCER (%)</th>
<th align="center" valign="top">ACER (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">1</td>
<td align="center" valign="top">1.47</td>
<td align="center" valign="top">5.27</td>
<td align="center" valign="top">3.37</td>
<td align="center" valign="middle">14.46</td>
<td align="center" valign="middle">6.78</td>
<td align="center" valign="middle">10.62</td>
</tr>
<tr>
<td align="left" valign="middle">2</td>
<td align="center" valign="top">1.39</td>
<td align="center" valign="top">3.99</td>
<td align="center" valign="top">2.69</td>
<td align="center" valign="middle">5.06</td>
<td align="center" valign="middle">10.46</td>
<td align="center" valign="middle">7.76</td>
</tr>
<tr>
<td align="left" valign="middle">3</td>
<td align="center" valign="middle">20.71</td>
<td align="center" valign="middle">65.23</td>
<td align="center" valign="middle">42.97</td>
<td align="center" valign="middle">25.36</td>
<td align="center" valign="middle">45.82</td>
<td align="center" valign="middle">35.59</td>
</tr>
<tr>
<td align="left" valign="middle">4</td>
<td align="center" valign="middle">20.07</td>
<td align="center" valign="middle">65.97</td>
<td align="center" valign="middle">43.02</td>
<td align="center" valign="middle">7.29</td>
<td align="center" valign="middle">35.41</td>
<td align="center" valign="middle">21.35</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Experiment 1: AlexNet networks without the pseudo-negative feature generator were tested with an intra-test on the OULU-NPU and MSU-MFSD datasets.</p>
<p>Experiment 2: AlexNet networks with the pseudo-negative feature generator were tested with an intra-test on the OULU-NPU and MSU-MFSD datasets.</p>
<p>Experiment 3: AlexNet networks without the pseudo-negative feature generator were tested with an inter-test on the OULU-NPU and MSU-MFSD datasets.</p>
<p>Experiment 4: AlexNet networks with the pseudo-negative feature generator were tested with an inter-test on the OULU-NPU and MSU-MFSD datasets.</p>
<p>To evaluate the effectiveness of our method, in <xref ref-type="table" rid="tab4">Table 4</xref>, the OULU-NPU dataset was used to train and test the AlexNet and AlexNet+our (AlexNet network using the pseudo-feature generator), respectively, and the performance evaluation metrics were calculated. The results indicated that the proposed method achieved comparable performance with state-of-the-art methods (LBP&#x2009;+&#x2009;SVM, GRADIANT, and MILHP). We tested our model on the Replay-Attack dataset, as shown in <xref ref-type="table" rid="tab5">Table 5</xref>. Compared with the state-of-the-art methods from the past 3 years (RGB+LBP and multilevel+ELBP), our model achieved superior performance in terms of accuracy and other evaluation metrics.</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>Comparable performance on the OULU-NPU dataset.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Protocol</th>
<th align="left" valign="top">Model</th>
<th align="center" valign="top">APCER (%)</th>
<th align="center" valign="top">BPCER (%)</th>
<th align="center" valign="top">ACER (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" rowspan="5">I</td>
<td align="left" valign="top">LBP+SVM (<xref ref-type="bibr" rid="ref13">George and Marcel, 2019</xref>)</td>
<td align="center" valign="middle">12.9</td>
<td align="center" valign="middle">51.7</td>
<td align="center" valign="middle">32.3</td>
</tr>
<tr>
<td align="left" valign="top">GRADIANT (<xref ref-type="bibr" rid="ref4">Boulkenafet et al., 2017b</xref>)</td>
<td align="center" valign="middle">1.3</td>
<td align="center" valign="middle">12.5</td>
<td align="center" valign="middle">6.9</td>
</tr>
<tr>
<td align="left" valign="top">MILHP (<xref ref-type="bibr" rid="ref21">Lin et al., 2018</xref>)</td>
<td align="center" valign="middle">8.3</td>
<td align="center" valign="middle">0.8</td>
<td align="center" valign="middle">4.6</td>
</tr>
<tr>
<td align="left" valign="top">AlexNet</td>
<td align="center" valign="middle">0.9</td>
<td align="center" valign="middle">79.9</td>
<td align="center" valign="middle">40.4</td>
</tr>
<tr>
<td align="left" valign="top">AlexNet+our</td>
<td align="center" valign="middle">0.0</td>
<td align="center" valign="middle">63.2</td>
<td align="center" valign="middle">31.6</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="5">II</td>
<td align="left" valign="top">LBP+SVM (<xref ref-type="bibr" rid="ref13">George and Marcel, 2019</xref>)</td>
<td align="center" valign="middle">30.0</td>
<td align="center" valign="middle">20.3</td>
<td align="center" valign="middle">25.1</td>
</tr>
<tr>
<td align="left" valign="top">GRADIANT (<xref ref-type="bibr" rid="ref4">Boulkenafet et al., 2017b</xref>)</td>
<td align="center" valign="middle">3.1</td>
<td align="center" valign="middle">1.9</td>
<td align="center" valign="middle">2.5</td>
</tr>
<tr>
<td align="left" valign="top">MILHP (<xref ref-type="bibr" rid="ref21">Lin et al., 2018</xref>)</td>
<td align="center" valign="middle">5.6</td>
<td align="center" valign="middle">5.3</td>
<td align="center" valign="middle">5.4</td>
</tr>
<tr>
<td align="left" valign="top">AlexNet</td>
<td align="center" valign="middle">14.5</td>
<td align="center" valign="middle">6.8</td>
<td align="center" valign="middle">10.6</td>
</tr>
<tr>
<td align="left" valign="top">AlexNet+our</td>
<td align="center" valign="middle">5.06</td>
<td align="center" valign="middle">10.46</td>
<td align="center" valign="middle">7.76</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="5">III</td>
<td align="left" valign="top">LBP+SVM (<xref ref-type="bibr" rid="ref13">George and Marcel, 2019</xref>)</td>
<td align="center" valign="middle">28.5&#x2009;&#x00B1;&#x2009;23.1</td>
<td align="center" valign="middle">23.3&#x2009;&#x00B1;&#x2009;18.0</td>
<td align="center" valign="middle">25.9&#x2009;&#x00B1;&#x2009;11.3</td>
</tr>
<tr>
<td align="left" valign="top">GRADIANT (<xref ref-type="bibr" rid="ref4">Boulkenafet et al., 2017b</xref>)</td>
<td align="center" valign="middle">2.6&#x2009;&#x00B1;&#x2009;3.9</td>
<td align="center" valign="middle">5.0&#x2009;&#x00B1;&#x2009;5.3</td>
<td align="center" valign="middle">3.8&#x2009;&#x00B1;&#x2009;2.4</td>
</tr>
<tr>
<td align="left" valign="top">MILHP (<xref ref-type="bibr" rid="ref21">Lin et al., 2018</xref>)</td>
<td align="center" valign="middle">1.5&#x2009;&#x00B1;&#x2009;1.2</td>
<td align="center" valign="middle">6.4&#x2009;&#x00B1;&#x2009;6.6</td>
<td align="center" valign="middle">4.0&#x2009;&#x00B1;&#x2009;2.9</td>
</tr>
<tr>
<td align="left" valign="top">AlexNet</td>
<td align="center" valign="middle">3.4&#x2009;&#x00B1;&#x2009;3.0</td>
<td align="center" valign="middle">11.6&#x2009;&#x00B1;&#x2009;7.6</td>
<td align="center" valign="middle">7.2&#x2009;&#x00B1;&#x2009;3.7</td>
</tr>
<tr>
<td align="left" valign="top">AlexNet+our</td>
<td align="center" valign="middle">2.3&#x2009;&#x00B1;&#x2009;2.3</td>
<td align="center" valign="middle">9.8&#x2009;&#x00B1;&#x2009;5.3</td>
<td align="center" valign="middle">6.0&#x2009;&#x00B1;&#x2009;1.5</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="5">IV</td>
<td align="left" valign="top">LBP+SVM (<xref ref-type="bibr" rid="ref13">George and Marcel, 2019</xref>)</td>
<td align="center" valign="middle">41.67&#x2009;&#x00B1;&#x2009;27.03</td>
<td align="center" valign="middle">55&#x2009;&#x00B1;&#x2009;21.21</td>
<td align="center" valign="middle">48.33&#x2009;&#x00B1;&#x2009;6.07</td>
</tr>
<tr>
<td align="left" valign="top">GRADIANT (<xref ref-type="bibr" rid="ref4">Boulkenafet et al., 2017b</xref>)</td>
<td align="center" valign="middle">5.0&#x2009;&#x00B1;&#x2009;4.5</td>
<td align="center" valign="middle">15.0&#x2009;&#x00B1;&#x2009;7.1</td>
<td align="center" valign="middle">10.0&#x2009;&#x00B1;&#x2009;5</td>
</tr>
<tr>
<td align="left" valign="top">MILHP (<xref ref-type="bibr" rid="ref21">Lin et al., 2018</xref>)</td>
<td align="center" valign="middle">15.8&#x2009;&#x00B1;&#x2009;12.8</td>
<td align="center" valign="middle">8.3&#x2009;&#x00B1;&#x2009;15.7</td>
<td align="center" valign="middle">12.0&#x2009;&#x00B1;&#x2009;6.2</td>
</tr>
<tr>
<td align="left" valign="top">AlexNet</td>
<td align="center" valign="middle">9.1&#x2009;&#x00B1;&#x2009;9.1</td>
<td align="center" valign="middle">58.9&#x2009;&#x00B1;&#x2009;33.9</td>
<td align="center" valign="middle">32.8&#x2009;&#x00B1;&#x2009;16.0</td>
</tr>
<tr>
<td align="left" valign="top">AlexNet+our</td>
<td align="center" valign="middle">3.5&#x2009;&#x00B1;&#x2009;3.5</td>
<td align="center" valign="middle">55.9&#x2009;&#x00B1;&#x2009;25.9</td>
<td align="center" valign="middle">29.7&#x2009;&#x00B1;&#x2009;11.2</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap position="float" id="tab5">
<label>Table 5</label>
<caption>
<p>Comparable performance on the Replay-Attack dataset.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Model</th>
<th align="center" valign="top">HTER(%)</th>
<th align="center" valign="top">EER(%)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">RGB+LBP (<xref ref-type="bibr" rid="ref2">Antil and Dhiman, 2023</xref>)</td>
<td align="center" valign="middle">4.58</td>
<td align="center" valign="middle">9.69</td>
</tr>
<tr>
<td align="left" valign="top">Multilevel+ELBP (<xref ref-type="bibr" rid="ref1">Antil and Dhiman, 2022</xref>)</td>
<td align="center" valign="middle">0.00</td>
<td align="center" valign="middle">0.00</td>
</tr>
<tr>
<td align="left" valign="top">Dropblock (<xref ref-type="bibr" rid="ref47">Wu et al., 2021</xref>)</td>
<td align="center" valign="middle">0.29</td>
<td align="center" valign="middle">0.00</td>
</tr>
<tr>
<td align="left" valign="top">Our</td>
<td align="center" valign="middle">0.00</td>
<td align="center" valign="middle">0.00</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As shown in <xref ref-type="table" rid="tab3">Table 3</xref>, the APCER of the AlexNet using a pseudo-negative feature generator decreased significantly on both within-set and cross-set tests, and BPCER also decreased, with only a few parts increasing slightly. The comparison results in <xref ref-type="table" rid="tab4">Table 4</xref> show that on the OULU-NPU dataset, the performance of AlexNet is not outstanding, and there is a significant performance gap with the mainstream methods. In contrast, the AlexNet using a pseudo-negative feature generator showed good performance in training and testing. The APCER and BPCER were significantly improved compared with those of AlexNet, and they were close to the performance evaluation indicators of mainstream methods.</p>
<p>To test the model&#x2019;s generalization performance, cross-dataset testing was conducted on the MSU-MFSD dataset (referred to as M), OULU-NPU dataset (referred to as O), Replay-Attack dataset (referred to as R), and CASIA-FASD dataset (referred to as C; <xref ref-type="bibr" rid="ref52">Zhang et al., 2012</xref>). Then, the results were compared with those of other mainstream experiments, as shown in <xref ref-type="table" rid="tab6">Table 6</xref>. To further verify the performance of the model, we reduced the data set used for training. The experimental results are shown in <xref ref-type="table" rid="tab7">Table 7</xref>. From <xref ref-type="table" rid="tab7">Table 7</xref>, it can be observed that, when using a smaller dataset, our method can achieve results close to or even surpass those obtained from training on larger datasets.</p>
<table-wrap position="float" id="tab6">
<label>Table 6</label>
<caption>
<p>Comparison of the results between our experiment and the state-of-the-art in cross-domain face anti-spoofing detection.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top" rowspan="2">Methods</th>
<th align="center" valign="top" colspan="2">O&#x0026;C&#x0026;R-to-M</th>
<th align="center" valign="top" colspan="2">O&#x0026;M&#x0026;R-to-C</th>
<th align="center" valign="top" colspan="2">O&#x0026;C&#x0026;M-to-R</th>
<th align="center" valign="top" colspan="2">R&#x0026;C&#x0026;M-to-O</th>
</tr>
<tr>
<th align="center" valign="top">ACER(%)</th>
<th align="center" valign="top">AUC(%)</th>
<th align="center" valign="top">ACER(%)</th>
<th align="center" valign="top">AUC(%)</th>
<th align="center" valign="top">ACER(%)</th>
<th align="center" valign="top">AUC(%)</th>
<th align="center" valign="top">ACER(%)</th>
<th align="center" valign="top">AUC(%)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">MADDG (<xref ref-type="bibr" rid="ref36">Shao et al., 2019</xref>)</td>
<td align="center" valign="middle">17.69</td>
<td align="center" valign="middle">88.06</td>
<td align="center" valign="middle">24.50</td>
<td align="center" valign="middle">84.51</td>
<td align="center" valign="middle">22.19</td>
<td align="center" valign="middle">84.99</td>
<td align="center" valign="middle">27.89</td>
<td align="center" valign="middle">80.02</td>
</tr>
<tr>
<td align="left" valign="middle">ANRL (<xref ref-type="bibr" rid="ref27">Liu et al., 2021b</xref>)</td>
<td align="center" valign="middle">10.83</td>
<td align="center" valign="middle">96.75</td>
<td align="center" valign="middle">17.85</td>
<td align="center" valign="middle">89.26</td>
<td align="center" valign="middle">16.03</td>
<td align="center" valign="middle">91.04</td>
<td align="center" valign="middle">15.67</td>
<td align="center" valign="middle">91.90</td>
</tr>
<tr>
<td align="left" valign="middle">SSAN (<xref ref-type="bibr" rid="ref44">Wang et al., 2022b</xref>)</td>
<td align="center" valign="middle">6.67</td>
<td align="center" valign="middle">98.75</td>
<td align="center" valign="middle">10.00</td>
<td align="center" valign="middle">96.67</td>
<td align="center" valign="middle">8.88</td>
<td align="center" valign="middle">96.79</td>
<td align="center" valign="middle">13.72</td>
<td align="center" valign="middle">93.63</td>
</tr>
<tr>
<td align="left" valign="middle">Our</td>
<td align="center" valign="middle">7.12</td>
<td align="center" valign="middle">98.06</td>
<td align="center" valign="middle">11.54</td>
<td align="center" valign="middle">99.21</td>
<td align="center" valign="middle">3.88</td>
<td align="center" valign="middle">98.17</td>
<td align="center" valign="middle">8.36</td>
<td align="center" valign="middle">98.78</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap position="float" id="tab7">
<label>Table 7</label>
<caption>
<p>Comparative cross-dataset testing results for similar models.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Experiment</th>
<th align="left" valign="top">Model</th>
<th align="center" valign="top">Train(videos)</th>
<th align="center" valign="top">HTER(%)</th>
<th align="center" valign="top">AUC(%)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">M to R</td>
<td align="left" valign="top">Multilevel+ELBP (<xref ref-type="bibr" rid="ref1">Antil and Dhiman, 2022</xref>)</td>
<td align="center" valign="middle">280</td>
<td align="center" valign="middle">24.3</td>
<td align="center" valign="middle">-</td>
</tr>
<tr>
<td align="left" valign="top">M to R</td>
<td align="left" valign="top">Our</td>
<td align="center" valign="middle">280</td>
<td align="center" valign="middle">21.10</td>
<td align="center" valign="middle">92.36</td>
</tr>
<tr>
<td align="left" valign="top">R&#x0026;M to O</td>
<td align="left" valign="top">SSDG (<xref ref-type="bibr" rid="ref15">Jia et al., 2020</xref>)</td>
<td align="center" valign="middle">1,480</td>
<td align="center" valign="middle">36.01</td>
<td align="center" valign="middle">66.88</td>
</tr>
<tr>
<td align="left" valign="top">R&#x0026;M to O</td>
<td align="left" valign="top">D<sup>2</sup>AN (<xref ref-type="bibr" rid="ref7">Chen et al., 2021</xref>)</td>
<td align="center" valign="middle">1,480</td>
<td align="center" valign="middle">27.70</td>
<td align="center" valign="middle">75.36</td>
</tr>
<tr>
<td align="left" valign="top">R&#x0026;M to O</td>
<td align="left" valign="top">DRDG (<xref ref-type="bibr" rid="ref28">Liu et al., 2021a</xref>)</td>
<td align="center" valign="middle">1,480</td>
<td align="center" valign="middle">33.35</td>
<td align="center" valign="middle">69.14</td>
</tr>
<tr>
<td align="left" valign="top">R&#x0026;M to O</td>
<td align="left" valign="top">ANRL (<xref ref-type="bibr" rid="ref27">Liu et al., 2021b</xref>)</td>
<td align="center" valign="middle">1,480</td>
<td align="center" valign="middle">30.73</td>
<td align="center" valign="middle">74.10</td>
</tr>
<tr>
<td align="left" valign="top">R&#x0026;M to O</td>
<td align="left" valign="top">SSAN (<xref ref-type="bibr" rid="ref44">Wang et al., 2022b</xref>)</td>
<td align="center" valign="middle">1,480</td>
<td align="center" valign="middle">29.44</td>
<td align="center" valign="middle">76.62</td>
</tr>
<tr>
<td align="left" valign="middle">M to O</td>
<td align="left" valign="top">Our</td>
<td align="center" valign="middle">280</td>
<td align="center" valign="middle">26.24</td>
<td align="center" valign="middle">83.77</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec14">
<label>5.3</label>
<title>Feature distribution</title>
<p>The feature visualization algorithm was utilized to extract and compute the features of the training images, whose cosine distance is depicted in <xref ref-type="fig" rid="fig10">Figure 10</xref>. Specifically, <xref ref-type="fig" rid="fig10">Figure 10A</xref> presents the distance between the attack and the <italic>bona fide</italic> samples in the training phase. It can be seen that there is a large distance between the <italic>bona fide</italic> samples and the attack samples, and there are many blank unknown regions between the two types of samples. Since the face anti-spoofing system in practical applications may encounter some new attack data that did not appear in training, this paper generated false negative samples between the <italic>bona fide</italic> and attack samples. As shown in <xref ref-type="fig" rid="fig10">Figure 10B</xref>, the pseudo-negative samples are closer to the <italic>bona fide</italic> samples, indicating that the classification boundary of the face anti-spoofing system, during training, is more biased toward the <italic>bona fide</italic> samples. In practical applications, the face anti-spoofing system can achieve a good identification effect for new attacks that have not appeared in the dataset.</p>
<fig position="float" id="fig10">
<label>Figure 10</label>
<caption>
<p>The feature cosine distance of images during training. <bold>(A)</bold> The training without using pseudo-negative features, <bold>(B)</bold> the training using pseudo-negative features.</p>
</caption>
<graphic xlink:href="fnins-18-1362286-g010.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="conclusions" id="sec15">
<label>6</label>
<title>Conclusion</title>
<p>In this paper, a face anti-spoofing algorithm is proposed based on generated pseudo-negative features. Through continuous iteration, the original face anti-spoofing system achieves higher accuracy and robustness. Meanwhile, by adding pseudo-negative features, good results have been obtained in detecting attack samples. It shows that adding pseudo-negative class features enables the model to detect negative samples, and this affects the detection of positive examples in some cases. In this study, by constantly adjusting the strategy, new features are continually generated based on the image&#x2019;s original features. Concurrently, a face anti-spoofing system is devised to counter emerging attacks within the feature space, resulting in the development of more effective strategies. Furthermore, this study promotes aggregation among <italic>bona fide</italic> examples while increasing scatter among attack examples, consequently bolstering the model&#x2019;s robustness in unfamiliar territories. In future work, we will focus on eliminating the influence on positive examples to improve their detection effect.</p>
</sec>
<sec sec-type="data-availability" id="sec16">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found at: Replay-Attack Database: <ext-link xlink:href="https://www.idiap.ch/dataset/replayattack" ext-link-type="uri">https://www.idiap.ch/dataset/replayattack</ext-link>, MSU-MFSD Database: <ext-link xlink:href="http://biometricscse.msu.edu/Publications/Databases/MSUMobileFaceSpoofing" ext-link-type="uri">http://biometricscse.msu.edu/Publications/Databases/MSUMobileFaceSpoofing</ext-link>, and Oulu-NPU Database: <ext-link xlink:href="https://sites.google.com/site/oulunpudatabase" ext-link-type="uri">https://sites.google.com/site/oulunpudatabase</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="sec17">
<title>Author contributions</title>
<p>YM: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing, Conceptualization, Data curation, Formal analysis, Funding acquisition, Methodology, Resources, Supervision, Validation. CL: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing, Data curation, Formal analysis, Investigation, Methodology, Software, Supervision, Validation. LL: Investigation, Resources, Supervision, Validation, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. YW: Data curation, Investigation, Software, Supervision, Validation, Writing &#x2013; review &#x0026; editing. YX: Formal analysis, Investigation, Resources, Supervision, Validation, Writing &#x2013; review &#x0026; editing.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="sec18">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This research was funded by Training Program for Young Backbone Teachers of Higher Education Institutions in Henan Province China &#x201C;Research on shadow adversarial sample attack method combining optimal point constraints&#x201D; (no. 2023GGJS116) and Henan Provincial Scientific and Technological Research Project &#x201C;Research on Cross-domain Face Anti-spoofing Technology for Diversified Attacks&#x201D; (242102210128).</p>
</sec>
<sec sec-type="COI-statement" id="sec19">
<title>Conflict of interest</title>
<p>YX was employed by China Telecom Corporation Ltd.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="sec100" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Antil</surname> <given-names>A.</given-names></name> <name><surname>Dhiman</surname> <given-names>C.</given-names></name></person-group> (<year>2022</year>). Two stream RGB-LBP based transfer learning model for face anti-spoofing. In: <italic>International Conference on Computer Vision Image Processing</italic>. pp. 364&#x2013;374. Cham: Springer Nature Switzerland.</citation></ref>
<ref id="ref2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Antil</surname> <given-names>A.</given-names></name> <name><surname>Dhiman</surname> <given-names>C.</given-names></name></person-group> (<year>2023</year>). <article-title>A two stream face anti-spoofing framework using multi-level deep features and ELBP features</article-title>. <source>Multimedia Systems</source> <volume>29</volume>, <fpage>1</fpage>&#x2013;<lpage>16</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s00530-023-01060-7</pub-id></citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Antil</surname> <given-names>A.</given-names></name> <name><surname>Dhiman</surname> <given-names>C.</given-names></name></person-group> (<year>2024</year>). <article-title>MF2ShrT: multi-modal feature fusion using shared layered transformer for face anti-spoofing</article-title>. <source>ACM Trans. Multimedia Comput. Commun. Appl.</source> <volume>20</volume>, <fpage>1</fpage>&#x2013;<lpage>21</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3640817</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Boulkenafet</surname> <given-names>Z.</given-names></name> <name><surname>Komulainen</surname> <given-names>J.</given-names></name> <name><surname>Akhtar</surname> <given-names>Z.</given-names></name> <name><surname>Benlamoudi</surname> <given-names>A.</given-names></name> <name><surname>Samai</surname> <given-names>D.</given-names></name> <name><surname>Bekhouche</surname> <given-names>S. E.</given-names></name> <etal/></person-group>. (<year>2017b</year>). A competition on generalized software-based face presentation attack detection in mobile scenarios. In: <italic>Proceedings of International Joint Conference on Biometrics</italic>. pp. 688&#x2013;696.</citation></ref>
<ref id="ref5"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Boulkenafet</surname> <given-names>Z.</given-names></name> <name><surname>Komulainen</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>L.</given-names></name> <name><surname>Feng</surname> <given-names>X.</given-names></name> <name><surname>Hadid</surname> <given-names>A.</given-names></name></person-group> (<year>2017a</year>). OULU-NPU: a mobile face presentation attack database with real-world variations. In: <italic>IEEE International Conference on Automatic Face &#x0026; Gesture Recognition</italic> pp. 612&#x2013;618.</citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cai</surname> <given-names>R.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Wan</surname> <given-names>R.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Hu</surname> <given-names>Y.</given-names></name> <name><surname>Kot</surname> <given-names>A. C.</given-names></name></person-group> (<year>2022</year>). <article-title>Learning meta pattern for face anti-spoofing</article-title>. <source>IEEE Trans. Inf. Forensics Security.</source> <volume>17</volume>, <fpage>1201</fpage>&#x2013;<lpage>1213</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TIFS.2022.3158551</pub-id>, PMID: <pub-id pub-id-type="pmid">34156938</pub-id></citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Yao</surname> <given-names>T.</given-names></name> <name><surname>Sheng</surname> <given-names>K.</given-names></name> <name><surname>Ding</surname> <given-names>S.</given-names></name> <name><surname>Tai</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Generalizable representation learning for mixture domain face anti-spoofing</article-title>. <source>AAAI Conf. Artif. Intell.</source> <volume>35</volume>, <fpage>1132</fpage>&#x2013;<lpage>1139</lpage>.</citation></ref>
<ref id="ref8"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Chingovska</surname> <given-names>I.</given-names></name> <name><surname>Anjos</surname> <given-names>A.</given-names></name> <name><surname>Marcel</surname> <given-names>S.</given-names></name></person-group> (<year>2012</year>). On the effectiveness of local binary patterns in face anti-spoofing. In: <italic>Proceedings of the International Conference of Biometrics Special Interest Group (BIOSIG)</italic>. pp. 1&#x2013;7.</citation></ref>
<ref id="ref9"><citation citation-type="other"><person-group person-group-type="author"><name><surname>de Freitas Pereira</surname> <given-names>T.</given-names></name> <name><surname>Anjos</surname> <given-names>A.</given-names></name> <name><surname>De Martino</surname> <given-names>J. M.</given-names></name> <name><surname>Marcel</surname> <given-names>S.</given-names></name></person-group> (<year>2013</year>) Can face anti-spoofing countermeasures work in a real world scenario?. In: <italic>IEEE International Conference on Biometrics</italic> pp. 1&#x2013;8.</citation></ref>
<ref id="ref10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>de Freitas Pereira</surname> <given-names>T.</given-names></name> <name><surname>Komulainen</surname> <given-names>J.</given-names></name> <name><surname>Anjos</surname> <given-names>A.</given-names></name> <name><surname>De Martino</surname> <given-names>J. M.</given-names></name> <name><surname>Hadid</surname> <given-names>A.</given-names></name> <name><surname>Pietik&#x00E4;inen</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Face liveness detection using dynamic texture</article-title>. <source>EURASIP J. Image Video Process.</source> <volume>2014</volume>:<fpage>2</fpage>. doi: <pub-id pub-id-type="doi">10.1186/1687-5281-2014-2</pub-id></citation></ref>
<ref id="ref11"><citation citation-type="other"><person-group person-group-type="author"><name><surname>De Marsico</surname> <given-names>M.</given-names></name> <name><surname>Nappi</surname> <given-names>M.</given-names></name> <name><surname>Riccio</surname> <given-names>D.</given-names></name> <name><surname>Dugelay</surname> <given-names>J. L.</given-names></name></person-group> (<year>2012</year>) Moving face spoofing detection via 3D projective invariants. In: <italic>IAPR International Conference on Biometrics (ICB)</italic> pp. 73&#x2013;78.</citation></ref>
<ref id="ref12"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Dong</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Cai</surname> <given-names>W.</given-names></name> <name><surname>Lv</surname> <given-names>P.</given-names></name> <name><surname>Yu</surname> <given-names>Z.</given-names></name></person-group> (<year>2021</year>) Open set face anti-spoofing in unseen attacks. In: <italic>ACM International Conference on Multimedia</italic>. pp. 4082&#x2013;4090.</citation></ref>
<ref id="ref13"><citation citation-type="other"><person-group person-group-type="author"><name><surname>George</surname> <given-names>A.</given-names></name> <name><surname>Marcel</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). Deep pixel-wise binary supervision for face presentation attack detection. In: <italic>Proceedings of IEEE International Conference on Biometrics</italic>. pp. 1&#x2013;8.</citation></ref>
<ref id="ref14"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>H. P.</given-names></name> <name><surname>Sun</surname> <given-names>D.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Chu</surname> <given-names>W. S.</given-names></name> <name><surname>Xiao</surname> <given-names>T.</given-names></name> <name><surname>Yuan</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2022</year>). <source>Adaptive transformers for robust few-shot cross-domain face anti-spoofing</source>. <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer Nature Switzerland</publisher-name>. Pp. <fpage>37</fpage>&#x2013;<lpage>54</lpage>.</citation></ref>
<ref id="ref15"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Jia</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Shan</surname> <given-names>S.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name></person-group> (<year>2020</year>). Single-side domain generalization for face anti-spoofing. In: <italic>IEEE Conference on Computer Vision Pattern Recognition</italic>. pp. 8484&#x2013;8493.</citation></ref>
<ref id="ref16"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>T.</given-names></name> <name><surname>Kim</surname> <given-names>Y.</given-names></name> <name><surname>Kim</surname> <given-names>I.</given-names></name> <name><surname>Kim</surname> <given-names>D.</given-names></name></person-group> (<year>2019</year>). Basn: enriching feature representation using bipartite auxiliary supervisions for face anti-spoofing. In: <italic>IEEE/CVF International Conference on Computer Vision Workshop (ICCVW)</italic>.</citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krizhevsky</surname> <given-names>A.</given-names></name> <name><surname>Sutskever</surname> <given-names>I.</given-names></name> <name><surname>Hinton</surname> <given-names>G. E.</given-names></name></person-group> (<year>2012</year>). <article-title>ImageNet classification with deep convolutional neural networks</article-title>. <source>Commun. ACM</source> <volume>25</volume>, <fpage>84</fpage>&#x2013;<lpage>90</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3065386</pub-id></citation></ref>
<ref id="ref18"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>L.</given-names></name> <name><surname>Feng</surname> <given-names>X.</given-names></name> <name><surname>Boulkenafez</surname> <given-names>Z.</given-names></name> <name><surname>Xia</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>M.</given-names></name> <name><surname>Hadid</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). An original face anti-spoofing approach using partial convolutional neural network. In: <italic>Sixth International Conference on Image Processing Theory, Tools and Applications (IPTA)</italic> pp. 1&#x2013;6.</citation></ref>
<ref id="ref19"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Lam</surname> <given-names>K. Y.</given-names></name> <name><surname>Kot</surname> <given-names>A. C.</given-names></name></person-group> (<year>2020</year>). Unseen face presentation attack detection with hypersphere loss. In: <italic>IEEE International Conference on Acoustics Speech Signal Processing</italic>. pp. 2852&#x2013;2856.</citation></ref>
<ref id="ref20"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Liao</surname> <given-names>C. H.</given-names></name> <name><surname>Chen</surname> <given-names>W. C.</given-names></name> <name><surname>Liu</surname> <given-names>H. T.</given-names></name> <name><surname>Yeh</surname> <given-names>Y. R.</given-names></name> <name><surname>Hu</surname> <given-names>M. C.</given-names></name> <name><surname>Chen</surname> <given-names>C. S.</given-names></name></person-group> (<year>2023</year>). Domain invariant vision transformer learning for face anti-spoofing. In: <italic>Proceedings of IEEE Winter Conference on Applications of Computer Vision</italic> pp. 6098&#x2013;6107.</citation></ref>
<ref id="ref21"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>C.</given-names></name> <name><surname>Liao</surname> <given-names>Z.</given-names></name> <name><surname>Zhou</surname> <given-names>P.</given-names></name> <name><surname>Hu</surname> <given-names>J.</given-names></name> <name><surname>Ni</surname> <given-names>B.</given-names></name></person-group> (<year>2018</year>). Live face verification with multiple Instantialized local homographic parameterization. In: <italic>IJCAI International Joint Conference on Artificial Intelligence</italic>. pp. 814&#x2013;820.</citation></ref>
<ref id="ref22"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Dai</surname> <given-names>W.</given-names></name> <name><surname>Gou</surname> <given-names>M.</given-names></name> <name><surname>Huang</surname> <given-names>C. T.</given-names></name> <name><surname>Xiong</surname> <given-names>H.</given-names></name></person-group> (<year>2022a</year>). <source>Source-free domain adaptation with contrastive domain alignment and self-supervised exploration for face anti-spoofing</source>. <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer Nature Switzerland</publisher-name>. Pp. <fpage>511</fpage>&#x2013;<lpage>528</lpage></citation></ref>
<ref id="ref23"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Dai</surname> <given-names>W.</given-names></name> <name><surname>Li</surname> <given-names>C.</given-names></name> <name><surname>Zou</surname> <given-names>J.</given-names></name> <name><surname>Xiong</surname> <given-names>H.</given-names></name></person-group> (<year>2022b</year>). Causal intervention for generalizable face anti-spoofing. In: <italic>IEEE International Conference on Multimedia Expo</italic>. pp. 01&#x2013;06.</citation></ref>
<ref id="ref24"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Gou</surname> <given-names>M.</given-names></name> <name><surname>Huang</surname> <given-names>C. T.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Dai</surname> <given-names>W.</given-names></name></person-group> (<year>2023</year>). Towards unsupervised domain generalization for face anti-spoofing. In: <italic>Proceedings of IEEE International Conference on Computer Vision</italic>. pp. 20654&#x2013;20664.</citation></ref>
<ref id="ref25"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Jourabloo</surname> <given-names>A.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name></person-group> (<year>2018</year>). Learning deep models for face anti-spoofing: binary or auxiliary supervision. In: <italic>IEEE/CVF Conference on Computer Vision and Pattern Recognition</italic> 389&#x2013;398.</citation></ref>
<ref id="ref26"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Stehouwer</surname> <given-names>J.</given-names></name> <name><surname>Jourabloo</surname> <given-names>A.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name></person-group> (<year>2019</year>). Deep tree learning for zero-shot face anti-spoofing. In: <italic>IEEE Conference on Computer Vision Pattern Recognition</italic>. pp. 4680&#x2013;4689.</citation></ref>
<ref id="ref27"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>S.</given-names></name> <name><surname>Zhang</surname> <given-names>K. Y.</given-names></name> <name><surname>Yao</surname> <given-names>T.</given-names></name> <name><surname>Bi</surname> <given-names>M.</given-names></name> <name><surname>Ding</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2021b</year>). Adaptive normalized representation learning for generalizable face anti-spoofing. In: <italic>ACM International Conference on Multimedia</italic>. pp. 1469&#x2013;1477.</citation></ref>
<ref id="ref28"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>S.</given-names></name> <name><surname>Zhang</surname> <given-names>K. Y.</given-names></name> <name><surname>Yao</surname> <given-names>T.</given-names></name> <name><surname>Sheng</surname> <given-names>K.</given-names></name> <name><surname>Ding</surname> <given-names>S.</given-names></name> <name><surname>Tai</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2021a</year>). Dual reweighting domain generalization for face presentation attack detection. In: <italic>IJCAI International Joint Conference on Artificial Intelligence</italic>.</citation></ref>
<ref id="ref29"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Lv</surname> <given-names>L.</given-names></name> <name><surname>Xiang</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Huang</surname> <given-names>H.</given-names></name> <name><surname>Ruan</surname> <given-names>R.</given-names></name> <name><surname>Xu</surname> <given-names>X.</given-names></name></person-group> (<year>2021</year>). Combining dynamic image and prediction ensemble for cross-domain face anti-spoofing. In: <italic>IEEE International Conference on Acoustics Speech and Signal Processing</italic>. pp. 2550&#x2013;2554.</citation></ref>
<ref id="ref30"><citation citation-type="other"><person-group person-group-type="author"><name><surname>M&#x00E4;&#x00E4;tt&#x00E4;</surname> <given-names>J.</given-names></name> <name><surname>Hadid</surname> <given-names>A.</given-names></name> <name><surname>Pietik&#x00E4;inen</surname> <given-names>M.</given-names></name></person-group> (<year>2011</year>). Face spoofing detection from single images using micro-texture analysis. In: <italic>IEEE International Conference on Biometrics</italic> pp. 1&#x2013;7.</citation></ref>
<ref id="ref31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Menotti</surname> <given-names>D.</given-names></name> <name><surname>Chiachia</surname> <given-names>G.</given-names></name> <name><surname>Pinto</surname> <given-names>A.</given-names></name> <name><surname>Schwartz</surname> <given-names>W. R.</given-names></name> <name><surname>Pedrini</surname> <given-names>H.</given-names></name> <name><surname>Falcao</surname> <given-names>A. X.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Deep representations for iris, face, and fingerprint spoofing detection</article-title>. <source>IEEE Trans. Inf. Forensics Secur.</source> <volume>10</volume>, <fpage>864</fpage>&#x2013;<lpage>879</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TIFS.2015.2398817</pub-id></citation></ref>
<ref id="ref32"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Nagpal</surname> <given-names>C.</given-names></name> <name><surname>Dubey</surname> <given-names>S. R.</given-names></name></person-group> (<year>2019</year>). A performance evaluation of convolutional neural networks for face anti spoofing. In: <italic>International Joint Conference on Neural Networks (IJCNN)</italic> pp. 1&#x2013;8.</citation></ref>
<ref id="ref33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pinto</surname> <given-names>A.</given-names></name> <name><surname>Pedrini</surname> <given-names>H.</given-names></name> <name><surname>Schwartz</surname> <given-names>W. R.</given-names></name> <name><surname>Rocha</surname> <given-names>A.</given-names></name></person-group> (<year>2015</year>). <article-title>Face spoofing detection through visual codebooks of spectral temporal cubes</article-title>. <source>IEEE Trans. Image Process.</source> <volume>24</volume>, <fpage>4726</fpage>&#x2013;<lpage>4740</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TIP.2015.2466088</pub-id>, PMID: <pub-id pub-id-type="pmid">26276988</pub-id></citation></ref>
<ref id="ref34"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Rehman</surname> <given-names>Y. A. U.</given-names></name> <name><surname>Po</surname> <given-names>L. M.</given-names></name> <name><surname>Liu</surname> <given-names>M.</given-names></name></person-group> (<year>2017</year>). Deep learning for face anti-spoofing: an end-to-end approach. In: <italic>Signal Processing: Algorithms, Architectures, Arrangements, and Applications (SPA)</italic>. pp. 195&#x2013;200.</citation></ref>
<ref id="ref35"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Saha</surname> <given-names>S.</given-names></name> <name><surname>Xu</surname> <given-names>W.</given-names></name> <name><surname>Kanakis</surname> <given-names>M.</given-names></name> <name><surname>Georgoulis</surname> <given-names>S.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Paudel</surname> <given-names>D. P.</given-names></name> <etal/></person-group>. (<year>2020</year>). Domain agnostic feature learning for image and video based face anti-spoofing. In: <italic>IEEE Conference on Computer Vision Pattern Recognition</italic>. pp. 802&#x2013;803.</citation></ref>
<ref id="ref36"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Shao</surname> <given-names>R.</given-names></name> <name><surname>Lan</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Yuen</surname> <given-names>P. C.</given-names></name></person-group> (<year>2019</year>). Multi-adversarial discriminative deep domain generalization for face presentation attack detection. In: <italic>IEEE Conference on Computer Vision Pattern Recognition</italic> pp. 10023&#x2013;10031.</citation></ref>
<ref id="ref37"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Simonyan</surname> <given-names>K.</given-names></name> <name><surname>Zisserman</surname> <given-names>A.</given-names></name></person-group> (<year>2014</year>). Two-stream convolutional networks for action recognition in videos. In: <italic>27th International Conference on Neural Information Processing Systems</italic> 27.</citation></ref>
<ref id="ref38"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Srivatsan</surname> <given-names>K.</given-names></name> <name><surname>Naseer</surname> <given-names>M.</given-names></name> <name><surname>Nandakumar</surname> <given-names>K.</given-names></name></person-group> (<year>2023</year>). FLIP: cross-domain face anti-spoofing with language guidance. In: <italic>Proceedings of IEEE International Conference on Computer Vision</italic> pp. 19685&#x2013;19696.</citation></ref>
<ref id="ref39"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Chu</surname> <given-names>W. S.</given-names></name></person-group> (<year>2023</year>). Rethinking domain generalization for face anti-spoofing: separability and alignment. in: <italic>Proceedings of IEEE International Conference on Computer Vision Pattern Recognition</italic>. pp. 24563&#x2013;24574.</citation></ref>
<ref id="ref40"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>W.</given-names></name> <name><surname>Zhao</surname> <given-names>H.</given-names></name> <name><surname>Jin</surname> <given-names>Z.</given-names></name></person-group> (<year>2016</year>). 3D convolutional neural networks for facial expression classification. In: <italic>Asian Conference on Computer Vision</italic> 528&#x2013;543.</citation></ref>
<ref id="ref41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>W.</given-names></name> <name><surname>Zhao</surname> <given-names>H.</given-names></name> <name><surname>Jin</surname> <given-names>Z.</given-names></name></person-group> (<year>2019</year>). <article-title>A facial expression recognition method based on ensemble of 3D convolutional neural networks</article-title>. <source>Neural Comput. Applic.</source> <volume>31</volume>, <fpage>2795</fpage>&#x2013;<lpage>2812</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s00521-017-3230-2</pub-id></citation></ref>
<ref id="ref42"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>C. Y.</given-names></name> <name><surname>Lu</surname> <given-names>Y. D.</given-names></name> <name><surname>Yang</surname> <given-names>S. T.</given-names></name> <name><surname>Lai</surname> <given-names>S. H.</given-names></name></person-group> (<year>2022</year>). Patchnet: a simple face anti-spoofing framework via fine-grained patch recognition. In: <italic>Proceedings of IEEE International Conference on Computer Vision Pattern Recognition</italic>. Pp. 20281-20290.</citation></ref>
<ref id="ref43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>Q.</given-names></name> <name><surname>Deng</surname> <given-names>W.</given-names></name> <name><surname>Guo</surname> <given-names>G.</given-names></name></person-group> (<year>2022a</year>). <article-title>Learning multi-granularity temporal characteristics for face anti-spoofing</article-title>. <source>IEEE Trans. Inf. Forensics Secur.</source> <volume>17</volume>, <fpage>1254</fpage>&#x2013;<lpage>1269</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TIFS.2022.3158062</pub-id></citation></ref>
<ref id="ref44"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Yu</surname> <given-names>Z.</given-names></name> <name><surname>Deng</surname> <given-names>W.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Gao</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2022b</year>). Domain generalization via shuffled style assembly for face anti-spoofing. In: <italic>ACM International Conference on Multimedia</italic>. pp. 4123&#x2013;4133.</citation></ref>
<ref id="ref45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Bian</surname> <given-names>Y.</given-names></name> <name><surname>Cai</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Pu</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>Self-domain adaptation for face anti-spoofing</article-title>. <source>AAAI Conf. Artif. Intell.</source> <volume>35</volume>, <fpage>2746</fpage>&#x2013;<lpage>2754</lpage>. doi: <pub-id pub-id-type="doi">10.1609/aaai.v35i4.16379</pub-id></citation></ref>
<ref id="ref46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wen</surname> <given-names>D.</given-names></name> <name><surname>Han</surname> <given-names>H.</given-names></name> <name><surname>Jain</surname> <given-names>A. K.</given-names></name></person-group> (<year>2015</year>). <article-title>Face spoof detection with image distortion analysis</article-title>. <source>IEEE Trans. Inf. Forensics Secur.</source> <volume>10</volume>, <fpage>746</fpage>&#x2013;<lpage>761</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TIFS.2015.2400395</pub-id></citation></ref>
<ref id="ref47"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>G.</given-names></name> <name><surname>Zhou</surname> <given-names>Z.</given-names></name> <name><surname>Guo</surname> <given-names>Z.</given-names></name></person-group> (<year>2021</year>). A robust method with dropblock for face anti-spoofing. In: <italic>International Joint Conference on Neural Networks</italic> pp. 1&#x2013;8.</citation></ref>
<ref id="ref48"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>J.</given-names></name> <name><surname>Lei</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>S. Z.</given-names></name></person-group> (<year>2014</year>). Learn convolutional neural network for face anti-spoofing. arXiv [Preprint].</citation></ref>
<ref id="ref49"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Yin</surname> <given-names>W.</given-names></name> <name><surname>Ming</surname> <given-names>Y.</given-names></name> <name><surname>Tian</surname> <given-names>L.</given-names></name></person-group> (<year>2016</year>). A face anti-spoofing method based on optical flow field. In: <italic>13th International Conference on Signal Processing (ICSP)</italic> pp. 1333&#x2013;1337.</citation></ref>
<ref id="ref50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>Z.</given-names></name> <name><surname>Qin</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Zhao</surname> <given-names>C.</given-names></name> <name><surname>Lei</surname> <given-names>Z.</given-names></name> <name><surname>Zhao</surname> <given-names>G.</given-names></name></person-group> (<year>2022</year>). <article-title>Deep learning for face anti-spoofing: a survey</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>45</volume>, <fpage>5609</fpage>&#x2013;<lpage>5631</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TPAMI.2022.3215850</pub-id></citation></ref>
<ref id="ref51"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>Z.</given-names></name> <name><surname>Zhao</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Qin</surname> <given-names>Y.</given-names></name> <name><surname>Su</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2020</year>). Searching central difference convolutional networks for face anti-spoofing. In: <italic>IEEE Conference on Computer Vision Pattern Recognition</italic>. pp. 5295&#x2013;5305.</citation></ref>
<ref id="ref52"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Yan</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>S.</given-names></name> <name><surname>Lei</surname> <given-names>Z.</given-names></name> <name><surname>Yi</surname> <given-names>D.</given-names></name> <name><surname>Li</surname> <given-names>S. Z.</given-names></name></person-group> (<year>2012</year>). A face antispoofing database with diverse attacks. In: <italic>IAPR International Conference on Biometrics</italic>. pp. 26&#x2013;31.</citation></ref>
<ref id="ref53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>L.</given-names></name> <name><surname>Luo</surname> <given-names>J.</given-names></name> <name><surname>Gao</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>W.</given-names></name> <name><surname>Lei</surname> <given-names>B.</given-names></name> <name><surname>Leng</surname> <given-names>J.</given-names></name></person-group> (<year>2021</year>). <article-title>Selective domain-invariant feature alignment network for face anti-spoofing</article-title>. <source>IEEE Trans Inf. Forensics Secur.</source> <volume>16</volume>, <fpage>5352</fpage>&#x2013;<lpage>5365</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TIFS.2021.3125603</pub-id></citation></ref>
<ref id="ref54"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Q.</given-names></name> <name><surname>Zhang</surname> <given-names>K. Y.</given-names></name> <name><surname>Yao</surname> <given-names>T.</given-names></name> <name><surname>Lu</surname> <given-names>X.</given-names></name> <name><surname>Yi</surname> <given-names>R.</given-names></name> <name><surname>Ding</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2023</year>). Instance-aware domain generalization for face anti-spoofing. In: <italic>Proceedings of IEEE Conference on Computer Vision Pattern Recognition</italic>. pp. 20453&#x2013;20463.</citation></ref>
<ref id="ref55"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Q.</given-names></name> <name><surname>Zhang</surname> <given-names>K. Y.</given-names></name> <name><surname>Yao</surname> <given-names>T.</given-names></name> <name><surname>Yi</surname> <given-names>R.</given-names></name> <name><surname>Ding</surname> <given-names>S.</given-names></name> <name><surname>Ma</surname> <given-names>L.</given-names></name></person-group> (<year>2022b</year>). Adaptive mixture of experts learning for generalizable face anti-spoofing. In: <italic>ACM International Conference on Multimedia</italic> pp. 6009&#x2013;6018.</citation></ref>
<ref id="ref56"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Q.</given-names></name> <name><surname>Zhang</surname> <given-names>K. Y.</given-names></name> <name><surname>Yao</surname> <given-names>T.</given-names></name> <name><surname>Yi</surname> <given-names>R.</given-names></name> <name><surname>Sheng</surname> <given-names>K.</given-names></name> <name><surname>Ding</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2022a</year>). <source>Generative domain adaptation for face anti-spoofing</source>. <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer Nature Switzerland</publisher-name>. Pp. <fpage>335</fpage>&#x2013;<lpage>356</lpage>.</citation></ref>
</ref-list>
</back>
</article>
