<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Hum. Neurosci.</journal-id>
<journal-title>Frontiers in Human Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Hum. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-5161</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnhum.2024.1471634</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Human Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Domain adaptation spatial feature perception neural network for cross-subject EEG emotion recognition</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Lu</surname> <given-names>Wei</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2874260/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Xiaobo</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Xia</surname> <given-names>Lingnan</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Ma</surname> <given-names>Hua</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2414087/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Tan</surname> <given-names>Tien-Ping</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Henan High-speed Railway Operation and Maintenance Engineering Research Center, Zhengzhou Railway Vocational and Technical College</institution>, <addr-line>Zhengzhou</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>School of Computer Sciences, Universiti Sains Malaysia</institution>, <addr-line>Penang</addr-line>, <country>Malaysia</country></aff>
<aff id="aff3"><sup>3</sup><institution>Jiangxi Vocational College of Finance and Economics</institution>, <addr-line>Jiujiang</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Jiahui Pan, South China Normal University, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Dong Cui, Yanshan University, China</p>
<p>Man Fai Leung, Anglia Ruskin University, United Kingdom</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Hua Ma <email>mahua&#x00040;zzrvtc.edu.cn</email></corresp>
<corresp id="c002">Tien-Ping Tan <email>tienping&#x00040;usm.my</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>17</day>
<month>12</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>18</volume>
<elocation-id>1471634</elocation-id>
<history>
<date date-type="received">
<day>08</day>
<month>08</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>11</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2024 Lu, Zhang, Xia, Ma and Tan.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Lu, Zhang, Xia, Ma and Tan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Emotion recognition is a critical research topic within affective computing, with potential applications across various domains. Currently, EEG-based emotion recognition, utilizing deep learning frameworks, has been effectively applied and achieved commendable performance. However, existing deep learning-based models face challenges in capturing both the spatial activity features and spatial topology features of EEG signals simultaneously. To address this challenge, a <bold>d</bold>omain-adaptation <bold>s</bold>patial-feature <bold>p</bold>erception-network has been proposed for cross-subject EEG emotion recognition tasks, named DSP-EmotionNet. Firstly, a <bold>s</bold>patial <bold>a</bold>ctivity <bold>t</bold>opological <bold>f</bold>eature <bold>e</bold>xtractor <bold>m</bold>odule has been designed to capture spatial activity features and spatial topology features of EEG signals, named SATFEM. Then, using SATFEM as the feature extractor, DSP-EmotionNet has been designed, significantly improving the accuracy of the model in cross-subject EEG emotion recognition tasks. The proposed model surpasses state-of-the-art methods in cross-subject EEG emotion recognition tasks, achieving an average recognition accuracy of 82.5% on the SEED dataset and 65.9% on the SEED-IV dataset.</p></abstract>
<kwd-group>
<kwd>affective computing</kwd>
<kwd>electroencephalography</kwd>
<kwd>emotion recognition</kwd>
<kwd>convolutional neural network</kwd>
<kwd>graph attention network</kwd>
<kwd>domain adaptation</kwd>
</kwd-group>
<contract-num rid="cn001">232102240089</contract-num>
<contract-num rid="cn001">232102240091</contract-num>
<contract-num rid="cn001">242102241064</contract-num>
<contract-num rid="cn002">23B520033</contract-num>
<contract-num rid="cn002">25B580004</contract-num>
<contract-sponsor id="cn001">Henan Provincial Science and Technology Research Project<named-content content-type="fundref-id">10.13039/501100017700</named-content></contract-sponsor>
<contract-sponsor id="cn002">Education Department of Henan Province<named-content content-type="fundref-id">10.13039/501100009101</named-content></contract-sponsor>
<counts>
<fig-count count="5"/>
<table-count count="4"/>
<equation-count count="30"/>
<ref-count count="37"/>
<page-count count="15"/>
<word-count count="9725"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Brain-Computer Interfaces</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>Introduction</title>
<p>Emotion recognition (Jia et al., <xref ref-type="bibr" rid="B12">2021</xref>; Tan et al., <xref ref-type="bibr" rid="B23">2020</xref>; Cimtay et al., <xref ref-type="bibr" rid="B6">2020</xref>; Doma and Pirouz, <xref ref-type="bibr" rid="B8">2020</xref>) has become an important task in affective computing. It has potential applications in areas like affective brain-computer interfaces, diagnosing affective disorders, detecting emotions in patients with consciousness disorders, emotion detection of drivers, mental workload estimation, and cognitive neuroscience. Emotion is a mental and physiological state that arises from various sensory and cognitive inputs, significantly influencing human behavior in daily life (Jia et al., <xref ref-type="bibr" rid="B12">2021</xref>). Emotion is a response to both internal and external stimuli. Physiological signals, such as Electrocardiography (ECG), Electromyography (EMG), and Electroencephalography (EEG), correspond to the physiological responses caused by emotions. They are more reliable indicators of emotional expression than non-physiological signals, such as speech, posture, and facial expression, which can be masked by humans (Tan et al., <xref ref-type="bibr" rid="B23">2020</xref>; Cimtay et al., <xref ref-type="bibr" rid="B6">2020</xref>). Among these physiological signals, EEG signals have a high temporal resolution and a wealth of information, which can reveal subtle changes in emotions, making them more suitable for emotion recognition than other physiological signals (Atkinson and Campos, <xref ref-type="bibr" rid="B2">2016</xref>). EEG-based emotion recognition methods are more accurate and objective, as some studies have verified the relationship between EEG signals and emotions (Xing et al., <xref ref-type="bibr" rid="B31">2019</xref>).</p>
<p>In recent years, EEG signals have gained widespread application in emotion recognition due to their ability to accurately reflect the genuine emotions of subjects (Jia et al., <xref ref-type="bibr" rid="B11">2020</xref>; Zhou et al., <xref ref-type="bibr" rid="B37">2023</xref>). Early approaches to EEG-based emotion recognition have relied on processes such as signal denoising, feature design, and classifier learning. For example, Wang et al. have introduced the Support Vector Machine (SVM) classifier (Wang et al., <xref ref-type="bibr" rid="B28">2011</xref>), while Bahari et al. have proposed the K-Nearest Neighbors (KNN) classifier (Bahari and Janghorbani, <xref ref-type="bibr" rid="B3">2013</xref>), both achieving effective emotion classification. However, traditional machine learning techniques have been constrained by intricate feature engineering and selection processes. To overcome these limitations, researchers have introduced deep learning techniques. The continuous refinement of deep learning algorithms has led to significant achievements in EEG-based emotion recognition. For example, Kwon et al. have utilized CNN to extract features from EEG signals, while Li et al. have obtained deep representations of all EEG electrode signals using Recurrent Neural Networks (RNN; Kwon et al., <xref ref-type="bibr" rid="B14">2018</xref>; Li et al., <xref ref-type="bibr" rid="B17">2020</xref>). Additionally, some researchers have adopted hybrid models combining Convolutional Neural Networks (CNN) and RNN. For instance, Ramzan et al. have proposed a parallel CNN and LSTM-RNN deep learning model for emotion recognition and classification (Ramzan and Dawn, <xref ref-type="bibr" rid="B20">2023</xref>). Although traditional neural network models such as CNN and RNN have achieved high accuracy in EEG emotion recognition tasks, they typically process data in the form of grid data. However, grid data cannot effectively represent connections between different brain regions, thus hindering models from directly capturing the spatial topological features of EEG signals. To better capture connections between brain regions and achieve improved performance in emotion recognition tasks, researchers have begun exploring the use of graph data to represent interactions between brain regions and employing Graph Neural Networks (GNNs) to process this data. For instance, Asadzadeh et al. have proposed an emotion recognition method based on EEG source signals using a Graph Neural Network approach (Asadzadeh et al., <xref ref-type="bibr" rid="B1">2023</xref>). However, models based on GNNs face challenges in accurately detecting local features and capturing the spatial activity features of EEG signals.</p>
<p>However, when applying deep learning models to interdisciplinary tasks such as EEG-based emotion recognition, significant challenges arise due to the limited number of subjects in EEG emotion datasets, coupled with individual differences among subjects. This often results in a notable decrease in the performance of deep learning models in cross-subject EEG emotion recognition tasks. To address the issue of poor performance of subjects in EEG emotion recognition, many researchers have begun exploring the application of transfer learning techniques. In cross-subject EEG emotion recognition tasks, transfer learning primarily addresses the issue of domain gaps caused by individual differences. Transfer learning mainly includes fine-tuning and domain adaptation. Fine-tuning, as an effective knowledge transfer method, has gained widespread adoption. Zhang et al. introduced the Self-Training Maximum Classifier Difference (SMCD) model, utilizing fine-tuning to apply a model trained on the source domain to the target domain (Zhang et al., <xref ref-type="bibr" rid="B33">2023</xref>). However, collecting a large amount of labeled data from the target domain requires considerable time, manpower, and financial resources. Especially in tasks like EEG emotion recognition, acquiring large-scale EEG datasets and labeling them is a complex and expensive task. In some cases, labeled data from the target domain may be extremely scarce, or even insufficient for fine-tuning, which limits the performance and generalization ability of the model on the target task. Researchers have begun exploring the application of domain adaptation in cross-disciplinary EEG emotion recognition. Li et al. proposed a Domain Adaptation method that enhances adaptability by minimizing source domain error and aligning latent representations (Li et al., <xref ref-type="bibr" rid="B15">2019</xref>). However, the majority of existing domain adaptation methods only focus on extracting shallow-level features, without effectively aligning deep-level features of different types. This greatly limits the ability of the model for cross-domain transfer learning.</p>
<p>The primary contributions of this paper can be outlined as follows:</p>
<list list-type="bullet">
<list-item><p>To accurately capture the activity states of different brain regions and their inter-regional connectivity, we have designed a dual-branch <bold>S</bold>patial <bold>A</bold>ctivity <bold>T</bold>opological <bold>F</bold>eature <bold>E</bold>xtractor <bold>M</bold>odule, named SATFEM. This module has been able to simultaneously extract spatial activity features and spatial topological features from EEG signals, significantly enhancing the recognition performance of the model.</p></list-item>
<list-item><p>To minimize the disparity between the source and target domains, we have devised a <bold>D</bold>omain-adaptation <bold>S</bold>patial-feature <bold>P</bold>erception-network for cross-subject EEG emotion recognition, resulting in the proposal of the DSP-EmotionNet model. This model is tailored to enhance the generalization of the model on the target domain, thereby elevating the accuracy of cross-subject EEG emotion recognition tasks.</p></list-item>
<list-item><p>The proposed DSP-EmotionNet model achieves accuracy rates of 82.5% and 65.9% on the SEED and SEED-IV datasets, respectively, for cross-subject EEG emotion recognition tasks. These rates surpass those of state-of-the-art models. Additionally, a series of ablation experiments have been conducted to investigate the contributions of key components within DSP-EmotionNet to the recognition performance of cross-subject EEG emotion recognition tasks.</p></list-item>
</list>
</sec>
<sec id="s2">
<title>1 Related work</title>
<p>Traditional EEG feature extractors, such as CNNs and RNNs, have limitations in capturing the connections between brain regions, which constrains their ability to extract spatial topological features. Although GNN models have made improvements in this area, they still face challenges in detecting subtle local variations. Domain adaptation techniques have shown success in cross-subject EEG emotion recognition tasks, but most existing domain adaptation-based methods focus predominantly on aligning shallow features, failing to effectively utilize deeper and more diverse feature types.</p>
<sec>
<title>1.1 EEG spatial activity feature extractor</title>
<p>In recent years, the application of EEG signals in the field of emotion recognition has significantly increased. This is mainly attributed to the accurate and authentic reflection of the true emotional states of individuals by EEG signals. With the development of deep learning, two popular deep learning models, CNN and RNN, have been widely applied in EEG emotion recognition. For instance, Kwon et al. have utilized CNN for feature extraction from EEG signals. In their model, the EEG signal undergoes preprocessing via wavelet transform before convolution, considering both the time and frequency aspects of the EEG signal (Kwon et al., <xref ref-type="bibr" rid="B14">2018</xref>). Li et al. have employed four directed RNNs based on two spatial directions to traverse the electrode signals of two different brain regions, obtaining a deep representation of all EEG electrode signals while preserving their inherent spatial dependencies (Li et al., <xref ref-type="bibr" rid="B17">2020</xref>). Moreover, some researchers have adopted hybrid models combining CNN and RNN. For example, Chakravarthi et al. have proposed a classification method that combines CNN and LSTM, aiming to recognize and classify different emotional states by analyzing EEG data (Chakravarthi et al., <xref ref-type="bibr" rid="B5">2022</xref>). Ramzan et al. have proposed a parallel CNN and Long Short-Term Memory Recurrent Neural Network (LSTM-RNN) deep learning model, which primarily utilizes CNN for extracting spatial features of EEG signals and LSTM-RNN for extracting temporal features of EEG signals, thus achieving emotion recognition and classification (Ramzan and Dawn, <xref ref-type="bibr" rid="B20">2023</xref>). However, EEG spatial activity feature extractors such as CNNs and RNNs typically process data in a grid format. While grid data can effectively reflect the spatial activity states of EEG signals, it fails to adequately represent the connections between different brain regions. This limitation hinders the model&#x00027;s ability to directly capture the spatial topological features of EEG signals.</p>
</sec>
<sec>
<title>1.2 EEG spatial topological feature extractor</title>
<p>Despite the high accuracy achieved by traditional neural network models such as CNN and RNN in EEG emotion recognition tasks, the data they handle is typically in the form of grid data. EEG data are usually captured from multiple electrodes on the scalp, with each electrode signal representing the activity of the corresponding brain region. However, grid data cannot effectively represent the connectivity between brain regions, thereby preventing the model from directly capturing the connections between different brain regions. Therefore, in order to better capture the connectivity between brain regions and achieve better performance in emotion recognition tasks, researchers have begun to explore the use of graph data to represent the connections between brain regions and leverage GNNs to process such graph data. For instance, Asadzadeh et al. have proposed an emotion recognition method based on EEG source signals using a Graph Neural Network node (ESB-G3N). This method treats EEG source signals as node signals in graph data, the relationships between EEG source signals as the adjacency matrix of the graph data and employs GNN for EEG emotion recognition (Asadzadeh et al., <xref ref-type="bibr" rid="B1">2023</xref>). However, GNN-based models have certain advantages as EEG spatial topological feature extractors in processing the spatial topological features of EEG signals, they face challenges in accurately detecting local features and subtle variations in brain activity.</p>
</sec>
<sec>
<title>1.3 Transfer learning for emotion recognition</title>
<p>Due to the potential applications of deep learning models in various fields, there is great interest in utilizing these models for EEG-based emotion recognition. However, when applying deep learning models to cross-subject EEG emotion recognition tasks, there is a significant challenge due to the limited number of subjects in EEG emotion datasets, coupled with individual differences between subjects. This often results in a significant drop in the performance of deep learning models in interdisciplinary EEG emotion recognition tasks. To address the issue of decreased performance of subjects in EEG emotion recognition, many researchers have begun to explore the application of transfer learning techniques. In interdisciplinary EEG emotion recognition tasks, transfer learning primarily addresses the problem of data domain gaps caused by individual differences. EEG signals from different subjects in the same emotional state may exhibit significant variations due to individual differences. In such cases, the target domain has represented the feature space of EEG data obtained from a certain number of subjects. In contrast, the source domain has included data collected from one or more different individuals. Li et al. have incorporated fine-tuning into emotion recognition networks and examined the extent to which the models can be shared among subjects (Li et al., <xref ref-type="bibr" rid="B16">2018</xref>). Wang et al. have proposed a method that utilizes fine-tuning to address the challenge of emotional differences across different datasets in deep model transfer learning, to construct a robust emotion recognition model (Wang et al., <xref ref-type="bibr" rid="B27">2020</xref>). These methods overcome subject differences by training on the source domain and fine-tuning on the target domain. Although existing transfer learning methods for EEG emotion recognition can achieve improved results, almost all existing work requires the use of a certain amount of labeled data from the target domain for fine-tuning training. However, collecting a large amount of labeled data from the target domain requires a considerable amount of time, manpower, and financial resources. Especially in tasks such as EEG emotion recognition, obtaining large-scale EEG datasets and labeling them is a complex and expensive task. In some cases, the labeled data from the target domain may be extremely scarce or even insufficient for fine-tuning, which may limit the performance and generalization ability of the model on the target task. Therefore, some researchers have begun exploring the application of domain adaptation for cross-subject eeg emotion recognition. For example, Jin et al. have proposed the utilization of the Domain Adaptation Network (DAN) for knowledge transfer in EEG-based emotion recognition to address the fundamental problem of mitigating differences between the source subject and target subject in order to eliminate subject variability (Jin et al., <xref ref-type="bibr" rid="B13">2017</xref>). Li et al. have proposed a domain adaptation method for EEG emotion recognition, which is optimized by minimizing the classification error on the source domain while simultaneously aligning the latent representations of the source and target domains to make them more similar (Li et al., <xref ref-type="bibr" rid="B15">2019</xref>). Wang et al. have proposed an efficient few-label domain adaptation method based on the multi-subject learning model for cross-subject emotion classification tasks with limited EEG data (Wang et al., <xref ref-type="bibr" rid="B29">2021</xref>). However, most existing domain adaptation-based methods for cross-subject EEG emotion recognition focus primarily on aligning shallow features, without effectively aligning and fully utilizing deeper, more diverse types of features.</p>
</sec>
</sec>
<sec sec-type="methods" id="s3">
<title>2 Methodology</title>
<sec>
<title>2.1 Overview</title>
<p>The overall architecture of the proposed model is illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>. We summarize three key ideas of the proposed DSP-EmotionNet model as follows: (1) Constructing EEG spatial activity features and EEG spatial topological features. (2) Integrating spatial activity feature extractor and spatial topological feature extractor to capture the connections between different brain regions and the subtle changes in brain activity, this module is named the SATFEM module. The SATFEM module enhances the generalization ability of the model in cross-subject EEG emotion recognition by extracting both spatial activation and spatial topological features, resulting in a more robust feature representation. Compared to traditional methods that focus on a single type of feature, this combined approach better captures the complexity of EEG data. (3) Utilizing the SATFEM module as a feature extractor, a domain adaptation spatial feature perception network was proposed for cross-subject EEG emotion recognition tasks, improving the generalization ability of the model. This method not only applies domain adaptation techniques but also employs a dual-branch feature extractor to ensure effective domain feature alignment between different subjects. This enables domain adaptation to go beyond merely aligning shallow features, allowing for the effective alignment of deeper and more diverse feature types.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>The overall architecture of DSP-EmotionNet for EEG emotion recognition is as follows. Initially, two distinct feature maps of the brain are constructed: one representing EEG spatial activity features and the other representing EEG spatial topological features. Subsequently, the spatial activity feature extractor is employed to detect subtle changes in brain activity, and the spatial topological feature extractor is used to capture the connectivity between different brain regions. Finally, a domain adaptation spatial feature perception network is proposed for cross-subject EEG emotion recognition tasks, aimed at enhancing the generalization capability of the model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnhum-18-1471634-g0001.tif"/>
</fig>
</sec>
<sec>
<title>2.2 EEG feature representations</title>
<p>In this section, we introduce two distinct EEG feature representations: EEG spatial activity feature representation and EEG spatial topological feature representation. These different feature representations reflect various spatial relationships within the brain. Specifically, We employ EEG spatial activity feature representation to illustrate spatial activation state distribution maps of the brain, which can reflect the activation states of different brain regions in space. We use EEG spatial topological feature representation to depict spatial topological functional connectivity maps of the brain, which can reflect the connectivity between different brain regions in space. These two EEG feature representations complement each other and effectively demonstrate the spatial relationships of EEG signals.</p>
<sec>
<title>2.2.1 EEG spatial activity feature representation</title>
<p>To construct the EEG spatial activity feature representation, we employ the temporal-frequency feature extraction method to derive the Differential Entropy (DE) of five frequency bands {&#x003B4;, &#x003B8;, &#x003B1;, &#x003B2;, &#x003B3;} from all EEG channels across EEG signal samples within 4-s segments. We denote <inline-formula><mml:math id="M1"><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:mi>B</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> as a frequency feature matrix comprising frequency bands extracted from the DE feature, where <italic>B</italic> &#x02208; {&#x003B4;, &#x003B8;, &#x003B1;, &#x003B2;, &#x003B3;} represents the frequency band and <italic>N</italic><sub><italic>e</italic></sub> &#x02208; {<italic>FP</italic>1, <italic>FPZ</italic>, ..., <italic>CB</italic>2} denotes the electrode. Subsequently, the selected data are mapped onto a frequency domain brain electrode location matrix <inline-formula><mml:math id="M2"><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>H</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>b</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mi>B</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, based on the electrode positions of the brain. Finally, the frequency-domain brain electrode position matrices corresponding to different frequencies are overlaid to generate the spatial-frequency feature representation of EEG signals. Thus, the construction of the EEG feature representation <inline-formula><mml:math id="M3"><mml:msup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>H</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>W</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>B</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> is completed. The construction process of EEG spatial activity feature representation is illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>The construction process of EEG spatial activity feature representation. We adopt a time-frequency feature extraction method to extract 4-s EEG signal DE features from EEG signal samples. Subsequently, based on the electrode positions of the brain, the selected data are mapped onto the brain electrode position matrix. Finally, the electrode position matrices corresponding to different frequencies are superimposed to generate a spatial activity feature representation of the EEG signal.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnhum-18-1471634-g0002.tif"/>
</fig>
</sec>
<sec>
<title>2.2.2 EEG spatial topological ferture representation</title>
<p>To construct the EEG spatial topological feature representation, we employ the temporal-frequency feature extraction method to derive the DE of five frequency bands {&#x003B4;, &#x003B8;, &#x003B1;, &#x003B2;, &#x003B3;} from all EEG channels across EEG signal samples within 4-s segments. We denote <inline-formula><mml:math id="M4"><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:mi>B</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> as a frequency feature matrix comprising frequency bands extracted from the DE feature, where <italic>B</italic> &#x02208; {&#x003B4;, &#x003B8;, &#x003B1;, &#x003B2;, &#x003B3;} represents the frequency band and <italic>N</italic><sub><italic>e</italic></sub> &#x02208; {<italic>FP</italic>1, <italic>FPZ</italic>, ..., <italic>CB</italic>2} denotes the electrode. Subsequently, the frequency domain brain electrode network is defined as a graph <italic>G</italic> &#x0003D; (<italic>V, E, A</italic>), where <italic>V</italic> represents the set of vertices, with each vertex representing an electrode in the brain; <italic>E</italic> denotes the set of edges, indicating the connections between vertices; and <italic>A</italic> denotes the adjacency matrix of the brain electrode network <italic>G</italic>. Finally, the frequency-domain brain electrode position graph corresponding to different frequencies is overlaid to generate the spatial-frequency feature representation of EEG signals. Thus, the construction of the EEG feature representation <inline-formula><mml:math id="M5"><mml:msup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> is completed. The construction process of EEG spatial topological feature representation is illustrated in <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>The construction process of EEG spatial topological feature representation. We employ a time-frequency feature extraction method to extract 4-s EEG signal DE features from EEG signal samples. Subsequently, the brain electrode position network is defined as a graph representation. Finally, the graph representations of electrode positions corresponding to different frequencies are superimposed to generate a spatial topological feature representation of the EEG signal.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnhum-18-1471634-g0003.tif"/>
</fig>
</sec>
</sec>
<sec>
<title>2.3 Spatial feature perception extractor</title>
<p>Using EEG spatial activity features and EEG spatial topological features as inputs, a dual-branch spatial-activity-topological feature extractor module named SATFEM is designed. SATFEM can simultaneously extract spatial activity features and spatial topological features. The features extracted from the dual branches are fused at the feature fusion layer. <xref ref-type="table" rid="T5">Algorithm 1</xref> shows the pseudocode for SATFEM. The SATFEM feature extractor consists of three main components: the spatial-topological feature extractor, the spatial activity feature extractor, and the feature fusion layer.</p>
<table-wrap position="float" id="T5">
<label>Algorithm 1</label>
<caption><p>SATFEM.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnhum-18-1471634-i0001.tif"/>
</table-wrap>
<sec>
<title>2.3.1 Spatial topological feature extractor</title>
<p>The Graph Attention Network (GAT) is proposed to address issues in deep GNN models, such as inefficient information propagation and unclear relationships between nodes (Velickovic et al., <xref ref-type="bibr" rid="B26">2017</xref>). GAT utilizes attention mechanisms to dynamically allocate weights between nodes, thereby enhancing the influence of important nodes and improving the efficiency of information propagation and clarity of relationships between nodes. Therefore, it is suitable for extracting EEG spatial topological feature representation as a feature extractor. This helps capture relationships between different functional areas in EEG feature representation, facilitating more accurate identification of different EEG signals. The input of GAT is the EEG spatial topological feature representation <inline-formula><mml:math id="M13"><mml:msup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>.</p>
<p>In graph, let any node <italic>v</italic><sub><italic>i</italic></sub> in the <italic>l</italic> &#x02212; <italic>th</italic> layer correspond to the feature vector <italic>h</italic><sub><italic>i</italic></sub>, where <inline-formula><mml:math id="M14"><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:msup></mml:math></inline-formula>, and <italic>d</italic><sup>(<italic>l</italic>)</sup> represents the feature dimension of the node. After an aggregation operation centered around the attention mechanism, the output is the new feature vector <inline-formula><mml:math id="M15"><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, where <inline-formula><mml:math id="M16"><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:msup></mml:math></inline-formula>, and <italic>d</italic><sup>(<italic>l</italic>&#x0002B;1)</sup> represents the length of the output feature vector. This aggregation operation is called the Graph Attention Layer(GAL).</p>
<p>Assuming the central node is <italic>v</italic><sub><italic>i</italic></sub>, let the weight coefficient from neighboring node <italic>v</italic><sub><italic>j</italic></sub> to <italic>v</italic><sub><italic>i</italic></sub> be denoted as <xref ref-type="disp-formula" rid="E1">Equation 1</xref>.</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M17"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>W</mml:mi><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The weight parameter <italic>W</italic> &#x02208; &#x0211D;<sup><italic>d</italic><sup>(<italic>l</italic>&#x0002B;1)</sup>&#x000D7;<italic>d</italic><sup>(<italic>l</italic>)</sup></sup> is used for the feature transformation of nodes in this layer. &#x003B1;(&#x000B7;) is the function used to compute the correlation between two nodes. The fully connected layer for a single layer is described as <xref ref-type="disp-formula" rid="E2">Equation 2</xref>.</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M18"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext>LeakyReLU</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mtext>T</mml:mtext></mml:mrow></mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02225;</mml:mo><mml:mi>W</mml:mi><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the weight parameter &#x003B1; &#x02208; &#x0211D;<sup>2<italic>d</italic><sup>(<italic>l</italic>&#x0002B;1)</sup></sup>, and the activation function is designed as the LeakyReLU function. To better distribute weights, it is necessary to uniformly normalize the relevance computed with all leaders, specifically through softmax normalization as shown in <xref ref-type="disp-formula" rid="E3">Equation 3</xref>.</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M19"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext>softma</mml:mtext><mml:msub><mml:mrow><mml:mtext>x</mml:mtext></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mi>&#x000D1;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The weight coefficient &#x003B1; is calculated such that <xref ref-type="disp-formula" rid="E3">Equation 3</xref> ensures that the sum of the weight coefficients for all neighbors is equal to 1. The complete formula for calculating the weight coefficients is described in <xref ref-type="disp-formula" rid="E4">Equation 4</xref>.</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M20"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext class="textrm" mathvariant="normal">LeakyReLU</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">T</mml:mtext></mml:mrow></mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02225;</mml:mo><mml:mi>W</mml:mi><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mi>&#x000D1;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext class="textrm" mathvariant="normal">LeakyReLU</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">T</mml:mtext></mml:mrow></mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02225;</mml:mo><mml:mi>W</mml:mi><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Following the calculation of the weight coefficients as described above, according to the weighted sum with attention mechanism, the new feature vector of node <italic>v</italic><sub><italic>i</italic></sub> is obtained as shown in <xref ref-type="disp-formula" rid="E5">Equation 5</xref>.</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M21"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mi>&#x000D1;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mi>W</mml:mi><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
<sec>
<title>2.3.2 Spatial activity feature extractor</title>
<p>The Residual Network (ResNet) is proposed to address the problem of degradation in deep CNN models. ResNet utilizes residual connections to link different convolutional layers, thereby enabling the propagation of shallow feature information to the deeper layers. Therefore, it is suitable for extracting EEG spatial Activity feature representation as a feature extractor.</p>
<p>The input of ResNet is the EEG spatial Activity feature representation <inline-formula><mml:math id="M22"><mml:msup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>H</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>W</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>B</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>. The EEG spatial Activity feature representation first goes through <italic>conv</italic>1, which consists of a 7 &#x000D7; 7 convolutional layer, a max pooling operation, and Batch Normalization. The <italic>conv</italic>1 layer is responsible for the initial processing of spatial information extraction and feature representation for EEG spatial Activity feature representation. Specifically, the input of <italic>conv</italic>1 is the spatial-frequency feature representation <inline-formula><mml:math id="M23"><mml:msup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>H</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>W</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>B</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, where the shape of the spatial Activity feature representation is <italic>H</italic> &#x000D7; <italic>W</italic> &#x000D7; <italic>C</italic>, with <italic>H</italic> representing the height, <italic>W</italic> representing the width, and <italic>C</italic> representing the number of channels. Due to the number of frequency bands being 5, <italic>C</italic> &#x0003D; 5. However, this does not meet the input requirements of the original ResNet model, as the first convolutional layer in the original ResNet model requires an input channel size of 3. If the original model is used directly to process data with 5 input channels, channel conversion or padding operations are required, which may result in the loss of important information from the original data. Therefore, we replaced the first half of the ResNet model with a new convolutional layer that has 5 input channels, 64 output channels, a kernel size of 7 &#x000D7; 7, a stride of 2, a padding of 4, and no bias. The equations for <italic>conv</italic>1 of ResNet are shown in equations <xref ref-type="disp-formula" rid="E6">Equation 6</xref>.</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M24"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext>MaxPool</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>ReLU</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>BN</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>Con</mml:mtext><mml:msub><mml:mrow><mml:mtext>v</mml:mtext></mml:mrow><mml:mrow><mml:mn>7</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mn>7</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>A</italic><sup><italic>M</italic></sup> is the input of <italic>conv</italic>1 in the CNN branch, <italic>C</italic>1 is the output of conv1 in the CNN branch. Conv<sub>7 &#x000D7; 7</sub>(&#x000B7;) represents the convolutional layer operation with an output channel of 64, kernel size of 7 &#x000D7; 7, the stride of 2, and padding of 4. BN(&#x000B7;) represents the batch normalization layer operation, which performs batch normalization on the output of the convolutional layer. ReLU(&#x000B7;) represents the ReLU activation function, which applies the ReLU activation function to the output of the batch normalization layer. MaxPool(&#x000B7;) represents the max pooling layer operation, which performs max pooling using a 3 &#x000D7; 3 pooling kernel, a stride of 2, and padding of 1.</p>
<p>The features output from <italic>conv</italic>1 are processed through <italic>conv</italic>2<sub><italic>x</italic></sub>, <italic>conv</italic>3<sub><italic>x</italic></sub>, <italic>conv</italic>4<sub><italic>x</italic></sub>, and <italic>conv</italic>5<sub><italic>x</italic></sub>, respectively. Each of <italic>conv</italic>2<sub><italic>x</italic></sub>, <italic>conv</italic>3<sub><italic>x</italic></sub>, <italic>conv</italic>4<sub><italic>x</italic></sub>, and <italic>conv</italic>5<sub><italic>x</italic></sub> consists of 2 BasicBlocks. In BasicBlock, the input feature is added to the main branch output feature via a shortcut connection before being passed through a ReLU activation function. The equation of the main branch, as shown in <xref ref-type="disp-formula" rid="E7">Equation 7</xref>.</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M25"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext>BN</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>Con</mml:mtext><mml:msub><mml:mrow><mml:mtext>v</mml:mtext></mml:mrow><mml:mrow><mml:mn>3</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>ReLU</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>BN</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>Con</mml:mtext><mml:msub><mml:mrow><mml:mtext>v</mml:mtext></mml:mrow><mml:mrow><mml:mn>3</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>I</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>X</italic><sub><italic>BasicIN</italic></sub> is the input of BasicBlock, <italic>X</italic><sub><italic>main</italic></sub> is the output of the main branch in the BasicBlock. Conv<sub>3&#x000D7;3</sub>(&#x000B7;) represents the convolutional layer operation with a kernel size of 3 &#x000D7; 3, the stride of 1, and padding of 1. BN(&#x000B7;) represents the batch normalization layer operation, which performs batch normalization on the output of the convolutional layer. ReLU(&#x000B7;) represents the ReLU activation function, which applies the ReLU activation function to the output of the batch normalization layer.</p>
<p>The shortcut connection allows the gradient to flow directly through the network, bypassing the convolutional layers in the main branch, which helps to prevent the vanishing gradient problem. The equation of the shortcut connection, as shown in <xref ref-type="disp-formula" rid="E8">Equation 8</xref>.</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M26"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>h</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>t</mml:mi><mml:mi>c</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>=</mml:mo><mml:mtext>BN</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>Con</mml:mtext><mml:msub><mml:mrow><mml:mtext>v</mml:mtext></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>Con</mml:mtext><mml:msub><mml:mrow><mml:mtext>v</mml:mtext></mml:mrow><mml:mrow><mml:mn>3</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>I</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>X</italic><sub><italic>BasicIN</italic></sub> is the input of BasicBlock, <italic>X</italic><sub><italic>shortcut</italic></sub> is the output of the shortcut connection in the BasicBlock. Conv<sub>3&#x000D7;3</sub>(&#x000B7;) represents the convolutional layer operation with a kernel size of 3 &#x000D7; 3, the stride of 1, and padding of 1. Conv<sub>1 &#x000D7; 1</sub>(&#x000B7;) represents the convolutional layer operation with a kernel size of 1 &#x000D7; 1. BN(&#x000B7;) represents the batch normalization layer operation. ReLU(&#x000B7;) represents the ReLU activation function.</p>
<p>The addition of the input feature to the main branch output feature allows the network to learn residual mappings, which can be easier to optimize during training. The equation of the addition, as shown in <xref ref-type="disp-formula" rid="E8">Equation 8</xref>.</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M27"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>O</mml:mi><mml:mi>U</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext>ReLU</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>h</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>t</mml:mi><mml:mi>c</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>X</italic><sub><italic>main</italic></sub> is the output of the main branch, <italic>X</italic><sub><italic>shortcut</italic></sub> is the output of the shortcut connection, and <italic>X</italic><sub><italic>BasicOUT</italic></sub> is the output of the BasicBlock. ReLU(&#x000B7;) represents the ReLU activation function.</p>
</sec>
<sec>
<title>2.3.3 Feature fusion layer</title>
<p>Utilizing EEG spatial activity feature representation and EEG spatial topological feature representation, the EEG spatial activity extractor and EEG spatial topological feature extractor respectively extract local features of EEG signals and functional connectivity of brain regions from EEG signals. Subsequently, the extracted EEG spatial activity feature representation information and EEG spatial topological feature representation information are fused in the Feature Fusion layer, as outlined in <xref ref-type="disp-formula" rid="E10">Equation 10</xref>. The fused dual-branch network module is referred to as the spatial-activity-topology feature extraction network module, abbreviated as SATFEM.</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M28"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">Fusion</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02225;</mml:mo><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x02225; represents the concatenate operation,<inline-formula><mml:math id="M29"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>, and <inline-formula><mml:math id="M30"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> respectively represent the EEG spatial activity feature and EEG spatial topological feature features extracted by the EEG spatial activity extractor and EEG spatial topological feature extractor branches, <italic>Y</italic><sub><italic>out</italic></sub> represents the fused feature output.</p>
</sec>
</sec>
<sec>
<title>2.4 Domain adaptation</title>
<p>A Domain Adversarial Neural Network (DANN) is used for implementing transfer learning. This framework was initially proposed by Ganin et al. for image classification (Ganin and Lempitsky, <xref ref-type="bibr" rid="B9">2015</xref>). Building upon the original DANN model, a domain adaptive learning model for EEG emotion recognition is proposed, utilizing SATFEM as the feature extractor, named DSP-EmotionNet. The aim is to address domain differences among different subjects. <xref ref-type="table" rid="T6">Algorithm 2</xref> shows the pseudocode for DSP-EmotionNet. The architecture of this model comprises three main components: the feature extractor, emotion classifier, and domain classifier.</p>
<table-wrap position="float" id="T6">
<label>Algorithm 2</label>
<caption><p>DSP-EmotionNet.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnhum-18-1471634-i0002.tif"/>
</table-wrap>
<p>The feature extractor is used to extract shared EEG emotion representations from both the source and target domain input data. For the model, the SATFEM module is selected as the feature extractor. The formula for the feature extractor in the model can be represented as <xref ref-type="disp-formula" rid="E23">Equation 23</xref>:</p>
<disp-formula id="E23"><label>(23)</label><mml:math id="M48"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>x</italic><sub><italic>i</italic></sub> represents the input sample, while <italic>H</italic><sub><italic>i</italic></sub> represents the output feature representation obtained from the feature extractor. The feature extractor utilizes parameters &#x003B8;<sub><italic>f</italic></sub> to map the input sample <italic>x</italic><sub><italic>i</italic></sub> to a high-level feature space that contains abstract features useful for the adversarial transfer learning task of EEG-based emotion recognition. These features are then passed to the emotion classifier and domain classifier for subsequent emotion recognition and domain adaptive learning tasks.</p>
<p>The emotion classifier is a classifier used for emotion classification. It takes the shared features extracted by the feature extractor as input and performs emotion classification on the source domain data. In this case, a fully connected layer is chosen as the classifier for emotion classification. The formula for the emotion classifier in the model can be represented as <xref ref-type="disp-formula" rid="E24">Equation 24</xref>:</p>
<disp-formula id="E24"><label>(24)</label><mml:math id="M49"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>H</italic><sub><italic>i</italic></sub> represents the output feature representation from the feature extractor, and <italic>Y</italic><sub><italic>i</italic></sub> represents the emotion prediction results of the model for the input sample <italic>x</italic><sub><italic>i</italic></sub>. The emotion classifier maps the feature representation <italic>H</italic><sub><italic>i</italic></sub> to a predicted probability distribution over emotion labels using the parameter &#x003D5;<sub><italic>y</italic></sub>.</p>
<p>The domain classifier is used to determine whether the input features are from the source domain or the target domain. It takes the shared features extracted by the feature extractor as input and attempts to correctly classify them as belonging to the source domain or the target domain. The objective of the domain classifier, achieved through adversarial training, is to make the extracted features indistinguishable in terms of the domain. The formula for the Domain Classifier in the model can be represented as <xref ref-type="disp-formula" rid="E25">Equation 25</xref>:</p>
<disp-formula id="E25"><label>(25)</label><mml:math id="M50"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>H</italic><sub><italic>i</italic></sub> represents the output feature representation from the Feature Extractor, and <italic>D</italic><sub><italic>i</italic></sub> represents the prediction results of the domain label for the input sample <italic>x</italic><sub><italic>i</italic></sub>. The Domain Classifier maps the feature representation <italic>H</italic><sub><italic>i</italic></sub> to a predicted probability distribution over domain labels using the parameter &#x003C8;<sub><italic>d</italic></sub>.</p>
<p>The model is capable of learning universal feature representations from EEG emotion data of different subjects, thereby improving the emotion recognition performance of both the source and target domains. Through domain adaptation training, this transfer learning model aligns the feature representations of the source and target domains, further enhancing the generalization ability and adaptability of the model to the target domain. The overall training objective of the model can be expressed as <xref ref-type="disp-formula" rid="E26">Equation 26</xref>.</p>
<disp-formula id="E26"><label>(26)</label><mml:math id="M51"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mrow><mml:mi mathvariant="script">E</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mrow><mml:mrow><mml:mi mathvariant="script">D</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>e</mml:mi><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>-</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mrow><mml:mrow><mml:mi mathvariant="script">D</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0222A;</mml:mo><mml:msubsup><mml:mrow><mml:mrow><mml:mi mathvariant="script">D</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>o</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B8;<sub><italic>f</italic></sub>, &#x003D5;<sub><italic>y</italic></sub>, and &#x003C8;<sub><italic>d</italic></sub> represent the parameters of the feature extractor <italic>F</italic><sub>&#x003B8;</sub>, the emotion classifier <italic>G</italic><sub>&#x003D5;</sub>, and the domain classifier <italic>D</italic><sub>&#x003C8;</sub>, respectively. <inline-formula><mml:math id="M52"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>e</mml:mi><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the emotion classification loss, while <inline-formula><mml:math id="M53"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>o</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> represents the domain classification loss. The emotion samples are denoted by <italic>x</italic><sub><italic>i</italic></sub>, and <inline-formula><mml:math id="M54"><mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> represents their corresponding true emotion labels. Additionally, <inline-formula><mml:math id="M55"><mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> represents their corresponding domain labels, where <inline-formula><mml:math id="M56"><mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> indicates that the sample <italic>x</italic><sub><italic>i</italic></sub> comes from the source domain, and <inline-formula><mml:math id="M57"><mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> indicates that the sample <italic>x</italic><sub><italic>i</italic></sub> comes from the target domain.</p>
<p>The model first optimizes the parameters &#x003B8;<sub><italic>f</italic></sub> and &#x003D5;<sub><italic>y</italic></sub> of the feature extractor <italic>F</italic><sub>&#x003B8;</sub> and emotion classifier <italic>G</italic><sub>&#x003D5;</sub> by minimizing the classification loss and the feature extractor loss. This is achieved through the following formula, as shown in <xref ref-type="disp-formula" rid="E27">Equation 27</xref>:</p>
<disp-formula id="E27"><label>(27)</label><mml:math id="M58"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:mo class="qopname">arg</mml:mo><mml:mo class="qopname">min</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munder><mml:mrow><mml:mi mathvariant="script">E</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Then, the model optimizes the parameters &#x003C8;<italic>d</italic> of the domain classifier <italic>D&#x003C8;</italic> by maximizing its loss. This is achieved through the following formula, as shown in <xref ref-type="disp-formula" rid="E28">Equation 28</xref>:</p>
<disp-formula id="E28"><label>(28)</label><mml:math id="M59"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:mo class="qopname">arg</mml:mo><mml:mo class="qopname">max</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munder><mml:mrow><mml:mi mathvariant="script">E</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The two steps mentioned above are alternated until the network converges. During the domain adaptive learning process, a gradient reversal layer is employed to induce the feature extractor to learn adversarial feature representations, as shown in <xref ref-type="disp-formula" rid="E29">Equation 29</xref>:</p>
<disp-formula id="E29"><label>(29)</label><mml:math id="M60"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003BB;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>x</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>During backpropagation, the gradient reversal is achieved by multiplying the gradient with a negative identity matrix, as shown in <xref ref-type="disp-formula" rid="E30">Equation 30</xref>.</p>
<disp-formula id="E30"><label>(30)</label><mml:math id="M61"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mtext class="textrm" mathvariant="normal">d</mml:mtext><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003BB;</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">d</mml:mtext><mml:mi>x</mml:mi></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:mi>I</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
</sec>
<sec id="s4">
<title>3 Experiments</title>
<sec>
<title>3.1 Datasets and settings</title>
<p>The study utilizes the SEED dataset (Zheng and Lu, <xref ref-type="bibr" rid="B36">2015</xref>) and the SEED-IV dataset (Zheng et al., <xref ref-type="bibr" rid="B35">2018</xref>) for research purposes. Both of these datasets are publicly available datasets used for EEG-based emotion recognition. The SEED dataset includes 15 Chinese movie clips as stimuli for the experiments. These movie clips contain three types of emotions: positive, neutral, and negative. Each clip has a duration of &#x0007E;4 min. There are a total of 15 trials in each experiment. In a session, there is a 5-s cue before each clip, followed by a self-assessment period of 45 s, and then a 15-s rest after each clip. Two movie clips with the same emotion are not presented consecutively. The EEG signals are collected using a 62-channel ESI Neuroscan system. The SEED-IV dataset comprises 72 movie clips as experimental stimuli. These movie clips include four types of emotions: happy, sad, fear, and neutral. A total of 15 participants took part in the experiment. For each participant, three experiments are conducted on different days, each containing 24 trials. In each trial, the participant watched one of the movie clips, while their EEG signals were recorded using a 62-channel ESI Neuroscan system. The EEG signals from 62 channels are recorded using the ESI Neuroscan system at a sampling rate of 1,000 Hz, which is downsampled to 200 Hz. Band-pass filtering is applied to the EEG data to remove noise and artifacts, and features such as DE are extracted from each segment in five frequency bands (&#x003B4;: 1&#x0007E;4Hz, &#x003B8;: 4&#x0007E;8Hz, &#x003B1;: 8&#x0007E;14Hz, &#x003B2;: 14&#x0007E;31Hz, &#x003B3;: 31&#x0007E;50Hz).</p>
<p>We train and test the DSP-EmotionNet model using a Tesla V100-SXM2-32GB GPU and implement it using the PyTorch framework. The training is conducted using an Adam optimizer, and the learning rate is set to 5e-4. The batch size is set to 64, and the dropout rate is set to 0.7. The number of classes to classify for the SEED dataset is 3, while for the SEED-IV dataset, it is 4. We adopt the leave-one-subject-out (LOSO) cross-validation strategy to partition the dataset. Specifically, we use all data from 14 subjects as the training set. The remaining 1 subject is treated as an unknown subject and used as the test set. The cross-entropy loss is used as a loss function in this paper. The summary of the hyper-parameter settings is as shown in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>The settings of hyper-parameters on the SEED and SEED-IV datasets are summarized.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="left"><bold>SEED</bold></th>
<th valign="top" align="left"><bold>SEED-IV</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Optimizer</td>
<td valign="top" align="left">Adam</td>
<td valign="top" align="left">Adam</td>
</tr> <tr>
<td valign="top" align="left">Learning rate</td>
<td valign="top" align="left">5e-4</td>
<td valign="top" align="left">5e-4</td>
</tr> <tr>
<td valign="top" align="left">Number of classes</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">4</td>
</tr> <tr>
<td valign="top" align="left">Batch size</td>
<td valign="top" align="left">64</td>
<td valign="top" align="left">64</td>
</tr> <tr>
<td valign="top" align="left">Loss function</td>
<td valign="top" align="left">Cross-entropy</td>
<td valign="top" align="left">Cross-entropy</td>
</tr>
<tr>
<td valign="top" align="left">Dropout rate</td>
<td valign="top" align="left">0.7</td>
<td valign="top" align="left">0.7</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<title>3.2 Baseline methods</title>
<p>In order to evaluate the effectiveness of the proposed model, a comparative analysis is conducted with several baseline methods using the SEED and SEED IV datasets. Brief introductions to each of these methods are provided below.</p>
<list list-type="bullet">
<list-item><p>SVM (Suykens and Vandewalle, <xref ref-type="bibr" rid="B22">1999</xref>): Support vector machine utilizes the least squares to perform classification.</p></list-item>
<list-item><p>RF (Breiman, <xref ref-type="bibr" rid="B4">2001</xref>): Random forest is an ensemble learning method that integrates numerous decision trees to improve classification accuracy.</p></list-item>
<list-item><p>MLP (Rumelhart et al., <xref ref-type="bibr" rid="B21">1986</xref>): A multilayer perceptron represents a fundamental type of feedforward neural network, characterized by its layered structure of neurons arranged in a sequence from input to output.</p></list-item>
<list-item><p>STRNN (Zhang et al., <xref ref-type="bibr" rid="B32">2019</xref>): The proposed framework, known as spatial&#x02013;temporal recurrent neural network (STRNN), integrates spatial and temporal data for the effective classification of human emotions.</p></list-item>
<list-item><p>3D-CNN (Zhao et al., <xref ref-type="bibr" rid="B34">2020</xref>): It introduced a 3D convolutional neural network model for emotion recognition using EEG signals, which automatically extracted spatial-temporal features to achieve high classification accuracy.</p></list-item>
<list-item><p>MMResLSTM (Ma et al., <xref ref-type="bibr" rid="B18">2019</xref>): It proposed a Multimodal Residual LSTM Network for emotion recognition, which leveraged shared weights between different modalities to capture temporal correlations in EEG signals, thus achieving high classification accuracy.</p></list-item>
<list-item><p>CDCN (Gao et al., <xref ref-type="bibr" rid="B10">2021</xref>): It proposed a channel-fused dense convolutional network for EEG-based emotion recognition. This network utilizes convolutional and dense structures to process the temporal and electrode-related features of EEG signals, enhancing the model&#x00027;s ability to capture time dependencies and electrode correlations.</p></list-item>
<list-item><p>ACRNN (Tao et al., <xref ref-type="bibr" rid="B24">2020</xref>): It proposed an attention-based convolutional recurrent neural network for EEG-based emotion recognition, which utilizes channel-wise attention to dynamically weigh channels and incorporates self-attention to improve feature extraction from EEG signals.</p></list-item>
<list-item><p>STFFNN (Wang et al., <xref ref-type="bibr" rid="B30">2022</xref>): It introduced the Spatial-Temporal Feature Fusion Neural Network for EEG-based emotion recognition. This network combines spatial dependency learning, temporal feature learning, and feature fusion using convolutional neural networks and bidirectional LSTM, aiming to enhance emotion recognition accuracy.</p></list-item>
<list-item><p>TSception (Ding et al., <xref ref-type="bibr" rid="B7">2022</xref>): It proposed a novel multi-scale convolutional neural network for EEG-based emotion recognition, which captures both the temporal dynamics and spatial asymmetry of brain activity.</p></list-item>
<list-item><p>MetaEmotionNet (Ning et al., <xref ref-type="bibr" rid="B19">2023</xref>): It integrates spatial-frequency-temporal features into a unified network architecture and utilizes meta-learning to achieve rapid adaptation to new tasks.</p></list-item>
</list>
</sec>
<sec>
<title>3.3 Experimental results and analysis</title>
<p><xref ref-type="table" rid="T2">Tables 2</xref>, <xref ref-type="table" rid="T3">3</xref> presents the cross-subject experimental results on the SEED and SEED-IV datasets, showcasing the average accuracy (ACC) and standard deviation (STD) of both the reference approaches and the proposed DSP-EmotionNet framework for emotion recognition based on EEG signals. Across the SEED dataset, our approach surpasses alternative methodologies in the inter-subject transfer scenario, achieving an ACC of 0.825 with an STD of 0.076. Regarding the SEED-IV dataset, which involves a four-category classification task, the performance of our technique is relatively lower compared to the SEED dataset. Specifically, for the SEED-IV dataset, our technique achieves an ACC of 0.659, with an STD of 0.078.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Performance comparison between the baseline methods and the proposed DSP-EmotionNet on the SEED datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>ACC/STD</bold></th>
<th valign="top" align="left"><bold>F1-score/STD</bold></th>
<th valign="top" align="left"><bold>Kappa/STD</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">RF (Breiman, <xref ref-type="bibr" rid="B4">2001</xref>)</td>
<td valign="top" align="left">0.533/0.087</td>
<td valign="top" align="left">0.487/0.115</td>
<td valign="top" align="left">0.299/0.129</td>
</tr> <tr>
<td valign="top" align="left">SVM (Suykens and Vandewalle, <xref ref-type="bibr" rid="B22">1999</xref>)</td>
<td valign="top" align="left">0.531/0.109</td>
<td valign="top" align="left">0.468/0.137</td>
<td valign="top" align="left">0.296/0.163</td>
</tr> <tr>
<td valign="top" align="left">MLP (Rumelhart et al., <xref ref-type="bibr" rid="B21">1986</xref>)</td>
<td valign="top" align="left">0.701/0.094</td>
<td valign="top" align="left">0.674/0.123</td>
<td valign="top" align="left">0.550/0.142</td>
</tr> <tr>
<td valign="top" align="left">STRNN (Zhang et al., <xref ref-type="bibr" rid="B32">2019</xref>)</td>
<td valign="top" align="left">0.732/0.111</td>
<td valign="top" align="left">0.723/0.116</td>
<td valign="top" align="left">0.598/0.166</td>
</tr> <tr>
<td valign="top" align="left">3D-CNN (Zhao et al., <xref ref-type="bibr" rid="B34">2020</xref>)</td>
<td valign="top" align="left">0.742/0.078</td>
<td valign="top" align="left">0.738/0.079</td>
<td valign="top" align="left">0.613/0.116</td>
</tr> <tr>
<td valign="top" align="left">MMResLSTM (Ma et al., <xref ref-type="bibr" rid="B18">2019</xref>)</td>
<td valign="top" align="left">0.744/0.104</td>
<td valign="top" align="left">0.713/0.106</td>
<td valign="top" align="left">0.616/0.157</td>
</tr> <tr>
<td valign="top" align="left">CDCN (Gao et al., <xref ref-type="bibr" rid="B10">2021</xref>)</td>
<td valign="top" align="left">0.693/0.094</td>
<td valign="top" align="left">0.668/0.116</td>
<td valign="top" align="left">0.540/0.142</td>
</tr> <tr>
<td valign="top" align="left">ACRNN (Tao et al., <xref ref-type="bibr" rid="B24">2020</xref>)</td>
<td valign="top" align="left">0.763/0.081</td>
<td valign="top" align="left">0.739/0.093</td>
<td valign="top" align="left">0.644/0.121</td>
</tr> <tr>
<td valign="top" align="left">STFFNN (Wang et al., <xref ref-type="bibr" rid="B30">2022</xref>)</td>
<td valign="top" align="left">0.720/0.090</td>
<td valign="top" align="left">0.0710/0.095</td>
<td valign="top" align="left">0.579/0.136</td>
</tr> <tr>
<td valign="top" align="left">TSception (Ding et al., <xref ref-type="bibr" rid="B7">2022</xref>)</td>
<td valign="top" align="left">0.643/0.098</td>
<td valign="top" align="left">0.636/0.107</td>
<td valign="top" align="left">0.465/0.148</td>
</tr> <tr>
<td valign="top" align="left">MetaEmotionNet (Ning et al., <xref ref-type="bibr" rid="B19">2023</xref>)</td>
<td valign="top" align="left">0.775/0.088</td>
<td valign="top" align="left">0.772/0.087</td>
<td valign="top" align="left">0.663/0.132</td>
</tr>
<tr>
<td valign="top" align="left"><bold>DSP-EmotionNet</bold></td>
<td valign="top" align="left"><bold>0.825/0.076</bold></td>
<td valign="top" align="left"><bold>0.824/0.072</bold></td>
<td valign="top" align="left"><bold>0.739/0.126</bold></td>
</tr></tbody>
</table>
</table-wrap>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Performance comparison between the baseline methods and the proposed DSP-EmotionNet on the SEED-IV datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>ACC/STD</bold></th>
<th valign="top" align="left"><bold>F1-score/STD</bold></th>
<th valign="top" align="left"><bold>Kappa/STD</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">RF (Breiman, <xref ref-type="bibr" rid="B4">2001</xref>)</td>
<td valign="top" align="left">0.347/0.066</td>
<td valign="top" align="left">0.297/0.083</td>
<td valign="top" align="left">0.134/0.083</td>
</tr> <tr>
<td valign="top" align="left">SVM (Suykens and Vandewalle, <xref ref-type="bibr" rid="B22">1999</xref>)</td>
<td valign="top" align="left">0.411/0.074</td>
<td valign="top" align="left">0.348/0.075</td>
<td valign="top" align="left">0.209/0.094</td>
</tr> <tr>
<td valign="top" align="left">MLP (Rumelhart et al., <xref ref-type="bibr" rid="B21">1986</xref>)</td>
<td valign="top" align="left">0.507/0.063</td>
<td valign="top" align="left">0.397/0.083</td>
<td valign="top" align="left">0.328/0.089</td>
</tr> <tr>
<td valign="top" align="left">STRNN (Zhang et al., <xref ref-type="bibr" rid="B32">2019</xref>)</td>
<td valign="top" align="left">0.532/0.074</td>
<td valign="top" align="left">0.517/0.076</td>
<td valign="top" align="left">0.372/0.100</td>
</tr> <tr>
<td valign="top" align="left">3D-CNN (Zhao et al., <xref ref-type="bibr" rid="B34">2020</xref>)</td>
<td valign="top" align="left">0.541/0.109</td>
<td valign="top" align="left">0.507/0.133</td>
<td valign="top" align="left">0.384/0.149</td>
</tr> <tr>
<td valign="top" align="left">MMResLSTM (Ma et al., <xref ref-type="bibr" rid="B18">2019</xref>)</td>
<td valign="top" align="left">0.511/0.114</td>
<td valign="top" align="left">0.461/0.112</td>
<td valign="top" align="left">0.348/0.149</td>
</tr> <tr>
<td valign="top" align="left">CDCN (Gao et al., <xref ref-type="bibr" rid="B10">2021</xref>)</td>
<td valign="top" align="left">0.545/0.131</td>
<td valign="top" align="left">0.521/0.150</td>
<td valign="top" align="left">0.393/0.172</td>
</tr> <tr>
<td valign="top" align="left">ACRNN (Tao et al., <xref ref-type="bibr" rid="B24">2020</xref>)</td>
<td valign="top" align="left">0.492/0.092</td>
<td valign="top" align="left">0.462/0.082</td>
<td valign="top" align="left">0.296/0.110</td>
</tr> <tr>
<td valign="top" align="left">STFFNN (Wang et al., <xref ref-type="bibr" rid="B30">2022</xref>)</td>
<td valign="top" align="left">0.567/0.073</td>
<td valign="top" align="left">0.550/0.075</td>
<td valign="top" align="left">0.418/0.099</td>
</tr> <tr>
<td valign="top" align="left">TSception (Ding et al., <xref ref-type="bibr" rid="B7">2022</xref>)</td>
<td valign="top" align="left">0.562/0.095</td>
<td valign="top" align="left">0.558/0.096</td>
<td valign="top" align="left">0.414/0.128</td>
</tr> <tr>
<td valign="top" align="left">MetaEmotionNet (Ning et al., <xref ref-type="bibr" rid="B19">2023</xref>)</td>
<td valign="top" align="left">0.612/0.083</td>
<td valign="top" align="left">0.589/0.102</td>
<td valign="top" align="left">0.479/0.114</td>
</tr>
<tr>
<td valign="top" align="left"><bold>DSP-EmotionNet</bold></td>
<td valign="top" align="left"><bold>0.659/0.078</bold></td>
<td valign="top" align="left"><bold>0.652/0.119</bold></td>
<td valign="top" align="left"><bold>0.542/0.108</bold></td>
</tr></tbody>
</table>
</table-wrap>
<p>Our proposed DSP-EmotionNet model excels in emotion recognition tasks. In contrast, traditional machine learning methods such as SVM and RF perform relatively poorly, primarily due to their inability to capture the rich information present in EEG signals. Deep learning-based CNN and RNN can better extract deep temporal or spatial features, hence methods like STRNN, 3D-CNN, CDCN, and MMResLSTM based on deep learning outperform traditional machine learning methods in terms of performance. Recently proposed methods like ACRNN, STFFNN, and TSception model both the temporal and spatial dimensions of EEG signals, improving classification stability by introducing attention mechanisms, integrating discriminative features, and capturing temporal dynamics, resulting in better results on the SEED or SEED-IV datasets. Our proposed DSP-EmotionNet model not only captures spatial local features of EEG signals but also captures the correlations between different regions of EEG signals. MetaEmotionNet utilizes two streams of spatial-temporal and spatial-frequency information, along with attention mechanisms, to comprehensively extract spatial-frequency-temporal features of EEG signals, thus achieving optimal performance in metrics. Moreover, meta-learning methods effectively enhance the adaptability of the model. Compared to the MetaEmotionNet method, our proposed DSP-EmotionNet model employs Domain Adversarial Neural Networks (DANN) to improve the generalization ability of the model. DANN technology enables the model to gradually adapt to the data distribution of new domains during the training process, thereby enhancing its generalization ability on new domains and improving the recognition rate of the model in cross-subject EEG emotion recognition tasks. Our proposed DSP-EmotionNet model not only exhibits high performance in emotion recognition tasks but also demonstrates stronger adaptability and generalization capabilities. For more detailed classification results, the confusion matrices of the proposed DSP-EmotionNet are respectively shown in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>The confusion matrices of DSP-EmotionNet on SEED and SEED IV datasets.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnhum-18-1471634-g0004.tif"/>
</fig>
</sec>
<sec>
<title>3.4 Ablation experiments</title>
<p>To validate the impact of different components in our proposed model on EEG emotion recognition tasks, we conduct ablation experiments on the SEED and SEED IV datasets. Our proposed method, named DSP-EmotionNet, consists primarily of three parts: spatial activity feature extractor, spatial topological feature extractor, and domain adversarial neural network. To verify the effectiveness of these three key components in our approach, we conduct ablation experiments on DSP-EmotionNet. <xref ref-type="table" rid="T4">Table 4</xref> illustrate the impact of these three key components of DSP-EmotionNet on cross-subject EEG emotion recognition tasks. &#x0201C;SAE&#x0201D; denotes using only the spatial activity feature extractor for cross-subject EEG emotion recognition tasks. &#x0201C;STE&#x0201D; denotes using only the spatial topological feature extractor for cross-subject EEG emotion recognition tasks. &#x0201C;SAE-STE&#x0201D; represents combining the spatial activity feature extractor and spatial topological feature extractor for cross-subject EEG emotion recognition tasks, excluding domain adversarial neural network. &#x0201C;SAE-DANN&#x0201D; represents combining the spatial activity feature extractor with domain adversarial neural network for cross-subject EEG emotion recognition tasks, excluding the spatial topological feature extractor. &#x0201C;STE-DANN&#x0201D; represents combining the spatial topological feature extractor with domain adversarial neural network for cross-subject EEG emotion recognition tasks, excluding the spatial activity feature extractor.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Ablation experiments on the major components of DSP-EmotionNet were conducted on the SEED and SEED IV datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="left"><bold>Accuracy (%)</bold></th>
<th valign="top" align="left"><bold>Another Metric (%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SAE</td>
<td valign="top" align="left">69.7%</td>
<td valign="top" align="left">51.4%</td>
</tr> <tr>
<td valign="top" align="left">STE</td>
<td valign="top" align="left">72.6%</td>
<td valign="top" align="left">55.6%</td>
</tr> <tr>
<td valign="top" align="left">SAE-STE</td>
<td valign="top" align="left">77.1%</td>
<td valign="top" align="left">60.1%</td>
</tr> <tr>
<td valign="top" align="left">SAE-DANN</td>
<td valign="top" align="left">72.2%</td>
<td valign="top" align="left">56.1%</td>
</tr> <tr>
<td valign="top" align="left">STE-DANN</td>
<td valign="top" align="left">76.7%</td>
<td valign="top" align="left">59.7%</td>
</tr>
<tr>
<td valign="top" align="left">DSP-EmotionNet</td>
<td valign="top" align="left">82.5%</td>
<td valign="top" align="left">65.9%</td>
</tr></tbody>
</table>
</table-wrap>
<p>The accuracy of &#x0201C;SAE&#x0201D; on the SEED dataset is 69.7%, and on the SEED IV dataset, it is 51.4%. In comparison, &#x0201C;STE&#x0201D; achieves an accuracy of 72.6% on the SEED dataset and 55.6% on the SEED IV dataset. This indicates that the spatial topological feature extractor performs better than the spatial activity feature extractor in cross-subject EEG emotion recognition tasks. Furthermore, &#x0201C;SAE-STE&#x0201D; outperforms &#x0201C;SAE&#x0201D; and &#x0201C;STE&#x0201D; on both the SEED and SEED IV datasets, demonstrating the effectiveness of combining EEG spatial activity features and EEG spatial topological features. On the SEED dataset, &#x0201C;SAE-DANN&#x0201D; and &#x0201C;STE-DANN&#x0201D; achieve accuracies of 72.2% and 76.7%, respectively, outperforming &#x0201C;SAE&#x0201D; and &#x0201C;STE&#x0201D;. On the SEED IV dataset, &#x0201C;SAE-DANN&#x0201D; and &#x0201C;STE-DANN&#x0201D; achieve accuracies of 56.1% and 59.7%, respectively, also outperforming &#x0201C;SAE&#x0201D; and &#x0201C;STE&#x0201D;. This indicates that DANN technology enables the model to gradually adapt to the data distribution of new domains during training, thereby improving the generalization ability of the model on new domains. DSP-EmotionNet achieves accuracies of 82.5% and 65.9% on the SEED and SEED IV datasets, respectively, surpassing the results of other methods in the ablation experiments. These results collectively demonstrate that the integration of feature fusion and domain adaptation contributes to the enhancement of model recognition performance in cross-subject EEG emotion recognition tasks.</p>
<p>To visually understand the effectiveness of DSP-EmotionNet, we randomly select a participant from the SEED dataset and use their EEG samples as the test set. We visualize the data using t-SNE (Van der Maaten and Hinton, <xref ref-type="bibr" rid="B25">2008</xref>) scatter plots, as shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. Specifically, we select six methods for visualization experiments: &#x0201C;SAE&#x0201D;, &#x0201C;STE&#x0201D;, &#x0201C;SAE-STE&#x0201D;, &#x0201C;SAE-DANN&#x0201D;, &#x0201C;STE-DANN&#x0201D;, and &#x0201C;DSP-EmotionNet.&#x0201D; Data points are color-coded to represent three different emotions: negative emotions in red, neutral emotions in green, and positive emotions in blue. It is worth noting that the range of the data after dimensionality reduction varies depending on the participant. Here, we only show the visualization results of our method. The figure displays scatter plots for the six different methods. As shown in <xref ref-type="fig" rid="F5">Figure 5a</xref>, data points corresponding to the three emotions clearly intermingle, exhibiting significant overlap. This suggests that &#x0201C;SAE&#x0201D; may face challenges in distinguishing emotions in cross-subject EEG emotion recognition tasks. In <xref ref-type="fig" rid="F5">Figure 5b</xref>, the clusters appear somewhat separated, but there is still considerable overlap between emotions, especially between negative and neutral states. This indicates that although &#x0201C;SAE-DANN&#x0201D; improves upon &#x0201C;SAE&#x0201D;, it may not be sufficient for optimal emotion recognition on its own. In <xref ref-type="fig" rid="F5">Figure 5c</xref>, clusters for each emotion seem more distinct compared to &#x0201C;SAE&#x0201D;, indicating that &#x0201C;STE&#x0201D; significantly enhances the discernibility of emotions. As shown in <xref ref-type="fig" rid="F5">Figure 5d</xref>, clusters for positive and neutral emotions are notably different and well-separated. In <xref ref-type="fig" rid="F5">Figure 5e</xref>, clusters for each emotion perform better than individual &#x0201C;SAE&#x0201D; and &#x0201C;STE&#x0201D; methods, indicating that &#x0201C;SAE-STE&#x0201D; can extract more effective features. As shown in <xref ref-type="fig" rid="F5">Figure 5f</xref>, DSP-EmotionNet exhibits notably distinct and well-separated clusters for each emotion compared to &#x0201C;SAE-STE,&#x0201D; particularly the positive (blue) cluster, which is almost completely isolated from the other two emotions. This further emphasizes that the integration of feature fusion and domain adaptation significantly contributes to enhancing the recognition performance of the model in cross-subject EEG emotion recognition tasks.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>The performance of various methods in cross-subject EEG emotion recognition tasks was visualized using t-SNE. <bold>(a)</bold> SAE. <bold>(b)</bold> SAE-DANN. <bold>(c)</bold> STE. <bold>(d)</bold> STE-DANN. <bold>(e)</bold> SAE-STE. <bold>(f)</bold> DSP-EmotionNet.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnhum-18-1471634-g0005.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="conclusions" id="s5">
<title>4 Conclusion</title>
<p>In this paper, we introduce a domain adaptation EEG signal spatial feature perception network, named DSP-EmotionNet, for cross-subject EEG emotion recognition tasks. Initially, we extract DE features and spatially map them based on electrode position distribution to generate representations of EEG signal spatial activity features, and similarly for spatial graph mapping to produce representations of EEG signal spatial topological features. These two features serve as the input for our proposed model. Then, we design a dual-branch network, named SATFEM, utilizing a spatial activity feature extractor branch to capture EEG signal spatial activity features and a spatial topological feature extractor branch to capture EEG signal spatial topological features. The features extracted from both branches are effectively fused and classified in the feature fusion and classification layer. Finally, we employ SATFEM as the feature extractor and design a domain adaptation network to better adapt the model to the features of the target domain, thereby enhancing the accuracy of the model on cross-subject EEG emotion recognition tasks. The proposed DSP-EmotionNet achieved average recognition accuracies of 82.5% and 65.9% on the SEED and SEED-IV datasets, respectively, surpassing state-of-the-art methods. To evaluate the impact of different components in DSP-EmotionNet on the EEG emotion recognition task, we conduct ablation experiments on the SEED and SEED-IV datasets. The experimental results show that the combination of the spatial activity feature extractor branch and the spatial topological feature extractor branch can effectively enhance the capability of the model for feature extraction, and applying a domain adaptation network allows the model to better adapt to the features of the target domain, improving the generalizability of the model. The proposed DSP-EmotionNet represents a new approach to cross-subject EEG emotion recognition. This method can also be easily applied to other EEG classification tasks, such as motor imagery and sleep stage classification. However, the current model still has some limitations in practical applications. For instance, the proposed dual-branch structure has higher computational complexity compared to single-branch models, and it also lacks the capability for real-time online processing. In future work, we will investigate model compression and acceleration, as well as the real-time online capabilities of DSP-EmotionNet in cross-subject EEG emotion recognition, aiming to further enhance the generalizability and practicality of the model.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>WL: Formal analysis, Methodology, Software, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. XZ: Conceptualization, Data curation, Formal analysis, Funding acquisition, Software, Writing &#x02013; review &#x00026; editing. LX: Conceptualization, Investigation, Methodology, Project administration, Resources, Validation, Writing &#x02013; review &#x00026; editing. HM: Investigation, Methodology, Project administration, Resources, Writing &#x02013; review &#x00026; editing. T-PT: Conceptualization, Funding acquisition, Methodology, Project administration, Resources, Software, Supervision, Validation, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This research was funded by Henan Provincial Science and Technology Research Project, China (Grant Nos. 232102240091, 232102240089, and 242102241064) and Key Scientific Research Project of Henan Province Higher Education Institutions, China (Grant Nos. 23B520033 and 25B580004).</p>
</sec>
<ack><p>The authors express their gratitude for the diligent efforts of all the reviewers and editorial staff. The authors thank the Shanghai Jiao Tong University for providing the Emotion EEG Datasets. We used generative AI, specifically ChatGPT, only for grammar correction in the manuscript.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Asadzadeh</surname> <given-names>S.</given-names></name> <name><surname>Rezaii</surname> <given-names>T. Y.</given-names></name> <name><surname>Beheshti</surname> <given-names>S.</given-names></name> <name><surname>Meshgini</surname> <given-names>S.</given-names></name></person-group> (<year>2023</year>). <article-title>Accurate emotion recognition utilizing extracted EEG sources as graph neural network nodes</article-title>. <source>Cogn. Comput</source>. <volume>15</volume>, <fpage>176</fpage>&#x02013;<lpage>189</lpage>. <pub-id pub-id-type="doi">10.1007/s12559-022-10077-5</pub-id><pub-id pub-id-type="pmid">35917638</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Atkinson</surname> <given-names>J.</given-names></name> <name><surname>Campos</surname> <given-names>D.</given-names></name></person-group> (<year>2016</year>). <article-title>Improving bci-based emotion recognition by combining EEG feature selection and Kernel classifiers</article-title>. <source>Expert Syst. Appl</source>. <volume>47</volume>, <fpage>35</fpage>&#x02013;<lpage>41</lpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2015.10.049</pub-id></citation>
</ref>
<ref id="B3">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Bahari</surname> <given-names>F.</given-names></name> <name><surname>Janghorbani</surname> <given-names>A.</given-names></name></person-group> (<year>2013</year>). <article-title>&#x0201C;EEG-based emotion recognition using recurrence plot analysis and K nearest neighbor classifier,&#x0201D;</article-title> in <source>2013 20th Iranian Conference on Biomedical Engineering (ICBME)</source> (<publisher-loc>Tehran</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>228</fpage>&#x02013;<lpage>233</lpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Breiman</surname> <given-names>L.</given-names></name></person-group> (<year>2001</year>). <article-title>Random forests</article-title>. <source>Machine Learn</source>. <volume>45</volume>, <fpage>5</fpage>&#x02013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chakravarthi</surname> <given-names>B.</given-names></name> <name><surname>Ng</surname> <given-names>S.-C.</given-names></name> <name><surname>Ezilarasan</surname> <given-names>M.</given-names></name> <name><surname>Leung</surname> <given-names>M.-F.</given-names></name></person-group> (<year>2022</year>). <article-title>Eeg-based emotion recognition using hybrid CNN and LSTM classification</article-title>. <source>Front. Comput. Neurosci</source>. <volume>16</volume>:<fpage>1019776</fpage>. <pub-id pub-id-type="doi">10.3389/fncom.2022.1019776</pub-id><pub-id pub-id-type="pmid">36277613</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cimtay</surname> <given-names>Y.</given-names></name> <name><surname>Ekmekcioglu</surname> <given-names>E.</given-names></name> <name><surname>Caglar-Ozhan</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>Cross-subject multimodal emotion recognition based on hybrid fusion</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>168865</fpage>&#x02013;<lpage>168878</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2020.3023871</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>Y.</given-names></name> <name><surname>Robinson</surname> <given-names>N.</given-names></name> <name><surname>Zhang</surname> <given-names>S.</given-names></name> <name><surname>Zeng</surname> <given-names>Q.</given-names></name> <name><surname>Guan</surname> <given-names>C.</given-names></name></person-group> (<year>2022</year>). <article-title>Tsception: capturing temporal dynamics and spatial asymmetry from EEG for emotion recognition</article-title>. <source>IEEE Trans. Affect. Comput</source>. <volume>2022</volume>:<fpage>3169001</fpage>. <pub-id pub-id-type="doi">10.1109/TAFFC.2022.3169001</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Doma</surname> <given-names>V.</given-names></name> <name><surname>Pirouz</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>A comparative analysis of machine learning methods for emotion recognition using EEG and peripheral physiological signals</article-title>. <source>J. Big Data</source> <volume>7</volume>, <fpage>1</fpage>&#x02013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.1186/s40537-020-00289-7</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ganin</surname> <given-names>Y.</given-names></name> <name><surname>Lempitsky</surname> <given-names>V.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Unsupervised domain adaptation by backpropagation,&#x0201D;</article-title> in <source>Proceedings of the 32nd International Conference on Machine Learning</source>, eds. F. Bach and D. Blei (<publisher-loc>Lille</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>1180</fpage>&#x02013;<lpage>1189</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Yang</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Ma</surname> <given-names>K.</given-names></name> <name><surname>Chen</surname> <given-names>G.</given-names></name></person-group> (<year>2021</year>). <article-title>A channel-fused dense convolutional network for EEG-based emotion recognition</article-title>. <source>IEEE Trans. Cogn. Dev. Syst</source>. <volume>2020</volume>, <fpage>945</fpage>&#x02013;<lpage>954</lpage>. <pub-id pub-id-type="doi">10.1109/TCDS.2020.2976112</pub-id><pub-id pub-id-type="pmid">32260445</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jia</surname> <given-names>Z.</given-names></name> <name><surname>Lin</surname> <given-names>Y.</given-names></name> <name><surname>Cai</surname> <given-names>X.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Gou</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Sst-emotionnet: Spatial-spectral-temporal based attention 3d dense network for eeg emotion recognition,&#x0201D;</article-title> in <source>Proceedings of the 28th ACM International Conference on Multimedia</source>, <fpage>2909</fpage>&#x02013;<lpage>2917</lpage>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jia</surname> <given-names>Z.</given-names></name> <name><surname>Lin</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Feng</surname> <given-names>Z.</given-names></name> <name><surname>Xie</surname> <given-names>X.</given-names></name> <name><surname>Chen</surname> <given-names>C.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;HetEmotionNet: two-stream heterogeneous graph recurrent neural network for multi-modal emotion recognition,&#x0201D;</article-title> in <source>Proceedings of the 29th ACM International Conference on Multimedia</source>, <fpage>1047</fpage>&#x02013;<lpage>1056</lpage>.</citation>
</ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Jin</surname> <given-names>Y.-M.</given-names></name> <name><surname>Luo</surname> <given-names>Y.-D.</given-names></name> <name><surname>Zheng</surname> <given-names>W.-L.</given-names></name> <name><surname>Lu</surname> <given-names>B.-L.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;EEG-based emotion recognition using domain adaptation network,&#x0201D;</article-title> in <source>2017 International Conference on Orange Technologies (ICOT)</source> (<publisher-loc>Singapore</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>222</fpage>&#x02013;<lpage>225</lpage>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kwon</surname> <given-names>Y.-H.</given-names></name> <name><surname>Shin</surname> <given-names>S.-B.</given-names></name> <name><surname>Kim</surname> <given-names>S.-D.</given-names></name></person-group> (<year>2018</year>). <article-title>Electroencephalography based fusion two-dimensional (2D)-convolution neural networks (CNN) model for emotion recognition system</article-title>. <source>Sensors</source> <volume>18</volume>:<fpage>1383</fpage>. <pub-id pub-id-type="doi">10.3390/s18051383</pub-id><pub-id pub-id-type="pmid">29710869</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Qiu</surname> <given-names>S.</given-names></name> <name><surname>Du</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>He</surname> <given-names>H.</given-names></name></person-group> (<year>2019</year>). <article-title>Domain adaptation for EEG emotion recognition based on latent representation similarity</article-title>. <source>IEEE Trans. Cogn. Dev. Syst</source>. <volume>12</volume>, <fpage>344</fpage>&#x02013;<lpage>353</lpage>. <pub-id pub-id-type="doi">10.1109/TCDS.2019.2949306</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>He</surname> <given-names>H.</given-names></name></person-group> (<year>2018</year>). <article-title>Hierarchical convolutional neural networks for EEG-based emotion recognition</article-title>. <source>Cogn. Comput</source>. <volume>10</volume>, <fpage>368</fpage>&#x02013;<lpage>380</lpage>. <pub-id pub-id-type="doi">10.1007/s12559-017-9533-x</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name> <name><surname>Zheng</surname> <given-names>W.</given-names></name> <name><surname>Zong</surname> <given-names>Y.</given-names></name> <name><surname>Qi</surname> <given-names>L.</given-names></name> <name><surname>Cui</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>A novel bi-hemispheric discrepancy model for EEG emotion recognition</article-title>. <source>IEEE Trans. Cogn. Dev. Syst</source>. <volume>13</volume>, <fpage>354</fpage>&#x02013;<lpage>367</lpage>. <pub-id pub-id-type="doi">10.1109/TCDS.2020.2999337</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>J.</given-names></name> <name><surname>Tang</surname> <given-names>H.</given-names></name> <name><surname>Zheng</surname> <given-names>W.-L.</given-names></name> <name><surname>Lu</surname> <given-names>B.-L.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Emotion recognition using multimodal residual LSTM network,&#x0201D;</article-title> in <source>Proceedings of the 27th ACM International Conference on Multimedia</source>, <fpage>176</fpage>&#x02013;<lpage>183</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ning</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Lin</surname> <given-names>Y.</given-names></name> <name><surname>Cai</surname> <given-names>X.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Gou</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>MetaemotionNet: spatial-spectral-temporal based attention 3D dense network with meta-learning for EEG emotion recognition</article-title>. <source>IEEE Trans. Instrument. Measur</source>. <volume>2023</volume>:<fpage>3338676</fpage>. <pub-id pub-id-type="doi">10.1109/TIM.2023.3338676</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ramzan</surname> <given-names>M.</given-names></name> <name><surname>Dawn</surname> <given-names>S.</given-names></name></person-group> (<year>2023</year>). <article-title>Fused CNN-LSTM deep learning emotion recognition model using electroencephalography signals</article-title>. <source>Int. J. Neurosci</source>. <volume>133</volume>, <fpage>587</fpage>&#x02013;<lpage>597</lpage>. <pub-id pub-id-type="doi">10.1080/00207454.2021.1941947</pub-id><pub-id pub-id-type="pmid">34121598</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rumelhart</surname> <given-names>D. E.</given-names></name> <name><surname>Hinton</surname> <given-names>G. E.</given-names></name> <name><surname>Williams</surname> <given-names>R. J.</given-names></name></person-group> (<year>1986</year>). <article-title>Learning internal representations by error propagation, parallel distributed processing, explorations in the microstructure of cognition</article-title>. <source>Biometrika</source> <volume>71</volume>, <fpage>599</fpage>&#x02013;<lpage>607</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Suykens</surname> <given-names>J. A. K.</given-names></name> <name><surname>Vandewalle</surname> <given-names>J.</given-names></name></person-group> (<year>1999</year>). <article-title>Least squares support vector machine classifiers</article-title>. <source>Neural Proc. Lett.</source> <volume>9</volume>, <fpage>293</fpage>&#x02013;<lpage>300</lpage>. <pub-id pub-id-type="doi">10.1023/a:1018628609742</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tan</surname> <given-names>C.</given-names></name> <name><surname>Ceballos</surname> <given-names>G.</given-names></name> <name><surname>Kasabov</surname> <given-names>N.</given-names></name> <name><surname>Puthanmadam Subramaniyam</surname> <given-names>N.</given-names></name></person-group> (<year>2020</year>). <article-title>Fusionsense: emotion classification using feature fusion of multimodal data and deep learning in a brain-inspired spiking neural network</article-title>. <source>Sensors</source> <volume>20</volume>:<fpage>5328</fpage>. <pub-id pub-id-type="doi">10.3390/s20185328</pub-id><pub-id pub-id-type="pmid">32957655</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tao</surname> <given-names>W.</given-names></name> <name><surname>Li</surname> <given-names>C.</given-names></name> <name><surname>Song</surname> <given-names>R.</given-names></name> <name><surname>Cheng</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Wan</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>EEG-based emotion recognition via channel-wise attention and self attention</article-title>. <source>IEEE Trans. Affect. Comput</source>. <volume>2020</volume>:<fpage>3025777</fpage>. <pub-id pub-id-type="doi">10.1109/TAFFC.2020.3025777</pub-id></citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Van der Maaten</surname> <given-names>L.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2008</year>). <article-title>Visualizing data using T-SNE</article-title>. <source>J. Machine Learn. Res</source>. <volume>9</volume>, <fpage>2579</fpage>&#x02013;<lpage>2605</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Velickovic</surname> <given-names>P.</given-names></name> <name><surname>Cucurull</surname> <given-names>G.</given-names></name> <name><surname>Casanova</surname> <given-names>A.</given-names></name> <name><surname>Romero</surname> <given-names>A.</given-names></name> <name><surname>Lio</surname> <given-names>P.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Graph attention networks</article-title>. <source>Stat</source> <volume>2017</volume>:<fpage>10903</fpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1710.10903</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>F.</given-names></name> <name><surname>Wu</surname> <given-names>S.</given-names></name> <name><surname>Zhang</surname> <given-names>W.</given-names></name> <name><surname>Xu</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Emotion recognition with convolutional neural network and EEG-based efdms</article-title>. <source>Neuropsychologia</source> <volume>146</volume>:<fpage>107506</fpage>. <pub-id pub-id-type="doi">10.1016/j.neuropsychologia.2020.107506</pub-id><pub-id pub-id-type="pmid">32497532</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>X.-W.</given-names></name> <name><surname>Nie</surname> <given-names>D.</given-names></name> <name><surname>Lu</surname> <given-names>B.-L.</given-names></name></person-group> (<year>2011</year>). <article-title>&#x0201C;EEG-based emotion recognition using frequency domain features and support vector machines,&#x0201D;</article-title> in <source>Neural Information Processing: 18th International Conference, ICONIP 2011, Shanghai, China, November 13-17, 2011, Proceedings, Part I 18</source> (<publisher-loc>Berlin</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>734</fpage>&#x02013;<lpage>743</lpage>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Ruan</surname> <given-names>Q.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name></person-group> (<year>2021</year>). <article-title>Cross-subject EEG emotion classification based on few-label adversarial domain adaption</article-title>. <source>Exp. Syst. Appl</source>. <volume>185</volume>:<fpage>115581</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2021.115581</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Hu</surname> <given-names>C.</given-names></name> <name><surname>Yin</surname> <given-names>Z.</given-names></name> <name><surname>Song</surname> <given-names>Y.</given-names></name></person-group> (<year>2022</year>). <article-title>Spatial-temporal feature fusion neural network for EEG-based emotion recognition</article-title>. <source>IEEE Trans. Instr. Measur</source>. <volume>71</volume>, <fpage>1</fpage>&#x02013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1109/TIM.2022.3165280</pub-id></citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xing</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Xu</surname> <given-names>T.</given-names></name> <name><surname>Shu</surname> <given-names>L.</given-names></name> <name><surname>Hu</surname> <given-names>B.</given-names></name> <name><surname>Xu</surname> <given-names>X.</given-names></name></person-group> (<year>2019</year>). <article-title>SAE&#x0002B; LSTM: a new framework for emotion recognition from multi-channel EEG</article-title>. <source>Front. Neurorobot</source>. <volume>13</volume>:<fpage>37</fpage>. <pub-id pub-id-type="doi">10.3389/fnbot.2019.00037</pub-id><pub-id pub-id-type="pmid">31244638</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>T.</given-names></name> <name><surname>Zheng</surname> <given-names>W.</given-names></name> <name><surname>Cui</surname> <given-names>Z.</given-names></name> <name><surname>Zong</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name></person-group> (<year>2019</year>). <article-title>Spatial&#x02014;temporal recurrent neural network for emotion recognition</article-title>. <source>IEEE Trans. Cybernet</source>. 839&#x02013;847. <pub-id pub-id-type="doi">10.1109/TCYB.2017.2788081</pub-id><pub-id pub-id-type="pmid">29994572</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Huang</surname> <given-names>D.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Xia</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name></person-group> (<year>2023</year>). <article-title>Self-training maximum classifier discrepancy for EEG emotion recognition</article-title>. <source>CAAI Trans. Intell. Technol</source>. <volume>2023</volume>:<fpage>12174</fpage>. <pub-id pub-id-type="doi">10.1049/cit2.12174</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>J.</given-names></name> <name><surname>Lin</surname> <given-names>J.</given-names></name> <name><surname>Yu</surname> <given-names>D.</given-names></name> <name><surname>Cao</surname> <given-names>X.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;A 3D convolutional neural network for emotion recognition based on EEG signals,&#x0201D;</article-title> in <source>2020 International Joint Conference on Neural Networks (IJCNN)</source> (<publisher-loc>Glasgow</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>6</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname> <given-names>W.-L.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Lu</surname> <given-names>Y.</given-names></name> <name><surname>Lu</surname> <given-names>B.-L.</given-names></name> <name><surname>Cichocki</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>Emotionmeter: a multimodal framework for recognizing human emotions</article-title>. <source>IEEE Trans. Cybernet</source>. <volume>49</volume>, <fpage>1110</fpage>&#x02013;<lpage>1122</lpage>. <pub-id pub-id-type="doi">10.1109/TCYB.2018.2797176</pub-id><pub-id pub-id-type="pmid">29994384</pub-id></citation></ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname> <given-names>W.-L.</given-names></name> <name><surname>Lu</surname> <given-names>B.-L.</given-names></name></person-group> (<year>2015</year>). <article-title>Investigating critical frequency bands and channels for EEG-based emotion recognition with deep neural networks</article-title>. <source>IEEE Trans. Auton. Ment. Dev</source>. <volume>7</volume>, <fpage>162</fpage>&#x02013;<lpage>175</lpage>. <pub-id pub-id-type="doi">10.1109/TAMD.2015.2431497</pub-id></citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>C.</given-names></name> <name><surname>Zhai</surname> <given-names>L.</given-names></name> <name><surname>Jia</surname> <given-names>Z.</given-names></name> <name><surname>Guan</surname> <given-names>C.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name></person-group> (<year>2023</year>). <article-title>Interpretable and robust AI in EEG systems: a survey</article-title>. <source>arXiv preprint arXiv:2304.10755</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2304.10755</pub-id></citation>
</ref>
</ref-list>
</back>
</article>