<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Physiol.</journal-id>
<journal-title>Frontiers in Physiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Physiol.</abbrev-journal-title>
<issn pub-type="epub">1664-042X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1515881</article-id>
<article-id pub-id-type="doi">10.3389/fphys.2025.1515881</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Physiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Multimodal diagnosis of Alzheimer&#x2019;s disease based on resting-state electroencephalography and structural magnetic resonance imaging</article-title>
<alt-title alt-title-type="left-running-head">Liu et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphys.2025.1515881">10.3389/fphys.2025.1515881</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Junxiu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/537248/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wu</surname>
<given-names>Shangxiao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2978343/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Fu</surname>
<given-names>Qiang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1402660/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Luo</surname>
<given-names>Xiwen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2991826/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Luo</surname>
<given-names>Yuling</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/878736/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Qin</surname>
<given-names>Sheng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2395934/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Huang</surname>
<given-names>Yiting</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2991814/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Zhaohui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2991793/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Guangxi Key Laboratory of Brain-inspired Computing and Intelligent Chips</institution>, <institution>School of Electronic and Information Engineering</institution>, <institution>Guangxi Normal University</institution>, <addr-line>Guilin</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Key Laboratory of Nonlinear Circuits and Optical Communications</institution>, <institution>Education Department of Guangxi Zhuang Autonomous Region</institution>, <institution>Guangxi Normal University</institution>, <addr-line>Guilin</addr-line>, <addr-line>Guangxi</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Xiangsihu College</institution>, <institution>Guangxi University for Nationalities</institution>, <addr-line>Nanning</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/462074/overview">Linwei Wang</ext-link>, Rochester Institute of Technology (RIT), United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1905844/overview">Shuqiang Wang</ext-link>, Chinese Academy of Sciences (CAS), China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2095429/overview">Raul Gonzalez-Gomez</ext-link>, Adolfo Ib&#xe1;&#xf1;ez University, Chile</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Qiang Fu, <email>qiangfu@gxnu.edu.cn</email>; Yuling Luo, <email>yuling0616@gxnu.edu.cn</email>; Xiwen Luo, <email>lxwrenai@163.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>12</day>
<month>03</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1515881</elocation-id>
<history>
<date date-type="received">
<day>08</day>
<month>11</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>14</day>
<month>02</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Liu, Wu, Fu, Luo, Luo, Qin, Huang and Chen.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Liu, Wu, Fu, Luo, Luo, Qin, Huang and Chen</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Multimodal diagnostic methods for Alzheimer&#x2019;s disease (AD) have demonstrated remarkable performance. However, the inclusion of electroencephalography (EEG) in such multimodal studies has been relatively limited. Moreover, most multimodal studies on AD use convolutional neural networks (CNNs) to extract features from different modalities and perform fusion classification. Regrettably, this approach often lacks collaboration and fails to effectively enhance the representation ability of features. To address this issue and explore the collaborative relationship among multimodal EEG, this paper proposes a multimodal AD diagnosis model based on resting-state EEG and structural magnetic resonance imaging (sMRI). Specifically, this work designs corresponding feature extraction models for EEG and sMRI modalities to enhance the capability of extracting modality-specific features. Additionally, a multimodal joint attention mechanism (MJA) is developed to address the issue of independent modalities. The MJA promotes cooperation and collaboration between the two modalities, thereby enhancing the representation ability of multimodal fusion. Furthermore, a random forest classifier is introduced to enhance the classification ability. The diagnostic accuracy of the proposed model can achieve 94.7%, marking a noteworthy accomplishment. This research stands as the inaugural exploration into the amalgamation of deep learning and EEG multimodality for AD diagnosis. Concurrently, this work strives to bolster the use of EEG in multimodal AD research, thereby positioning itself as a hopeful prospect for future advancements in AD diagnosis.</p>
</abstract>
<kwd-group>
<kwd>Alzheimer&#x2019;s disease</kwd>
<kwd>electroencephalography</kwd>
<kwd>magnetic resonance imaging</kwd>
<kwd>multimodal</kwd>
<kwd>joint attention mechanism</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Physiology and Medicine</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>AD is a neurodegenerative disease with a high incidence rate, currently affecting about 51.6 million people worldwide <xref ref-type="bibr" rid="B25">Schlachetzki et al. (2013)</xref>, which brings a heavy burden to society. According to reports, 6.7 million Americans aged 65 and older are currently living with Alzheimer&#x2019;s dementia. This number is likely to grow to 13.8 million by 2060. Meanwhile, the total cost of healthcare, long-term care, and hospice services for people with dementia aged 65 and over will reach an estimated <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>$</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>345 billion in 2023 <xref ref-type="bibr" rid="B24">Saykin et al. (2010)</xref>. So far, many markers of AD have been discovered. Many studies have focused on various aspects such as biomarker discovery, diagnosis methods, and therapeutic strategies. For example, studies with the <xref ref-type="bibr" rid="B40">Zong et al. (2024)</xref> proposed innovative approaches in analyzing neuroimaging data for AD diagnosis by integrating advanced image processing algorithms and machine learning techniques, aiming to improve the accuracy and efficiency of diagnosis. Another study with <xref ref-type="bibr" rid="B41">Zuo et al. (2024)</xref> focused on using multimodal data fusion methods to extract more comprehensive features from different sources of AD-related data for better understanding the disease progression. Moreover, the research with <xref ref-type="bibr" rid="B21">Pan et al. (2024)</xref> explored the potential of using specific neural network architectures to enhance the performance of AD diagnosis based on neuroimaging data. And the work with <xref ref-type="bibr" rid="B42">Zuo et al. (2023)</xref> investigated how to utilize time-series information in different modalities to capture the dynamic changes of AD, which is also quite inspiring for the field. The Alzheimer&#x2019;s Disease Neuroimaging Initiative (ADNI) has played a significant role in biomarker research, serving as a milestone in the field. Its primary objective is to the development of AD research by collecting various candidate biomarkers. ADNI combines magnetic resonance imaging (MRI) <xref ref-type="bibr" rid="B8">Elshafey et al. (2014)</xref> and positron emission tomography (PET) scans to study AD. It encompasses a vast amount of information related to the genetics, cerebrospinal fluid, and other biomarkers associated with AD <xref ref-type="bibr" rid="B1">a. Ill&#xe1;n et al. (2011)</xref>. But these modalities lack temporal resolution, and their analysis is only focused on traditional visual inspection. In recent years, some studies related to AD have been exploring the use of electroencephalography (EEG) to detect AD <xref ref-type="bibr" rid="B6">Darves-Bornoz et al. (2023)</xref>. At the same time, studies have also shown that EEG patterns are also one of the biomarkers of AD. In recent years, there has been a growing interest in AD diagnosis research using medical neuroimaging. Both machine learning (<xref ref-type="bibr" rid="B23">Peng et al., 2022</xref>; <xref ref-type="bibr" rid="B29">Uysal and Ozturk, 2020</xref>; <xref ref-type="bibr" rid="B11">Franciotti et al., 2023</xref>) and deep learning methods <xref ref-type="bibr" rid="B2">Alorf and Khan (2022)</xref>; <xref ref-type="bibr" rid="B17">Leela et al. (2023)</xref>; <xref ref-type="bibr" rid="B7">EL-Geneedy et al. (2022)</xref>; <xref ref-type="bibr" rid="B34">Yao et al. (2023a)</xref> have been widely explored in this field. However, medical neuroimaging lacks time resolution in the resting state, and it is difficult to form a continuous onset period time. To address this issue, researchers have turned to EEG as a potential marker for AD diagnosis, as EEG patterns provide temporal resolution. However, extracting meaningful representations from EEG patterns remains a significant challenge. Fortunately, deep learning models have also been applied to automatic feature extraction of EEG modalities <xref ref-type="bibr" rid="B3">Bi and Wang (2019)</xref>; <xref ref-type="bibr" rid="B33">Xia et al. (2023)</xref>, which can reduce the problems caused by the feature extraction process. The AD EEG data of the corresponding channel is selected, the corresponding channel data features are extracted and learned and finally classified by deep learning or machine learning classifier. Both medical neuroimaging and EEG data possess distinct modal characteristics. Medical neuroimaging captures changes in blood oxygen levels and changes in the hippocampus, while EEG provides high temporal resolution information. Integrating these two modes has been a challenging task for researchers. In recent years, there has been rapid development in multimodal AD diagnosis (<xref ref-type="bibr" rid="B30">Velazquez and Lee, 2022</xref>; <xref ref-type="bibr" rid="B9">Eslami et al., 2023</xref>; <xref ref-type="bibr" rid="B4">Chen et al., 2023</xref>; <xref ref-type="bibr" rid="B18">Leng et al., 2023</xref>), with most studies focusing on combining medical neuroimaging with clinical data. However, most multimodal models are trained independently for each modality, failing to capture the correlation and dependence between modalities. Only a few studies <xref ref-type="bibr" rid="B15">Jesus et al. (2021)</xref>; <xref ref-type="bibr" rid="B5">Colloby et al. (2016)</xref> have explored multimodal approaches using EEG data, employing machine learning methods for AD prediction. However, these studies rely on manual feature extraction, which is time-consuming, lacks interactivity, and is subject to subjective bias. Furthermore, identify and incorporate hidden features between the data. This study addresses the following issues. First, the absence of automatic feature extraction capabilities, compounded by the subjective nature of feature extraction, poses significant hurdles in identifying and extracting latent features from the data. Second, the current practice of training each modality model independently overlooks the interdependence and correlation between modalities. To overcome these obstacles, our work proposes a novel multimodal model that integrates sMRI and EEG patterns. Leveraging the unique characteristics of sMRI and EEG modalities, sMRI-based convolutional neural network (sCNN-sMRI) and EEG-based convolutional neural network (sCNN-EEG) are designed to extract features from respective modal data and promote the interdependence and correlation between different features. Furthermore, this work introduces an attention mechanism to bridge the semantic gap between models, thereby enhancing the learning capacity of the overall model. The main contributions of this paper are as follows. a. A multimodal AD diagnosis model based on the convolutional neural network (MCNNRF) is proposed. b. This work devises dedicated network architectures, namely sCNN-EEG and sCNN-sMRI, tailored for processing EEG and sMRI data, respectively. c. To handle the complexity of feature mapping and unveil latent features, stacked Random Forests (RF) is used for classification tasks d. A groundbreaking multimodal joint attention mechanism (MJA) is introduced to address the intricacies of feature extraction across different modalities. This mechanism fosters synergistic feature extraction while facilitating collaboration between modalities, thereby enhancing the model&#x2019;s ability to represent features effectively.</p>
<p>The rest of this work is organized as follows. The related work is discussed in Section II. The model used and constructed are provided in Section III. The model evaluation and experiments are presented in Section IV, and Section V concludes this paper.</p>
</sec>
<sec id="s2">
<title>2 Related Works</title>
<p>Currently, there are several studies focusing on AD diagnosis using unimodal data, primarily focusing on medical neuroimaging techniques. In <xref ref-type="bibr" rid="B19">Liu et al. (2023)</xref>, a diagnostic method using two - sample t - tests to detect AD is proposed. First, it uses two - sample t - tests to detect AD - related regions in MRI, then extracts the features of related regions through an unsupervised learning neural network, and finally classifies AD using a clustering algorithm. In <xref ref-type="bibr" rid="B20">Mehmood et al. (2021)</xref>, a layer - by - layer transfer learning model for AD diagnosis is developed.</p>
<p>However, the above-mentioned studies are all unimodal studies, lacking the interaction between modes and not considering the complementarity between multi-modalities. Multimodality has gained significant popularity in recent years, and a plethora of studies explore the potential of combining multiple modalities to enhance analysis and understanding. In <xref ref-type="bibr" rid="B5">Colloby et al. (2016)</xref>, multimodal EEG - MRI in the differential diagnosis of AD and dementia with Lewy bodies is proposed. The MRI index in this work is derived from the medial temporal lobe atrophy (MTA) score. Logistic regression analysis identified EEG predictors for AD and DLB. A joint EEG - MRI model is then generated to examine whether there is an improvement in classification compared to the individual patterns. In <xref ref-type="bibr" rid="B15">Jesus et al. (2021)</xref>, a multimodal prediction of Alzheimer&#x2019;s disease severity based on resting - state EEG and structural MRI is proposed. This work investigates the multimodal prediction of Mini - Mental State Examination (MMSE) scores using resting - state electroencephalography (EEG) and structural magnetic resonance imaging (MRI) scans. Evaluation is performed by three feature selection algorithms and four machine learning algorithms. Compared with <xref ref-type="bibr" rid="B5">Colloby et al. (2016)</xref>, this study is not only focused on the differential diagnosis between AD and other diseases but also aims to build a general multimodal diagnosis model for AD. In terms of methods, <xref ref-type="bibr" rid="B5">Colloby et al. (2016)</xref> relies on manually extracted MRI indicators and logistic regression analysis, while this study automatically extracts features from EEG and sMRI through deep learning, improving the accuracy and efficiency of diagnosis. Compared with <xref ref-type="bibr" rid="B15">Jesus et al. (2021)</xref>, this study innovatively proposes the Multimodal MJA, which effectively promotes the collaboration between different modalities. The MJA is more efficient in feature extraction and fusion, thus improving the diagnostic performance. In summary, previous AD diagnosis studies have achieved certain results in both unimodal and multimodal fields. However, most studies suffer from insufficient collaboration between modalities and less intelligent feature extraction methods. This study addresses these issues by designing dedicated feature - extraction models sCNN - EEG and sCNN - sMRI, combined with the innovative MJA, providing a more effective method for AD diagnosis.</p>
</sec>
<sec sec-type="methods" id="s3">
<title>3 Methods</title>
<p>In this section, the dataset is described in Section A. The feature selection method is provided in Section B. The sCNN-EEG model is proposed in Section C. The sCNN-sMRI model is proposed in Section D. The MJA module is described in Section E, and finally, the MCNNRF model is proposed in Section F.</p>
<sec id="s3-1">
<title>3.1 Dataset</title>
<p>The data set used in this work was provided by <xref ref-type="bibr" rid="B4">Chen et al. (2023)</xref>. The acquired data underwent text data processing using the Statistical Package for Social Sciences software (SPSS ver. 22.0, <ext-link ext-link-type="uri" xlink:href="http://www01.ibm.com/software/analytics/spss/products/statistics/">http://www01.ibm.com/software/analytics/spss/products/statistics/</ext-link>). During the data processing process, unknown and null values in MRI and EEG were estimated and filled using the weighted nearest neighbor algorithm <xref ref-type="bibr" rid="B28">Troyanskaya et al. (2001)</xref>. Subsequently, min-max normalization was applied to all data within the range [0,1].</p>
</sec>
<sec id="s3-2">
<title>3.2 Feature selection</title>
<p>Given that both modes contain hidden features in addition to observed features, utilizing too many features could lead to significant overfitting problems in the model. Therefore, this work uses two feature selection methods to address this concern. MRMR <xref ref-type="bibr" rid="B39">Zhao et al. (2019)</xref> feature selection is used for dimensionality reduction of the MRI and EEG datasets. By examining the score values of different subsets of the data set, the highest and most optimal feature set is selected. To streamline the process, a grading strategy of 10 is used for feature selection. The MRMR algorithm is run for each grading label, and N optimal features are chosen within the range of 10 500 through evaluating the score value. Following the selection of the N best features, a feature importance algorithm is employed to verify and further optimize these selected features.</p>
</sec>
<sec id="s3-3">
<title>3.3 Model for unimodal EEG data</title>
<p>In this work, the sCNN-EEG is designed to extract the important and hidden feature extraction from AD EEG data. In <xref ref-type="fig" rid="F1">Figure 1</xref>, the input undergoes convolution two kernels of size of 2, resulting in the generation of matrix <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is then processed through two different branches, where stacked convolutions with kernel sizes of 3 and 4 are applied, producing <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, respectively. Next, matrix multiplication is performed between <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, yielding a matrix graph <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> that contains both important and hidden features. A similar approach is used to combine <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, resulting in the formation of <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The matrix graph <inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can be calculated by <xref ref-type="disp-formula" rid="e1">Equation 1</xref>
<disp-formula id="e1">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Overall Structure Diagram of sCNN-EEG.</p>
</caption>
<graphic xlink:href="fphys-16-1515881-g001.tif"/>
</fig>
<p>The convolution kernels of the convolutional layer are all initialized with constants. A stride value of 1 is used to move the kernel window and perform the convolution across the entire input matrix. The sCNN-EEG maintains the size of the convolutional feature map, akin to the feature map of the previous convolution, and preserves the shape of the input data by setting the padding variables to be the same. The same padding variable can ensure that there will be no matrix problems during subsequent fusion. Since the second half of the pooling layer has the same structure, take the part of the model where <inline-formula id="inf13">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is an example. The feature map <inline-formula id="inf14">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is taken as two inputs, which are processed by the max pooling layer and the average pooling layer, respectively. After processing, two feature maps (<inline-formula id="inf15">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf16">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) are obtained. Both <inline-formula id="inf17">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf18">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> subjected to an identical convolution process, employing a kernel size of 2, resulting in the creation of two additional feature maps, <inline-formula id="inf19">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf20">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Then, a matrix multiplication operation is performed on <inline-formula id="inf21">
<mml:math id="m22">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf22">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to yield the composite feature map <inline-formula id="inf23">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. By using pooling operations such as max pooling and average pooling, important spatial information from the input feature maps is preserved while reducing dimensionality. These pooled feature maps, <inline-formula id="inf24">
<mml:math id="m25">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf25">
<mml:math id="m26">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, are then processed using convolution, which helps in extracting meaningful features. Finally, matrix multiplication is applied to combine these features, capturing the relationships between the different pooled representations and creating a composite feature map. This process is formally represented by <xref ref-type="disp-formula" rid="e2">Equation 2</xref>, which describes the mathematical operation involved in computing <inline-formula id="inf26">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.<disp-formula id="e2">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>V</mml:mi>
<mml:mi>G</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>M</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>X</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where AVG here is the average pooling layer, and MAX is the maximum pooling layer.</p>
<p>The two deep feature maps are matrix multiplied and concatenated with the original features to form the final feature map. The purpose of this is to ensure the integrity of feature information and to dig out deep features. Finally, the final feature maps are passed to the max pooling and connection layers. At the same time, the stacked feature maps are flattened using a flattening layer before the connection layer. This allows the feature maps to be transformed into a one-dimensional representation. The connection layer consists of 150, 100, and 50 units, which the flattened feature maps are connected to. Additionally, there is a hidden layer with a 50% dropout rate, which helps prevent overfitting by randomly dropping out half of the units during training. To ensure that the features are non-linear, the connection layer uses the hyperbolic tangent function as the activation function and initializes its weights with the Glorot normal initializer <xref ref-type="bibr" rid="B13">Glorot and Bengio (2010)</xref>. Therefore, the cost function can be obtained by<disp-formula id="e3">
<mml:math id="m29">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:mi>L</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mfenced open="(" close="">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:mspace width="1em"/>
<mml:mfenced open="" close=")">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mspace width="0.2em"/>
<mml:mi>log</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>&#x2202;</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where L is cost function is the combination of the binary cross-entropy and L2 regularization term. <inline-formula id="inf27">
<mml:math id="m30">
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a hyper-parameter which represents the regularization coefficient. <inline-formula id="inf28">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is true class label. <inline-formula id="inf29">
<mml:math id="m32">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is predicted class label. N is batch size. <inline-formula id="inf30">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the <inline-formula id="inf31">
<mml:math id="m34">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> weight parameter of the model. K is the number of weight matrices.</p>
</sec>
<sec id="s3-4">
<title>3.4 Model for unimodal sMRI data</title>
<p>In this work, the sCNN-sMRI is focused on important feature extraction and hidden feature extraction for sMRI. In <xref ref-type="fig" rid="F2">Figure 2</xref>, the sMRI data is taken as input, first passing through two identical convolution processes, using a convolution kernel of size 2. Then it goes through the feature extraction modules of two different convolution kernels. One of the paths consists of a set of convolution kernels 3 and 4. It aims to create two different feature maps (represented as <inline-formula id="inf32">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf33">
<mml:math id="m36">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> respectively). The other path consists of convolution kernels 4 and 2, which focuses on obtaining two different feature maps (represented as <inline-formula id="inf34">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf35">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> respectively). Multiple feature maps containing different feature information are created and fused through different feature extractors. Now all feature map features of the same feature extractor are fused together to form new feature maps <inline-formula id="inf36">
<mml:math id="m39">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf37">
<mml:math id="m40">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The <inline-formula id="inf38">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf39">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can be obtained by <xref ref-type="disp-formula" rid="e4">Equations 4</xref>, <xref ref-type="disp-formula" rid="e5">5</xref>
<disp-formula id="e4">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Verall Structure Diagram of sCNN-sMRI.</p>
</caption>
<graphic xlink:href="fphys-16-1515881-g002.tif"/>
</fig>
<p>In this model, multiple feature maps obtained from each branch are fused through matrix multiplication. Multiple feature maps <inline-formula id="inf40">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf41">
<mml:math id="m46">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are concatenated to obtain the feature map <inline-formula id="inf42">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. After cascading, <inline-formula id="inf43">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is calculated by the maximum pooling and the average pooling respectively, and obtains a multi-pooling feature matrix map. Multi-pooling features are convolved to obtain deep feature maps (each convolution kernel size of 2). The final feature map <inline-formula id="inf44">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is obtained by matrix multiplication of the maximum feature map and the average feature map. The <inline-formula id="inf45">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf46">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can be obtained by <xref ref-type="disp-formula" rid="e6">Equations 6</xref>, <xref ref-type="disp-formula" rid="e7">7</xref>
<disp-formula id="e6">
<mml:math id="m52">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa9;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="e7">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>V</mml:mi>
<mml:mi>G</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>M</mml:mi>
<mml:mi>A</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where <inline-formula id="inf47">
<mml:math id="m54">
<mml:mrow>
<mml:mtext>&#xa9;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> is the concatenation symbol, AVG is the average pooling layer, and MAX is the maximum pooling layer, W is the convolution kernel size. Finally, the <inline-formula id="inf48">
<mml:math id="m55">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is calculated by the max pooling and fully connection layers. After the maximum pooling layer, the structure of the connection layer is consistent with the connection of sCNN-EEG. Both consist of 200, 150, 50% and 50% dropout. The training cost function for this model is the same as <xref ref-type="disp-formula" rid="e3">Equation 3</xref>.</p>
</sec>
<sec id="s3-5">
<title>3.5 Multimodal joint attention mechanism</title>
<p>During the experiment, it is found that in the multi-branch feature extraction process of MCNNRF, the feature extraction process of the two modalities is independent of each other, and the lack of relevant cooperation may cause the extracted features to be independent of each other. At the same time, it may cause poor representation ability after multimodal fusion. To this end, this work proposes a fusion module of MJA, which is mainly used to explore the deep cooperation of two modalities to enhance the representation ability of extracted features. The structure of the MJA fusion module is shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. The design of the module is inspired by the spatial attention mechanism in <xref ref-type="bibr" rid="B12">Fu et al. (2019)</xref>. Specifically, the module takes the EEG data branching model (denoted as A) and the sMRI branching model (denoted as B) as input sources. To simplify the description, only the spatial attention unit of branch A is explained in detail in this paper. In the module, the structure of A and B is similar, and each branch consists of three convolutional layers and an S-shaped activation function; the convolutional layers are used to extract the features of each model, and the S-shaped activation function is used for nonlinear transformation. Given the input <inline-formula id="inf49">
<mml:math id="m56">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf50">
<mml:math id="m57">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf51">
<mml:math id="m58">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf52">
<mml:math id="m59">
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denote the number of channels, height, and width of the features, respectively, firstly, <inline-formula id="inf53">
<mml:math id="m60">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">query</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf54">
<mml:math id="m61">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">key</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf55">
<mml:math id="m62">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">value</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are generated by three <inline-formula id="inf56">
<mml:math id="m63">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> convolution operations, respectively, and the dimensionality of these outputs are all <inline-formula id="inf57">
<mml:math id="m64">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>/</mml:mo>
<mml:mn>8</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. In order to reduce the computational cost, the <inline-formula id="inf58">
<mml:math id="m65">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> convolution reduces the number of channels to 1/8 of the original, thus reducing the computational effort. Next, the attention score matrix is obtained by multiplying the transpose of <inline-formula id="inf59">
<mml:math id="m66">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">query</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with <inline-formula id="inf60">
<mml:math id="m67">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">key</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Then, the sigmoid activation function is used to generate the spatial attention graph <inline-formula id="inf61">
<mml:math id="m68">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, which reflects the spatial importance of the input features. Next, for branch B, the same operation is performed to obtain the corresponding spatial attention map <inline-formula id="inf62">
<mml:math id="m69">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, which is used to characterize the spatial feature importance of branch B. Unlike the use of softmax to generate <inline-formula id="inf63">
<mml:math id="m70">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in the original spatial attention mechanism, this paper employs an S-shaped activation function, which is designed to capture hidden features over a wider range and to be consistent with the activation function in multimodal models. In addition, in the traditional spatial attention mechanism, <inline-formula id="inf64">
<mml:math id="m71">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is only used to refine A. In this design, <inline-formula id="inf65">
<mml:math id="m72">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is not only used to refine B, but also achieves a deep fusion of the two modes by combining it with <inline-formula id="inf66">
<mml:math id="m73">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, which enables the two modes to work more closely together, thus improving the overall feature representation capability. Therefore, the MJA fusion module is able to better coordinate the feature learning of the two modalities by introducing a joint attention mechanism, which not only weights the features of the respective modalities at the spatial level, but also interactively fuses between the two modalities, thus improving the model&#x2019;s ability to comprehend and process multimodal data. This design allows the model to extract and fuse information more effectively, enhancing the robustness and accuracy of the final representation.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Chematic diagram of MJA module.</p>
</caption>
<graphic xlink:href="fphys-16-1515881-g003.tif"/>
</fig>
<p>Specifically, first perform three <inline-formula id="inf67">
<mml:math id="m74">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> convolutions to generate <inline-formula id="inf68">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">query</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf69">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">key</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf70">
<mml:math id="m77">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">value</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> respectively, and make their dimensions controlled at <inline-formula id="inf71">
<mml:math id="m78">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> Then <inline-formula id="inf72">
<mml:math id="m79">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">value</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf73">
<mml:math id="m80">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">value</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and the corresponding generated <inline-formula id="inf74">
<mml:math id="m81">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf75">
<mml:math id="m82">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are matrix multiplied to obtain two attention feature maps <inline-formula id="inf76">
<mml:math id="m83">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf77">
<mml:math id="m84">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The specific formulas of <inline-formula id="inf78">
<mml:math id="m85">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf79">
<mml:math id="m86">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can be calculated by <xref ref-type="disp-formula" rid="e8">Equations 8</xref>, <xref ref-type="disp-formula" rid="e9">9</xref>
<disp-formula id="e8">
<mml:math id="m87">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">value</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
<disp-formula id="e9">
<mml:math id="m88">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">value</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
</p>
<p>In the formula, <inline-formula id="inf80">
<mml:math id="m89">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>&#x3f5;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the stacked EEG feature guided by the sMRI feature, and <inline-formula id="inf81">
<mml:math id="m90">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>&#x3f5;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the stacked sMRI feature guided by the EEG feature. Finally, this work reshapes <inline-formula id="inf82">
<mml:math id="m91">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf83">
<mml:math id="m92">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> into <inline-formula id="inf84">
<mml:math id="m93">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and performs feature concatenation on <inline-formula id="inf85">
<mml:math id="m94">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with A, and <inline-formula id="inf86">
<mml:math id="m95">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with B to obtain the final stacked features. It can be calculated by <xref ref-type="disp-formula" rid="e10">Equations 10</xref>, <xref ref-type="disp-formula" rid="e11">11</xref>
<disp-formula id="e10">
<mml:math id="m96">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">refined</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mtext>&#xa9;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
<disp-formula id="e11">
<mml:math id="m97">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">refined</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>M</mml:mi>
<mml:mtext>&#xa9;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>where <inline-formula id="inf87">
<mml:math id="m98">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">refined</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>&#x3f5;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf88">
<mml:math id="m99">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">refined</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>&#x3f5;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the final EEG features and sMRI features, while N and M represent the initial input EEG features and sMRI features, respectively.</p>
</sec>
<sec id="s3-6">
<title>3.6 Multimodal model structure</title>
<p>The final framework of this work is a combination of sCNN-EEG, sCNN-sMRI models, and fused modality models. The sCNN-EEG and sCNN-sMRI models are responsible for extracting features from the corresponding modalities. The MAJ module is used to solve the interconnection and matching between multimodal features and to fuse multimodal features. MCNNRF is shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. It is divided into three phases. The first stage is to extract features from the corresponding modalities using single modality models (i.e. sCNN-EEG and sCNN-sMRI). The second phase aims to address the lack of interconnectivity and fusion in the multimodal information extraction process. The features extracted from the two single models are used as the input source of the MAJ module. The third stage is the stacked features formed after fusion and the stacked RF is used for classification.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Schematic diagram of MCNNRF module.</p>
</caption>
<graphic xlink:href="fphys-16-1515881-g004.tif"/>
</fig>
</sec>
</sec>
<sec id="s4">
<title>4 Experiments</title>
<p>In this section, the experimental environment and dataset are presented in Section A. Unimodal feature extraction model comparison is provided in Section B. Ablation work is provided in Section C. A Comparison between unimodal and multimodal model is provided in Section D. robustness analysis is provided in SectionE. Finally, Comparison with existing researches is provided in Section F.</p>
<sec id="s4-1">
<title>4.1 Experimental environment and dataset</title>
<p>This work is implemented by using TensorFlow library on an NVIDIA RTX A6000 GPU. The dataset used in this work is provided by the research of <xref ref-type="bibr" rid="B5">Colloby et al. (2016)</xref>. The dataset contains electroencephalogram (EEG) data from 99 Alzheimer&#x2019;s disease (AD) patients. However, due to the lack of data in 5 cases, magnetic resonance imaging (MRI) scan images are only available for 89 patients. Among the available cases, there are 45 females with an average age of 75.8<inline-formula id="inf89">
<mml:math id="m100">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>7.3 years. The EEG data of CNHCs (Healthy ControlsCognitively Normal) used is from a public static EEG dataset for epilepsy, and the MRI data is from the public ADNI dataset. Despite the data set being small, the model is relatively intricate, and this often culminates in overfitting of the model. Therefore, the experiments in this work use 10-fold cross-validation to deal with these problems. At the same time, the data set will be divided into 8:2 corresponding to the training set and the test set, where the training set is used for training the model, and the test set is used for testing and evaluating the model. The receiver operating characteristic curve (ROC) <xref ref-type="bibr" rid="B27">Tharwat (2018)</xref> is used as the main metric for hyperparameter tuning and finding the best model. This work also evaluated some secondary indicators such as sensitivity (Sn), specificity (Sp), accuracy (Acc), precision (Pre), and Matthew correlation coefficient (Mcc). These indicators can be calculated by <xref ref-type="disp-formula" rid="e12">Equations 12</xref>&#x2013;<xref ref-type="disp-formula" rid="e16">16</xref>
<disp-formula id="e12">
<mml:math id="m101">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>/</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
<disp-formula id="e13">
<mml:math id="m102">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>/</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
<disp-formula id="e14">
<mml:math id="m103">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>/</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>
<disp-formula id="e15">
<mml:math id="m104">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>c</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>/</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>
<disp-formula id="e16">
<mml:math id="m105">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#xd7;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#xd7;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#xd7;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>where <inline-formula id="inf90">
<mml:math id="m106">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is true positive, <inline-formula id="inf91">
<mml:math id="m107">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is true negative, <inline-formula id="inf92">
<mml:math id="m108">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is false positive and <inline-formula id="inf93">
<mml:math id="m109">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents false negative values. They are calculated from the confusion matrix of the predicted results.</p>
</sec>
<sec id="s4-2">
<title>4.2 Unimodal feature extraction model comparison</title>
<p>This work focuses on the feature extraction of EEG and sMRI datausing sCNN-EEG and sCNN-sMRI models, respectively. Prior to determining the sCNN-EEG and sCNN-sMRI models, this work designed some feature extraction model strategies for two modalities, named CNN-EEG and CNN-sMRI respectively. Compared with sCNN-EEG and sCNN-sMRI, CNN-EEG and CNN-sMRI only lacks different multiple pooling layer modules. In this section, the performance of different networks (CNN-EEG, sCNN-EEG, CNN-sMRI, and sCNN-sMRI) is compared. The Receiver Operating Characteristic (ROC) is shown in <xref ref-type="fig" rid="F5">Figure 5</xref>, and the Area Under the Curve (AUC) is calculated. The AUC of sCNN-sMRI is 0.33% higher than that of CNN-sMRI, while the AUC of sCNN-EEG is 2.63% higher than that of CNN-EEG. Other performance indicators are shown in <xref ref-type="table" rid="T1">Table 1</xref>. Compared with CNN-sMRI, the precision sCNN-sMRI improves to 75.97%, an Mcc improves to 51.35%, and a relatively flat sensitivity. Compared with CNN-EEG, sCNN-EEG has a relatively flat accuracy and a sensitivity improvement of 18.76%. Compared with CNN-EEG/CNN-sMRI, sCNN-EEG, and sCNN-sMRI not only have more multi-branch different convolutional kernels for feature extraction but also strengthen the learning of weak features and use multi-pooling modules to extract deep-level features. At the same time, using stacked connections in the connection allows the extracted features to perform stacked features, which can better integrate hidden features into it.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>OC of sCNN-EEG, sCNN-sMRI, CNN-EEG and CNN-sMRI.</p>
</caption>
<graphic xlink:href="fphys-16-1515881-g005.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>sCNN-EEG, sCNN-sMRI, various performance indicators of CNN-EEG and CNN-sMRI.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Model</th>
<th align="center">Accuracy (%)</th>
<th align="center">Precision (%)</th>
<th align="center">Sensitivity (%)</th>
<th align="center">Mcc (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">CNN-sMRI</td>
<td align="center">86.51</td>
<td align="center">71.35</td>
<td align="center">34.07</td>
<td align="center">44.72</td>
</tr>
<tr>
<td align="left">sCNN-sMRI</td>
<td align="center">86.84</td>
<td align="center">75.97</td>
<td align="center">36.79</td>
<td align="center">51.35</td>
</tr>
<tr>
<td align="left">CNN-EEG</td>
<td align="center">75.80</td>
<td align="center">61.57</td>
<td align="center">22.91</td>
<td align="center">29.37</td>
</tr>
<tr>
<td align="left">sCNN-EEG</td>
<td align="center">78.43</td>
<td align="center">61.04</td>
<td align="center">41.67</td>
<td align="center">29.57</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-3">
<title>4.3 Ablation study</title>
<p>After analyzing the various performance indicators of the centralized single-mode feature extractor, a model for the feature extractor is selected. The feature extractor is used to extract multimodal features, which are then stacked together. The multimodal features are fused and classified using different strategies. Initially, a simple concatenation matrix method is used to fuse the multimodal features, and the performance of the model is evaluated. However, it is observed that simple splicing methods does not directly improve and enhance the performance and classification ability of multimodal models. For this purpose, this work has designed several strategies for multimodal fusion. For example, using a bimodal attention mechanism for fusion. The multimodal model of EEG and sMRI can solve the correlation and cooperation between the modalities and can also classify AD well. <xref ref-type="table" rid="T2">Table 2</xref> shows the performance indicators of each strategy, and their ROC curves are shown in <xref ref-type="fig" rid="F6">Figure 6</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Performance indicators of multimodal models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">Accuracy (%)</th>
<th align="center">Precision (%)</th>
<th align="center">Sensitivity (%)</th>
<th align="center">Mcc (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">MCNNBA</td>
<td align="center">83.73</td>
<td align="center">73.68</td>
<td align="center">41.67</td>
<td align="center">40.14</td>
</tr>
<tr>
<td align="center">MCNNcRF</td>
<td align="center">62.93</td>
<td align="center">51.20</td>
<td align="center">51.09</td>
<td align="center">37.64</td>
</tr>
<tr>
<td align="center">MCNNBARF</td>
<td align="center">84.43</td>
<td align="center">81.88</td>
<td align="center">72.36</td>
<td align="center">43.10</td>
</tr>
<tr>
<td align="center">MCNNRF</td>
<td align="center">94.75</td>
<td align="center">85.12</td>
<td align="center">80.88</td>
<td align="center">75.34</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Multimodal strategy ROC</p>
</caption>
<graphic xlink:href="fphys-16-1515881-g006.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F6">Figure 6</xref> shows the performance curve of different multimodality models. Even with RF as the classifier, MCNNcRF accuracy is only 62.93%. MCNNBA uses dual attention fusion for multiple modalities, with an accuracy of 83.73%. Compared with MCNNcRF, the accuracy of MCNNBA is much higher than that of MCNNcRF. The main reason is that MCNNBA&#x2019;s dual attention fusion module is more focused on the connections between multimodals. When MCNN uses Bi-Attention and adds RF for classification, the accuracy rate is 84.43%, because RF enhances its ability to classify stacked features. Finally, when MCNNBA adds RF, the accuracy rate reaches 94.75%. Compared with the previous strategies, this model uses the fusion module to cooperate and deeply fuse the features after extracting the two modal features. <xref ref-type="table" rid="T2">Table 2</xref> shows the MCNNRF strategy outperformed all other strategy models in this experiment, demonstrating superior performance in terms of accuracy, precision, sensitivity and Mcc values. When compared with MCNNcRF, the accuracy of MCNNRF is elevated by 31.82%, precision is amplified by 11.44%, and sensitivity is increased by 39.21%. As shown in <xref ref-type="table" rid="T2">Table 2</xref> and <xref ref-type="fig" rid="F6">Figure 6</xref>, among different multimodal fusion strategies, MCNNRF with the MJA module shows the best performance in accuracy, precision, sensitivity and Mcc values. Although both MCNNBA and MCNNBARF use bimodal attention mechanisms for fusion, MCNNBARF outperforms MCNNBA in all performance metrics due to its utilization of RF to enhance classification capabilities. Nevertheless, MCNNBARF still falls short of MCNNRF&#x2019;s performance, as the bimodal attention mechanism only enhances feature extraction capabilities without exploring and amplifying the intercommunication and complementarity between modalities.</p>
</sec>
<sec id="s4-4">
<title>4.4 Comparison between unimodal and multimodal model</title>
<p>EEG modalities and sMRI modalities are fed into the model by the multimodal model as output sources. Compared to the unimodal model, the multimodal model diversifies the input. At the same time, the multimodal model fuses the characteristic enhancement features between the different modalities, and obtains a higher accuracy. Their various performance indicators are shown in <xref ref-type="table" rid="T3">Table 3</xref>. It is obvious that the multimodal model is the optimal model, and the accuracy is increased by 8.91% and 16.32% compared with the sCNN-sMRI and the sCNN-EEGl, respectively. At the same time, other parameters have been greatly improved. Overall in this experiment, MCNNRF is far superior to the unimodal model.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Performance indicators of multimodal models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">Accuracy (%)</th>
<th align="center">Precision (%)</th>
<th align="center">Sensitivity (%)</th>
<th align="center">Mcc (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">sCNN-sMRI</td>
<td align="center">86.84</td>
<td align="center">75.97</td>
<td align="center">36.79</td>
<td align="center">51.35</td>
</tr>
<tr>
<td align="center">sCNN-EEG</td>
<td align="center">78.43</td>
<td align="center">61.04</td>
<td align="center">41.67</td>
<td align="center">29.57</td>
</tr>
<tr>
<td align="center">MCNNRF</td>
<td align="center">94.75</td>
<td align="center">85.12</td>
<td align="center">80.88</td>
<td align="center">75.34</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-5">
<title>4.5 Robustness analysis</title>
<p>To analyze the robustness of the proposed models, 10 independent experiments are performed on each model. The experimental results, measured in terms of accuracy, are presented in <xref ref-type="fig" rid="F7">Figure 7</xref> The MCNNRF model exhibits obvious advantages, outperforming both unimodal models (sCNN-EEG and sCNN-sMRI) consistently. Even the worst-performing MCNN model surpasses the best-performing sCNN-EEG and sCNN-sMRI models by 7.07% and 2.02%, respectively. Additionally, the curves of all three models demonstrate stability and consistency throughout the experiments.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Performance indicators of sCNN-sMRI, sCNN-EEG, and MCNNRF.</p>
</caption>
<graphic xlink:href="fphys-16-1515881-g007.tif"/>
</fig>
</sec>
<sec id="s4-6">
<title>4.6 Comparison with existing researches</title>
<sec id="s4-6-1">
<title>4.6.1 Compared to unimodal</title>
<p>In this section, the performance of MCNNRF is compared with advanced AD unimodal diagnostic models. The comparison results are shown in <xref ref-type="table" rid="T4">Table 4</xref>. The accuracy performance comparison of each model shows that the MCNNRF is the best performing architecture with the highest accuracy. However, when considering the single EEG mode, the accuracy does not show significant differences. Although MCNNRF achieves approximately 1.75% higher accuracy than the other models, there is still a gap compared to DPCNN in terms of accuracy. The reason for this is that their dataset is relatively small and they chose to build their model using DPCNN, which is more suitable for one-dimensional data. The purpose is to increase the convolutional kernel to enhance the learning ability of one-dimensional data and prevent overfitting and gradient explosion issues. LMCN achieves an accuracy of 98%, reaching high levels of precision and sensitivity. The MCNNRF model achieves an accuracy of only 94.75%. This discrepancy primarily stems from LMCN&#x2019;s application of bidirectional long short-term memory networks to analyze time series predicated on EEG characteristics. Concurrently, LMCN leverages CNN to probe into the relationship between different channels and brain signals. The fusion of these techniques fully harnesses the characteristics of EEG, leading to high accuracy. Adazd-Net achieves an accuracy of 98.51%, a precision of 97.29%, and a sensitivity of 1. This remarkable performance is attributed to Adazd-Net using an interpretable boosting machine as a predictor and employing a designed adaptive and flexible Analytic Dyadic Zernike (ADZ) wavelet transformation for processing EEG data. The adaptive and flexible ADZ wavelet transformation automatically adapts to EEG variations and identifies the most discriminative channels. Compared to LMCN and Adazd-Net, this work significantly differs in EEG data processing. They focus more on the impact of the relationship between channels and EEG on AD, while this work emphasizes the relationships between multiple modalities and does not delve into a detailed analysis of EEG channels. The accuracy of MCNNRF is 6.05% and 4.35% higher than Fuzzy-VGG and VGG16 respectively. However, when compared to Fuzzy-VGG, the accuracy and sensitivity of this work are still lag slightly behind. The reason is that Fuzzy-VGG uses fuzzy C-means to modify image pixels for MRI to achieve the effect of implicitly marking the lesion area. At the same time, the Fuzzy-VGG adopts stacked small kernel convolution, which can obtain more useful information in complex images in a given area. Although the method of Fuzzy-VGG achieves a better result, it cannot directly detect the given area and reduce the interference of useless information. MCNNRF is 3.35% more accurate compared to VGG16. VGG16 uses a large data set. However, the data set of MCNNRF is small, and the feature extraction ability of the model has not been enhanced, so there are not enough features for learning classification. To compared with MRN, MCNNRF is 2.89% less accurate and 2.45% less sensitive. The reason for this significant gap is that MRN uses multi relational inference networks to learn MRI through spatial information correlation and topology. Therefore, MRN is possible to obtain multiple types of inter-regional relationships. MCNNRF extracts deep features from MRI data. The accuracy of IDA-Net is 2.05% lower than that of MCNNRF, but the sensitivity is 11.02% higher. IDA-Net uses the Transformer structure to classify AD, and the dataset used in this method is relatively large.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Performance indicators of multimodal models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">Modal type</th>
<th align="center">Accuracy (%)</th>
<th align="center">Precision (%)</th>
<th align="center">Sensitivity (%)</th>
<th align="center">Mcc (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">DPCNN <xref ref-type="bibr" rid="B10">Fouad and Labib (2023)</xref>
</td>
<td align="center">EEG</td>
<td align="center">93.0</td>
<td align="center">95.8</td>
<td align="center">-</td>
<td align="center">-</td>
</tr>
<tr>
<td align="center">LMCN <xref ref-type="bibr" rid="B14">Imani (2023)</xref>
</td>
<td align="center">EEG</td>
<td align="center">98</td>
<td align="center">1.00</td>
<td align="center">97</td>
<td align="center">-</td>
</tr>
<tr>
<td align="center">Adazd-Net <xref ref-type="bibr" rid="B16">Khare and Acharya (2023)</xref>
</td>
<td align="center">EEG</td>
<td align="center">98.51</td>
<td align="center">97.29</td>
<td align="center">100</td>
<td align="center">-</td>
</tr>
<tr>
<td align="center">Fuzzy-VGG <xref ref-type="bibr" rid="B35">Yao et al. (2023b)</xref>
</td>
<td align="center">MRI</td>
<td align="center">88.7</td>
<td align="center">92.9</td>
<td align="center">91.7</td>
<td align="center">72.5</td>
</tr>
<tr>
<td align="center">VGG16 <xref ref-type="bibr" rid="B26">Sharma et al. (2022)</xref>
</td>
<td align="center">MRI</td>
<td align="center">90.4</td>
<td align="center">90.5</td>
<td align="center">-</td>
<td align="center">-</td>
</tr>
<tr>
<td align="center">MRN <xref ref-type="bibr" rid="B37">Zhang et al. (2023b)</xref>
</td>
<td align="center">MRI</td>
<td align="center">97.64</td>
<td align="center">-</td>
<td align="center">83.33</td>
<td align="center">-</td>
</tr>
<tr>
<td align="center">IDA-Net <xref ref-type="bibr" rid="B38">Zhao et al. (2023)</xref>
</td>
<td align="center">MRI</td>
<td align="center">92.7</td>
<td align="center">-</td>
<td align="center">91.9</td>
<td align="center">-</td>
</tr>
<tr>
<td align="center">MCNNRF</td>
<td align="center">EEG &#x2b; sMRI</td>
<td align="center">94.75</td>
<td align="center">85.12</td>
<td align="center">80.88</td>
<td align="center">75.34</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s4-7">
<title>4.7 Compared to multimodal</title>
<p>This work is the first work to explore the application of deep learning in combination with EEG multimodality for AD diagnosis. Therefore, this work compares with most advanced multimodal methods. The performance comparison between the different models is shown in <xref ref-type="table" rid="T5">Table 5</xref>. Compared to CNN &#x2b; ANN model, MCNNRF shows relatively lower accuracy and sensitivity. The reason is that CNN &#x2b; ANN has conducted deep mining of clinical and biological information. Firstly, CNN is used to extract features from images, and then a fusion module is designed using ANN to fuse and classify features. MCNNRF model lacks the supplementation of auxiliary information like clinical data and does not utilize feature transformation techniques to reduce feature dimensionality differences. Compared to HMGD, MCNNRF shows relatively lower accuracy and sensitivity. The specific reason is that HMGD employs graph diffusion methods to enhance the representation capability of multimodal data, thereby strengthening the measurement of multimodal similarity. However, MCNNRF is more focused on cross-modal collaboration and correlation. In the comparison on Accuracy, the performance of MCNNRF is on par with OLFG. The difference between MCNNRF and OLFG lies in one utilizing a multimodal combination of EEG and MRI, while the other employs MRI and PET. OLFG focuses more on the variations of various information in brain images. MCNNRF considers changes in brain image information, while also focusing on information differences that occur over time. In summary, this work explores the application of multimodal EEG in AD. Compared to MCNNRF, the accuracy of 3D-CNN-BRNN increases 1.25% and the sensitivity value increases 11.12%. The main reason is that the 3D-CNN-BRNN dataset owns a clear time series, with mobile MRI data spanning 6 months. At the same time, bidirectional recurrent neural networks are used to recognize the time series. 3DCNN is used to extract MRI features, and then AD is classified by auxiliary information. However, MCNNRF has the different time span as this method for controlling datasets, and there is also no corresponding time series for recognition. Compared to MCNNRF, MCAD has a 0.68% lower accuracy, and MCAD uses MRI and PET as well as some auxiliary modal information. MCAD uses a cross attention mechanism to fuse modalities, while MCNNRF performs deep feature mining on modalities and finally performs fusion.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Comparison between MCNNRF and multimodal models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">Modal type</th>
<th align="center">Accuracy (%)</th>
<th align="center">Precision (%)</th>
<th align="center">Sensitivity (%)</th>
<th align="center">Mcc (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">CNN &#x2b; ANN <xref ref-type="bibr" rid="B31">Wang et al. (2023a)</xref>
</td>
<td align="center">MRI &#x2b; profile ect</td>
<td align="center">96.2</td>
<td align="center">97.4</td>
<td align="center">-</td>
<td align="left"/>
</tr>
<tr>
<td align="center">HMGD <xref ref-type="bibr" rid="B32">Wang et al. (2023b)</xref>
</td>
<td align="center">PET &#x2b; gene</td>
<td align="center">96.4</td>
<td align="center">97.8</td>
<td align="center">-</td>
<td align="left"/>
</tr>
<tr>
<td align="center">OLSL <xref ref-type="bibr" rid="B4">Chen et al. (2023)</xref>
</td>
<td align="center">MRI &#x2b; PET</td>
<td align="center">94.7</td>
<td align="center">89.0</td>
<td align="center">-</td>
<td align="left"/>
</tr>
<tr>
<td align="center">3D-CNN-BRNN <xref ref-type="bibr" rid="B22">Pang et al. (2021)</xref>
</td>
<td align="center">MRI &#x2b; dc &#x2b; cs</td>
<td align="center">96.00</td>
<td align="center">92.00</td>
<td align="center">-</td>
<td align="left"/>
</tr>
<tr>
<td align="center">MCAD <xref ref-type="bibr" rid="B36">Zhang et al. (2023a)</xref>
</td>
<td align="center">sMRI &#x2b; PET &#x2b; CSF</td>
<td align="center">94.07</td>
<td align="center">-</td>
<td align="center">-</td>
<td align="left"/>
</tr>
<tr>
<td align="center">MCNNRF</td>
<td align="center">EEG &#x2b; sMRI</td>
<td align="center">94.75</td>
<td align="center">85.12</td>
<td align="center">80.88</td>
<td align="center">75.34</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="s5">
<title>5 Discussion</title>
<p>The high accuracy of the MCNNRF model can be attributed to the effective cooperation between sCNN - EEG and sCNN - sMRI in feature extraction. The MJA module plays a crucial role in enhancing the correlation between modalities, enabling the model to capture more comprehensive information related to AD. For example, the EEG data provides high - temporal - resolution information, while the sMRI data reflects the structural changes of the brain. The MJA module effectively combines these two types of information, leading to improved diagnostic performance. However, the MCNNRF model also has some limitations. The relatively small dataset used in this study may limit the generalization ability of the model. Additionally, the model only considers EEG and sMRI data, ignoring other potentially important information such as patient history and genetic factors. Future research could focus on expanding the dataset and incorporating more modalities to improve the model&#x2019;s performance. Previous studies mostly used single - modality data or independent training of multimodal models, lacking the exploration of the correlation between modalities. In contrast, our MCNNRF model uses the MJA module to promote the collaboration between EEG and sMRI modalities. Compared with the study in <xref ref-type="bibr" rid="B5">Colloby et al. (2016)</xref> that uses manual feature extraction, our model automatically extracts features through deep learning, reducing subjective bias. And compared with <xref ref-type="bibr" rid="B15">Jesus et al. (2021)</xref>, our model shows better performance in multimodal fusion and classification. This study is the first to explore the combination of deep learning and EEG multimodality for AD diagnosis. The proposed MCNNRF model provides a new approach for AD diagnosis, which has potential application value in clinical practice. The model&#x2019;s high - performance multimodal fusion and classification ability can help doctors make more accurate AD diagnoses, contributing to the early detection and treatment of AD.</p>
</sec>
<sec sec-type="conclusion" id="s6">
<title>6 Conclusion</title>
<p>In conclusion, this work presents a multimodal AD diagnostic model integrating EEG and sMRI data. It designs sCNN-EEG and sCNN-sMRI for feature extraction, and the classification performance is improved by incorporating RF into the classifier. Comparative experimental results also demonstrate that the proposed diagnostic model is competitive with the state-of-the-art methods for multimodality-based AD diagnosis. Simultaneously, this work pioneers the exploration of deep learning amalgamated with EEG multimodality in the realm of AD diagnosis. It holds promising potential to serve as a viable option for Alzheimer&#x2019;s Disease diagnosis in the forthcoming future. The results show that MCNNRF achieves state-of-the-art overall performance compared to existing multimodal AD diagnostic models. Furthermore, the results of the ablation experiment demonstrate the effectiveness of the MJA block and deep introduction of RF. It is important to acknowledge that this work has two limitations. On the one hand, MCNNRF only takes EEG and sMRI as input, while ignoring the patterns of patient history. On the other hand, MCNNRF can only handle complete multimodal data and is not suitable for the absence of a certain modality. Therefore, future work will focus on introducing patient history into the proposed framework and adjusting the model structure to handle missing patterns.</p>
</sec>
<sec id="s7">
<title>7 Futuer Works</title>
<p>The multimodal joint attention mechanism (MJA fusion module) proposed in this study provides an effective framework for combining electroencephalogram (EEG) and structural magnetic resonance imaging (sMRI) data with significant improvements in feature extraction and fusion. Future studies can further explore more complex multimodal data fusion strategies, such as the introduction of functional magnetic resonance imaging (fMRI) and near-infrared spectroscopy (NIRS), and the application of deep learning techniques, such as self-attention mechanisms and graph neural networks, to improve the expressiveness and robustness of multimodal fusion and enhance the accuracy of clinical diagnosis. For Alzheimer&#x2019;s disease (AD), the MJA module can be extended to be applied to early diagnosis and prediction of AD, combining EEG and sMRI data to more comprehensively assess EEG activity and structural changes, constructing a multimodal early diagnostic system, and realizing dynamic tracking of AD patients and evaluation of treatment effects. The potential of EEG as a biomarker for AD should be further explored to provide data support for personalized prediction models. In addition, future research should also focus on the personalization of the model, customizing the fusion model based on the patient&#x2019;s age, gender, and genetic background, as well as improving the interpretability and clinical applicability of the model to achieve real-time, automated AD detection and integration with healthcare information systems to provide adjunctive diagnostic support.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s8">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec sec-type="author-contributions" id="s9">
<title>Author contributions</title>
<p>JL: Conceptualization, Funding acquisition, Resources, Supervision, Methodology, Writing&#x2013;review and editing. SW: Investigation, Methodology, Writing&#x2013;original draft, Formal Analysis, Validation, Visualization. QF: Formal Analysis, Funding acquisition, Visualization, Writing&#x2013;review and editing. XL: Data curation, Writing&#x2013;review and editing. YL: Conceptualization, Resources, Software, Validation, Writing&#x2013;review and editing. SQ: Visualization, Writing&#x2013;review and editing, Conceptualization, Formal Analysis. YH: Visualization, Writing&#x2013;review and editing. ZC: Software, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s10">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This research was supported by the National Natural Science Foundation of China under Grant 62462009, Guangxi Natural Science Foundation under Grants 2022GXNSFAA035632, 2022GXNSFFA035028 and 2025GXNSFBA069292, the Guangxi Science and Technology Projects under Grant GuiKeAD24010047, the Basic Ability Enhancement Program for Young and Middle-aged Teachers of Guangxi under Grant 2024KY0074, and a grant (No. BCIC-24-Z7) from Guangxi Key Laboratory of Brain-inspired Computing and Intelligent Chips.</p>
</sec>
<sec sec-type="COI-statement" id="s11">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s12">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s13">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alorf</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>M. U. G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Multi-label classification of alzheimer&#x2019;s disease stages from resting-state fmri-based correlation connectivity data and deep learning</article-title>. <source>Comput. Biol. Med.</source> <volume>151</volume>, <fpage>106240</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.106240</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bi</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Early alzheimer&#x2019;s disease diagnosis based on eeg spectral images using deep learning</article-title>. <source>Neural Netw.</source> <volume>114</volume>, <fpage>119</fpage>&#x2013;<lpage>135</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2019.02.005</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Orthogonal latent space learning with feature weighting and graph learning for multimodal alzheimer&#x2019;s disease diagnosis</article-title>. <source>Med. Image Anal.</source> <volume>84</volume>, <fpage>102698</fpage>. <pub-id pub-id-type="doi">10.1016/j.media.2022.102698</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Colloby</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Cromarty</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Peraza</surname>
<given-names>L. R.</given-names>
</name>
<name>
<surname>Johnsen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>J&#xf3;hannesson</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bonanni</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Multimodal eeg-mri in the differential diagnosis of alzheimer&#x2019;s disease and dementia with lewy bodies</article-title>. <source>J. Psychiatric Res.</source> <volume>78</volume>, <fpage>48</fpage>&#x2013;<lpage>55</lpage>. <pub-id pub-id-type="doi">10.1016/j.jpsychires.2016.03.010</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Darves-Bornoz</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Barbeau</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Calvel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Denuelle</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Guines</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Celebrini</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>N&#xb0;329 &#x2013; going beyond the empirical and historical parameters of intracranial electrical brain stimulations to improve the yield of stereo-electroencephalography</article-title>. <source>Clin. Neurophysiol.</source> <volume>150</volume>, <fpage>e174</fpage>&#x2013;<lpage>e175</lpage>. <pub-id pub-id-type="doi">10.1016/j.clinph.2023.03.304</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>El-Geneedy</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Moustafa</surname>
<given-names>H. E. D.</given-names>
</name>
<name>
<surname>Khalifa</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Khater</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>AbdElhalim</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>An mri-based deep learning approach for accurate detection of alzheimer&#x2019;s disease</article-title>. <source>Alexandria Eng. J.</source>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1016/j.aej.2022.07.062</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elshafey</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hassanien</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Khalil</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Allah</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Saad</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Baghdadi</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Hippocampus, caudate nucleus and entorhinal cortex volumetric mri measurements in discrimination between alzheimer&#x2019;s disease, mild cognitive impairment, and normal aging</article-title>. <source>Egypt. J. Radiology Nucl. Med.</source> <volume>45</volume>, <fpage>511</fpage>&#x2013;<lpage>518</lpage>. <pub-id pub-id-type="doi">10.1016/j.ejrnm.2013.12.011</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eslami</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tabarestani</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Adjouadi</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A unique color-coded visualization system with multimodal information fusion and deep learning in a longitudinal study of alzheimer&#x2019;s disease</article-title>. <source>Artif. Intell. Med.</source> <volume>140</volume>, <fpage>102543</fpage>. <pub-id pub-id-type="doi">10.1016/j.artmed.2023.102543</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fouad</surname>
<given-names>I. A.</given-names>
</name>
<name>
<surname>Labib</surname>
<given-names>F. E.-Z. M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Identification of alzheimer&#x2019;s disease from central lobe eeg signals utilizing machine learning and residual neural network</article-title>. <source>Biomed. Signal Process. Control</source> <volume>86</volume>, <fpage>105266</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2023.105266</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Franciotti</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Nardini</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Russo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Onofrj</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sensi</surname>
<given-names>S. L.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Comparison of machine learning-based approaches to predict the conversion to alzheimer&#x2019;s disease from mild cognitive impairment</article-title>. <source>Neuroscience</source> <volume>514</volume>, <fpage>143</fpage>&#x2013;<lpage>152</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroscience.2023.01.029</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Fu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <source>Dual attention network for scene segmentation</source>, <fpage>3146</fpage>&#x2013;<lpage>3154</lpage>.</citation>
</ref>
<ref id="B13">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Glorot</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>Understanding the difficulty of training deep feedforward neural networks</article-title>,&#x201d; in <source>Proceedings of the thirteenth international conference on artificial intelligence and statistics</source>, <fpage>249</fpage>&#x2013;<lpage>256</lpage>.</citation>
</ref>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ill&#xe1;n</surname>
<given-names>I. A.</given-names>
</name>
<name>
<surname>G&#xf3;rriz</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Ram&#xed;rez</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Salas-Gonzalez</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>L&#xf3;pez</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Segovia</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group>(<year>2011</year>). <article-title>18f-fdg pet imaging analysis for computer aided alzheimer&#x2019;s diagnosis</article-title>. <source>Inf. Sci.</source> <volume>181</volume>, <fpage>903</fpage>&#x2013;<lpage>916</lpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2010.10.027</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Imani</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Alzheimer&#x2019;s diseases diagnosis using fusion of high informative bilstm and cnn features of eeg signal</article-title>. <source>Biomed. Signal Process. Control</source> <volume>86</volume>, <fpage>105298</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2023.105298</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jesus</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Cassani</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>McGeown</surname>
<given-names>W. J.</given-names>
</name>
<name>
<surname>Cecchi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fadem</surname>
<given-names>K. C.</given-names>
</name>
<name>
<surname>Falk</surname>
<given-names>T. H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Multimodal prediction of alzheimer&#x2019;s disease severity level based on resting-state eeg and structural mri</article-title>. <source>Front. Hum. Neurosci.</source> <volume>15</volume>, <fpage>700627</fpage>. <pub-id pub-id-type="doi">10.3389/fnhum.2021.700627</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khare</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Acharya</surname>
<given-names>U. R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Adazd-net: automated adaptive and explainable alzheimer&#x2019;s disease detection system using eeg signals</article-title>. <source>Knowledge-Based Syst.</source> <volume>278</volume>, <fpage>110858</fpage>. <pub-id pub-id-type="doi">10.1016/j.knosys.2023.110858</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Leela</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Helenprabha</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Sharmila</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Prediction and classification of alzheimer disease categories using integrated deep transfer learning approach</article-title>. <source>Meas. Sensors</source> <volume>27</volume>, <fpage>100749</fpage>. <pub-id pub-id-type="doi">10.1016/j.measen.2023.100749</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Leng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Multimodal cross enhanced fusion network for diagnosis of alzheimer&#x2019;s disease and subjective memory complaints</article-title>. <source>Comput. Biol. Med.</source> <volume>157</volume>, <fpage>106788</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.106788</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Mazumdar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bath</surname>
<given-names>P. A.</given-names>
</name>
</person-group>
<collab>Alzheimer&#x27;s Disease Neuroimaging Initiative</collab> (<year>2023</year>). <article-title>An unsupervised learning approach to diagnosing alzheimer&#x2019;s disease using brain magnetic resonance imaging scans</article-title>. <source>Int. J. Med. Inf.</source> <volume>173</volume>, <fpage>105027</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2023.105027</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mehmood</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>yang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>feng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ahmad</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>khan</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>A transfer learning approach for early diagnosis of alzheimer&#x2019;s disease on mri images</article-title>. <source>Neuroscience</source> <volume>460</volume>, <fpage>43</fpage>&#x2013;<lpage>52</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroscience.2021.01.002</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zuo</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C. P.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Decgan: decoupling generative adversarial network for detecting abnormal neural circuits in alzheimer&#x2019;s disease</article-title>. <source>IEEE Trans. Artif. Intell.</source> <volume>5</volume>, <fpage>5050</fpage>&#x2013;<lpage>5063</lpage>. <pub-id pub-id-type="doi">10.1109/TAI.2024.3416420</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>A multi-modal data platform for diagnosis and prediction of alzheimer&#x2019;s disease using machine learning methods</article-title>. <source>Mob. Netw. Appl.</source> <volume>26</volume>, <fpage>2341</fpage>&#x2013;<lpage>2352</lpage>. <pub-id pub-id-type="doi">10.1007/s11036-021-01834-1</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>18f-fdg-pet radiomics based on white matter predicts the progression of mild cognitive impairment to alzheimer disease: a machine learning study</article-title>. <source>Acad. Radiol.</source> <volume>30</volume>, <fpage>1874</fpage>&#x2013;<lpage>1884</lpage>. <pub-id pub-id-type="doi">10.1016/j.acra.2022.12.033</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saykin</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Foroud</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Potkin</surname>
<given-names>S. G.</given-names>
</name>
<name>
<surname>Swaminathan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Alzheimer&#x2019;s disease neuroimaging initiative biomarkers as quantitative phenotypes: genetics core aims, progress, and plans</article-title>. <source>Alzheimer&#x2019;s and Dementia</source> <volume>6</volume>, <fpage>265</fpage>&#x2013;<lpage>273</lpage>. <pub-id pub-id-type="doi">10.1016/j.jalz.2010.03.013</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schlachetzki</surname>
<given-names>J. C.</given-names>
</name>
<name>
<surname>Saliba</surname>
<given-names>S. W.</given-names>
</name>
<name>
<surname>de Oliveira</surname>
<given-names>A. C. P.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Studying neurodegenerative diseases in culture models</article-title>. <source>Rev. Bras. Psiquiatr.</source> <volume>35</volume>, <fpage>92</fpage>&#x2013;<lpage>100</lpage>. <pub-id pub-id-type="doi">10.1590/1516-4446-2013-1159</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sharma</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Guleria</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Tiwari</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A deep learning based convolutional neural network model with vgg16 feature extractor for the detection of alzheimer disease using mri scans</article-title>. <source>Meas. Sensors</source> <volume>24</volume>, <fpage>100506</fpage>. <pub-id pub-id-type="doi">10.1016/j.measen.2022.100506</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tharwat</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Classification assessment methods</article-title>. <source>Appl. Comput. Inf.</source> <volume>17</volume>, <fpage>168</fpage>&#x2013;<lpage>192</lpage>. <pub-id pub-id-type="doi">10.1016/j.aci.2018.08.003</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Troyanskaya</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Cantor</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sherlock</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Hastie</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tibshirani</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2001</year>). <article-title>Missing value estimation methods for dna microarrays</article-title>. <source>Bioinformatics</source> <volume>17</volume>, <fpage>520</fpage>&#x2013;<lpage>525</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/17.6.520</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uysal</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Ozturk</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Hippocampal atrophy based alzheimer&#x2019;s disease diagnosis via machine learning methods</article-title>. <source>J. Neurosci. Methods</source> <volume>337</volume>, <fpage>108669</fpage>. <pub-id pub-id-type="doi">10.1016/j.jneumeth.2020.108669</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Velazquez</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Multimodal ensemble model for alzheimer&#x2019;s disease conversion prediction from early mild cognitive impairment subjects</article-title>. <source>Comput. Biol. Med.</source> <volume>151</volume>, <fpage>106201</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.106201</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2023a</year>). <article-title>Hypergraph-regularized multimodal learning by graph diffusion for imaging genetics based alzheimer&#x2019;s disease diagnosis</article-title>. <source>Med. Image Anal.</source> <volume>89</volume>, <fpage>102883</fpage>. <pub-id pub-id-type="doi">10.1016/j.media.2023.102883</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>M. W. M.</given-names>
</name>
<name>
<surname>Shao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2023b</year>). <article-title>Hypergraph-regularized multimodal learning by graph diffusion for imaging genetics based alzheimer&#x2019;s disease diagnosis</article-title>. <source>Med. Image Anal.</source> <volume>89</volume>, <fpage>102883</fpage>. <pub-id pub-id-type="doi">10.1016/j.media.2023.102883</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xia</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Usman</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A novel method for diagnosing alzheimer&#x2019;s disease using deep pyramid cnn based on eeg signals</article-title>. <source>Heliyon</source> <volume>9</volume>, <fpage>148588</fpage>&#x2013;<lpage>e14912</lpage>. <pub-id pub-id-type="doi">10.1016/j.heliyon.2023.e14858</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2023a</year>). <article-title>Fuzzy-vgg: a fast deep learning method for predicting the staging of alzheimer&#x2019;s disease based on brain mri</article-title>. <source>Inf. Sci.</source> <volume>642</volume>, <fpage>119129</fpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2023.119129</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2023b</year>). <article-title>Fuzzy-vgg: a fast deep learning method for predicting the staging of alzheimer&#x2019;s disease based on brain mri</article-title>. <source>Inf. Sci.</source> <volume>642</volume>, <fpage>119129</fpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2023.119129</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Qing</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2023a</year>). <article-title>Multi-modal cross-attention network for alzheimer&#x2019;s disease diagnosis with multi-modality data</article-title>. <source>Comput. Biol. Med.</source> <volume>162</volume>, <fpage>107050</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.107050</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Qing</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023b</year>). <article-title>Multi-relation graph convolutional network for alzheimer&#x2019;s disease diagnosis using structural mri</article-title>. <source>Knowledge-Based Syst.</source> <volume>270</volume>, <fpage>110546</fpage>. <pub-id pub-id-type="doi">10.1016/j.knosys.2023.110546</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Ida-net: inheritable deformable attention network of structural mri for alzheimer&#x2019;s disease diagnosis</article-title>. <source>Biomed. Signal Process. Control</source> <volume>84</volume>, <fpage>104787</fpage>&#x2013;<lpage>104813</lpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2023.104787</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Anand</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Maximum relevance and minimum redundancy feature selection methods for a marketing machine learning platform</source>, <fpage>442</fpage>&#x2013;<lpage>452</lpage>.</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zuo</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Ng</surname>
<given-names>M. K.-P.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A new brain network construction paradigm for brain disorder via diffusion-based graph contrastive learning</article-title>. <source>IEEE Trans. Pattern Analysis Mach. Intell.</source> <volume>46</volume>, <fpage>10389</fpage>&#x2013;<lpage>10403</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2024.3442811</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zuo</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C. L. P.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Prior-guided adversarial learning with hypergraph for predicting abnormal connections in alzheimer&#x2019;s disease</article-title>. <source>IEEE Trans. Cybern.</source> <volume>54</volume>, <fpage>3652</fpage>&#x2013;<lpage>3665</lpage>. <pub-id pub-id-type="doi">10.1109/TCYB.2023.3344641</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zuo</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Brain structure-function fusing representation learning using adversarial decomposed-vae for analyzing mci</article-title>. <source>IEEE Trans. Neural Syst. Rehabilitation Eng.</source> <volume>31</volume>, <fpage>4017</fpage>&#x2013;<lpage>4028</lpage>. <pub-id pub-id-type="doi">10.1109/TNSRE.2023.3323432</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>