<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurol.</journal-id>
<journal-title>Frontiers in Neurology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurol.</abbrev-journal-title>
<issn pub-type="epub">1664-2295</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fneur.2025.1626922</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neurology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>PlgFormer: parallel extraction of local-global features for AD diagnosis on sMRI using a unified CNN-transformer architecture</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Wang</surname> <given-names>Guoxin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2761602/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Li</surname> <given-names>Yuxia</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Zhou</surname> <given-names>Zhiyi</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3176458/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>An</surname> <given-names>Shan</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2478870/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Cao</surname> <given-names>Xuyang</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3176084/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Jin</surname> <given-names>Yuxin</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Sun</surname> <given-names>Zhengqin</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Chen</surname> <given-names>Guanqun</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2756046/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Mingkai</given-names></name>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3064274/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Li</surname> <given-names>Zhixiong</given-names></name>
<xref ref-type="aff" rid="aff8"><sup>8</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1334644/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Yu</surname> <given-names>Feng</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c003"><sup>&#x0002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>College of Biomedical Engineering and Instrument Science, Zhejiang University</institution>, <addr-line>Hangzhou</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Tangshan Central Hospital, Tangshan</institution>, <addr-line>Hebei</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Electrical and Information Engineering, Tianjin University</institution>, <addr-line>Tianjin</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>School of Electrical and Information Engineering, Tianjin University</institution>, <addr-line>Tianjin</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>JD Health International Inc.</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff6"><sup>6</sup><institution>Department of Neurology, Beijing Chao-Yang Hospital, Capital Medical University</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff7"><sup>7</sup><institution>Department of Neurology, XuanWu Hospital of Capital Medical University</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff8"><sup>8</sup><institution>Karamay Integrated Traditional Chinese and Western Medicine Hospital (People&#x00027;s Hospital of Karamay)</institution>, <addr-line>Karamay</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Shang-Ming Zhou, University of Plymouth, United Kingdom</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Rakeshkumar Mahto, California State University, Fullerton, United States</p>
<p>Wang Liao, The Second Affiliated Hospital of Guangzhou Medical University, China</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Shan An <email>anshan&#x00040;tju.edu.cn</email></corresp>
<corresp id="c002">Zhixiong Li <email>865818683&#x00040;qq.com</email></corresp>
<corresp id="c003">Feng Yu <email>osfengyu&#x00040;zju.edu.cn</email></corresp>
<fn fn-type="equal" id="fn001"><p>&#x02020;These authors have contributed equally to this work</p></fn></author-notes>
<pub-date pub-type="epub">
<day>29</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1626922</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Wang, Li, Zhou, An, Cao, Jin, Sun, Chen, Zhang, Li and Yu.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Wang, Li, Zhou, An, Cao, Jin, Sun, Chen, Zhang, Li and Yu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Structural magnetic resonance imaging (sMRI) is an important tool for the early diagnosis of Alzheimer&#x00027;s disease (AD). Previous methods based on voxel, region of interests (ROIs) or patch have limitations in characterizing discriminative features in sMRI for AD as they can only focus on specific local or global features.</p></sec>
<sec>
<title>Methods</title>
<p>We propose a computer-aided AD diagnosis method based on sMRI, named PlgFormer, which considers the extraction of both local and global features. By using a combination of convolution and self-attention, we can extract context features at both local and global levels. In the decision-making layer of the model, we design a feature fusion module that adaptively selects context features through a gating mechanism. Additionally, to account for changes in image input resolution during the downsampling operation, we embed a dynamic embedding block at each stage of the network, which can adaptively adjust the weights of the inputs with different resolutions.</p></sec>
<sec>
<title>Results</title>
<p>We evaluated the performance of our method on dichotomous AD vs. normal control (NC) and mild cognitive impairment (MCI) vs. NC, as well as trichotomous AD vs. MCI vs. NC classification tasks, using publicly available ADNI and XWNI datasets that we collected. On the ADNI dataset, the proposed method achieves classification accuracies of 0.9431 for AD vs. NC, 0.8216 for MCI vs. CN, and 0.6228 for the AD vs. MCI vs. CN task. On the XWNI dataset, the corresponding accuracies are 0.9307, 0.8600, and 0.8672, respectively. The experimental results demonstrate the high precision and robustness of our method in diagnosing people with different stages of cognitive impairment.</p></sec>
<sec>
<title>Conclusion</title>
<p>The findings in our experimental results underscore the clinical potential of our proposed PlgFormer as a reliable and interpretable framework for supporting early and accurate diagnosis of AD using sMRI.</p></sec></abstract>
<kwd-group>
<kwd>Alzheimer&#x00027;s disease diagnosis</kwd>
<kwd>attention mechanism</kwd>
<kwd>computer-aided diagnosis</kwd>
<kwd>sMRI</kwd>
<kwd>multi-level feature fusion</kwd>
</kwd-group>
<counts>
<fig-count count="7"/>
<table-count count="8"/>
<equation-count count="13"/>
<ref-count count="46"/>
<page-count count="15"/>
<word-count count="10777"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Artificial Intelligence in Neurology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Alzheimer&#x00027;s disease (AD) is a neurodegenerative disorder that progressively impairs cognitive functions such as memory, thinking, and behavior. It is the most common cause of dementia and affects the normal lives of over 30 million people worldwide (<xref ref-type="bibr" rid="B1">1</xref>), primarily older adults. As the disease advances, individuals may experience difficulties with language, disorientation, mood changes, and eventually lose the ability to perform daily activities independently. These challenges not only degrade the quality of life but also place a heavy burden on families and healthcare systems. Early and accurate diagnosis of AD is therefore essential for effective intervention and management.</p>
<p>Diagnosis AD is based primarily on clinical evaluation, including medical history, physical examination, and cognitive tests (<xref ref-type="bibr" rid="B2">2</xref>, <xref ref-type="bibr" rid="B3">3</xref>). Unfortunately, effective treatments for AD have yet to be discovered. Therefore, early diagnosis is imperative not only for improving the quality of life for patients but also for supporting the development of more effective treatment methods and intervention measures.</p>
<p>Structural magnetic resonance imaging (sMRI) is a noninvasive medical imaging technique that captures the physical structure of the brain and provides detailed images of its tissues, including gray and white matter. Unlike functional magnetic resonance imaging (fMRI), which measures changes in blood flow and neural activity, sMRI uses strong magnetic fields and radio waves. sMRI has become a valuable tool in diagnosing AD due to its ability to detect changes in brain structure associated with the disease (<xref ref-type="bibr" rid="B4">4</xref>&#x02013;<xref ref-type="bibr" rid="B7">7</xref>), such as shrinkage in the hippocampus and other areas of the brain. However, its use is often limited by physician experience and time-consuming processes.</p>
<p>To solve this dilemma, hopes have gradually been placed on computer-aided diagnosis of AD, which has a long history and has achieved good performance. Existing methods for computer-aided diagnosis of AD using sMRI can be broadly classified into three categories: (1) Voxel-based methods; (2) Regions of interest (ROIs)-based methods; and (3) Patch-based methods. However, these methods face challenges due to the high dimensionality of sMRI data and the discrete distribution of AD lesions.</p>
<p>Voxel-based methods for computer-aided diagnosis of AD utilize the entire sMRI as input and extract global features for AD diagnosis. Hinrichs et al. (<xref ref-type="bibr" rid="B8">8</xref>) used gray matter density to extract discriminative features and employed a linear programming boosting method to classify AD and normal control (NC). Kao et al. (<xref ref-type="bibr" rid="B9">9</xref>) detected white matter changes throughout the whole sMRI to diagnose AD. Vounou et al. (<xref ref-type="bibr" rid="B10">10</xref>) detected markers associated with longitudinal changes in brain voxels caused by AD and used a sparse reduced-rank regression model to classify them. However, these methods extract features across the whole sMRI, resulting in high computational complexity and overfitting of the model due to limited data.</p>
<p>In contrast, ROI-based methods focus on extracting features from pre-segmented regions with lower feature dimensionality compared to voxel-based methods. Zhang et al. (<xref ref-type="bibr" rid="B11">11</xref>) adaptively extracted 93 ROIs using the atlas warping algorithm and identified AD with support vector machines (SVM). Liu et al. (<xref ref-type="bibr" rid="B12">12</xref>) proposed an ensemble classification model to construct gray matter density features within regions by using multiple spatially normalized templates. However, ROI selection is heavily dependent on specialist knowledge, making it difficult for developers to master. Additionally, ROI-based methods can only capture local discriminative features, making it difficult to capture global features such as ventricular volume and gyral sulcus morphological changes, which is insufficient for computer-aided AD diagnosis.</p>
<p>Patch-based methods extract features with intermediate feature dimensions, focusing more effectively on local discriminative features. Qiu et al. (<xref ref-type="bibr" rid="B13">13</xref>) trained a FCN to adaptively and randomly select meaningful patches, which were fed to a multilayer perceptron (MLP) for individual-level AD diagnosis. Zhang et al. (<xref ref-type="bibr" rid="B14">14</xref>) selected discriminative patches by computing shapley values and extracted local-global context features in sMRI using Convolutional Neural Network (CNN). However, how to combine local patch features with global representations is still a problem that needs to be explored.</p>
<p>We propose PlgFormer, a new method for AD diagnosis using sMRI. PlgFormer utilizes convolution modules and transformer modules in parallel to extract discriminative local-global context features in a unified manner. To adaptively adjust parameter size for inputs of different sizes, we embed dynamic convolutional layers in early stages of the model. This operation also reduces the number of parameters 72 compared to conventional convolution, as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>Various existing models were evaluated on ADNI dataset for the binary classification of AD and CN, and their performance was compared in relation to the number of model parameters. The graphical area corresponded to the number of model parameters. Our PlgFormer achieved optimal classification performance with the lightest number of parameters. The classification performance is illustrated in <xref ref-type="table" rid="T1">Table 1</xref>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fneur-16-1626922-g0001.tif">
<alt-text>Scatter plot illustrating model accuracy versus parameters in megabytes. Each point represents a model, differentiated by color and shape according to a legend. Accuracy ranges from 86% to 96%, and parameters range from 0 to 70MB. The PlgFormer model, represented by a pink circle, achieves the highest accuracy of approximately 95% with 4.36MB. The largest model, 3D-ResNet-34, is marked by a large green triangle with 60.53MB.</alt-text>
</graphic>
</fig>
<p>In summary, the major contributions of this paper can be summarized as follows.</p>
<list list-type="order">
<list-item><p>We design a novel dynamic embedding block (DEB) in our model, which is a combination of traditional convolution and dynamic convolution. Compared to traditional convolution, introducing dynamic convolution can help the model adaptively adjust the size of parameters based on the size of input, which reduces the model parameters while enhancing its robustness.</p></list-item>
<list-item><p>A dual-branch structure extracts local-global context features in sMRI in parallel. Considering feature alignment, we design both dual-branch operations as a multi-head architecture. Additionally, we introduce a feature fusion module (F2M) that adaptively selects local or global features at the decision-making layer of the model.</p></list-item>
<list-item><p>We collected a large-scale sMRI dataset called the XWNI dataset. Our extensive experiments, which included both the publicly available ADNI dataset and the collected XWNI dataset, show that our proposed PlgFormer substantially outperforms the existing baseline models while utilizing minimal parameters.</p></list-item>
</list></sec>
<sec id="s2">
<title>2 Related works</title>
<p>In this section, we provide a summary of current methods for computer-aided diagnosis of AD using sMRI data.</p>
<sec>
<title>2.1 Traditional machine learning for AD diagnosis</title>
<p>Early methods for computer-aided diagnosis of AD using sMRI data mainly focused on extracting features and using traditional machine learning methods to analyze and classify them. Ashburner and Friston (<xref ref-type="bibr" rid="B15">15</xref>) compared differences in brain structure between individuals or groups to identify brain regions associated with specific diseases or cognitive functions. Kl&#x000F6;ppel et al. (<xref ref-type="bibr" rid="B16">16</xref>) transformed brain MRI into statistical features and employed SVM to identify early structural changes in the brain of AD patients. Similarly, Fan et al. (<xref ref-type="bibr" rid="B17">17</xref>) split MRI into specified regions and selected the most discriminative regions for AD diagnosis based on the statistical characteristics of each region and the inter-relationship between them, and performed the classification using SVM. Hinrichs et al. (<xref ref-type="bibr" rid="B8">8</xref>) randomly augmented the raw ADNI data and adopted a linear programming boosting method for classification, achieving more robust results. sMRI data are always over dimensional. Cao et al. (<xref ref-type="bibr" rid="B18">18</xref>) proposed a multi-kernel-based method combined with marginal fisher analysis to reduce the feature dimensionality of MRI and establish a complex mapping relationship from image to disease. Zhang et al. (<xref ref-type="bibr" rid="B11">11</xref>) used principal component analysis (PCA) for feature extraction and dimensionality reduction and conducted classification of AD and CN instances using SVM. Abuhmed et al. (<xref ref-type="bibr" rid="B19">19</xref>) proposed two novel hybrid deep learning models, DFBL and MRBL, which integrate multivariate BiLSTM architecture with traditional machine learning models to enhance the prediction of Alzheimer&#x00027;s disease progression using multimodal time-series data.</p>
<p>However, these methods mentioned above present two limitations: (1) manual feature extraction is reliant on human experience and may ignore discriminative features; (2) high-dimensional features may lead to overfitting.</p></sec>
<sec>
<title>2.2 CNN-based methods for AD diagnosis</title>
<p>CNNs have achieved considerable amount of success in the field of computer vision, with their strong inductive bias performs well in tasks including image classification (<xref ref-type="bibr" rid="B20">20</xref>&#x02013;<xref ref-type="bibr" rid="B22">22</xref>), object detection, semantic segmentation, and video understanding. Naturally, these CNN-based methods were applied to AD diagnosis using MRI data, achieving satisfactory performances. Lian et al. (<xref ref-type="bibr" rid="B23">23</xref>) proposed a hierarchical full convolutional network (H-FCN) to automatically identify discriminative local patches associated with AD in the whole brain sMRI. Zhu et al. (<xref ref-type="bibr" rid="B24">24</xref>) proposed a dual attention multi-instance deep learning network to extract discriminative features from local patches and aggregate these features by attention-aware weights. Wu et al. (<xref ref-type="bibr" rid="B25">25</xref>) proposed a 3D CNN model to extract and integrate robust multiscale spatial features to promote AD computer-aided diagnosis. Zhang et al. (<xref ref-type="bibr" rid="B26">26</xref>) proposed a residual self-attention deep neural network to capture local spatial features in sMRI, while attention mechanisms are still necessary to jointly construct global representations. A multi-branch convolutional network was proposed in (<xref ref-type="bibr" rid="B27">27</xref>), where each branch extracts its own features independently and uses a fully connected layer for feature aggregation to obtain global representations. Zhang et al. (<xref ref-type="bibr" rid="B28">28</xref>) proposed a multi-relation reasoning network (MRN) that constructs brain graphs from sMRI data to capture spatial and topological relationships, improving Alzheimer&#x00027;s disease diagnosis through enhanced feature representation and global reasoning.</p>
<p>In addition to specialized CNN-based models, general models such as VGGNet (<xref ref-type="bibr" rid="B21">21</xref>) and ResNet (<xref ref-type="bibr" rid="B22">22</xref>) have also been applied to AD diagnosis due to their successful performance in natural visual scene analysis. Billones et al. (<xref ref-type="bibr" rid="B29">29</xref>) utilized a modified VGGNet model for AD diagnosis using sMRI data. Li et al. (<xref ref-type="bibr" rid="B30">30</xref>) employed the ResNet model to the ADNI dataset, focusing specifically on local discriminative features in hippocampal regions. These methods all use the strong inductive bias of CNNs to extend the conventional 2D convolutional kernels to 3D and extract the local representations associated with AD in sMRI.</p>
<p>The strong inductive bias of CNNs enables them to capture local features effectively even with a limited number of samples. However, when constructing global representations by integrating local information, attention mechanisms are typically still required.</p></sec>
<sec>
<title>2.3 Attention mechanism-based methods for AD diagnosis</title>
<p>Attention mechanisms excel at uniting different local features to construct global representations. Thanks to this, Wu et al. (<xref ref-type="bibr" rid="B25">25</xref>) aggregated the local features extracted by the convolution module under the supervision of the attention weights. similarly, Zhang et al. (<xref ref-type="bibr" rid="B26">26</xref>) directed the network to focus on critical information in the CNN feature maps and suppress non-essential features by using self-attention scores. Liu et al. (<xref ref-type="bibr" rid="B31">31</xref>) aggregated the semantic features between different channels by the classical Squeeze and Excitation (SE) module. Zhang et al. (<xref ref-type="bibr" rid="B32">32</xref>) embedded a connection-wise attention mechanism in DesNet for associating context features of the model. Jin et al. (<xref ref-type="bibr" rid="B33">33</xref>) constructed a simple convolutional layer for computing attention weights to retrace global features for local features extracted by convolution. Transformer is a pure attention mechanism architecture, originally proposed for natural language processing tasks (<xref ref-type="bibr" rid="B34">34</xref>) and has since been extended to the field of computer vision (<xref ref-type="bibr" rid="B35">35</xref>, <xref ref-type="bibr" rid="B36">36</xref>) demonstrating progressively satisfactory performance.</p>
<p>However, Jang and Hwang (<xref ref-type="bibr" rid="B37">37</xref>) have experimentally demonstrated that the coarse embedding of Transformer into a mature CNN architecture may not always yield satisfactory performances due to its lack of focus on local features, implying that attention may not be all need for AD diagnosis with sMRI data.</p></sec></sec>
<sec id="s3">
<title>3 Method</title>
<p>In this section, we present the details of the proposed method. The overview of the method is elaborated in Section 3.1. After that, dynamic embedding block, convolution and self-attention parallel extraction of local-global context features and the F2M are described in Section 3.2, Section 2.3, and Section 3.4, respectively.</p>
<sec>
<title>3.1 Overview of the proposed method</title>
<p>The strong inductive bias of the convolution is beneficial for local features extraction, while the self-attention is significant for the representation of global features. Both local and global features are important for AD diagnosis from sMRI, which means we cannot simply conduct convolution or self-attention operations. To achieve this, we propose an architecture that extracts local-global context features in parallel with a uniform multi-head self-attention mechanism. Our designed F2M in the decision-making layer of the network selects discriminative local-global context features through a gating mechanism. Additionally, since downsampling changes the input resolution at each stage, we embed a dynamic embedding block in each stage of the network to adjust the weights for different input resolutions adaptively.</p>
<p><xref ref-type="fig" rid="F2">Figure 2</xref> provides an overview of our proposed PlgFormer algorithm. Simply described, we take a vanilla 3-D sMRI <italic>X</italic>&#x02208;&#x0211D;<sup><italic>D</italic> &#x000D7; <italic>H</italic> &#x000D7; <italic>W</italic></sup> as an example. The training process of PlgFormer is depicted in <xref ref-type="table" rid="T9">Algorithm 1</xref>. The proposed PlgFormer consists of multi-stages, each of which conducts a downsampling operation to reduce the dimensionality of the data. To adapt to the change of input size, DEBs are introduced in each stage of the model to adaptively adjust the weights of convolution kernels for various resolutions. In the early stages of the model, Multi-Head-Self-Attention (MHSA) was introduced in a purely convolutional operation paired with multi-head style, without considering the computationally burdensome self-attention mechanism due to the large input size. Local and global discriminative features are then extracted using the convolution and self-attentive modules. To facilitate feature alignment before fusion, the convolution operation is designed as a multi-head structure unified with the self-attention mechanism. Local-global context features are fused with F2M and passed through global average pooling and full connectivity to obtain the one-hot tensor at the decision-making layer of the network.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>Overview of the proposed PlgFormer. This method comprises four stages of feature extraction. Each stage employs dynamic convolutional style to encode location information, MHSA<sub><italic>l</italic></sub> or MHSA<sub><italic>g</italic></sub> to extract local or global features, and a feature fusion module to combine these features and produce classification results. For further elaboration, please refer to Section 3. The figure schematically illustrates the data dimensions as a binary classification of AD and CN.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fneur-16-1626922-g0002.tif">
<alt-text>Flowchart showing a neural network architecture for image processing. It begins with a stack of four grayscale brain scans labeled as dimensions 1  &#x000D7;  D &#x000D7; H &#x000D7; W. These pass through four stages, each with patch embedding and MHS-A with FFN components. The stages progressively halve the dimensions from 16&#x000D7;D4&#x000D7;H4&#x000D7;W4 to 128&#x000D7;D32&#x000D7;H32&#x000D7;W32. Outputs are combined and fed into a fully connected layer for analysis.</alt-text>
</graphic>
</fig>
<table-wrap position="float" id="T9"> 
<label>Algorithm 1</label>
<caption><p>The training procedure of PlgFormer.</p></caption>
<table frame="hsides" rules="groups">
<tbody>
<tr><td align="left" valign="top"><monospace><bold>Input:</bold> input images <italic>X</italic>&#x02208;&#x0211D;<sup><italic>D</italic> &#x000D7; <italic>H</italic> &#x000D7; <italic>W</italic></sup></monospace></td></tr>
<tr><td align="left" valign="top"><monospace><bold>Output:</bold> outputs &#x00176;&#x02208;&#x0211D;<sup><italic>n</italic></sup> (n: number of classes)</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>1: <bold>repeat</bold></monospace></td></tr>
<tr><td align="left" valign="top"><monospace>2: Let stage <italic>k</italic> &#x0003D; 1, <italic>loss</italic> &#x0003D; 0.0</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>3: <italic>H</italic><sup>0</sup>&#x02190;DEB(<italic>X</italic>)</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>4: <bold>for</bold> <italic>k</italic> &#x0003D; 1 &#x02192; <italic>m</italic> <bold>do</bold> (<italic>m</italic>:number of stages)</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>5: &#x000A0;&#x000A0;&#x000A0;<bold>if</bold> <italic>k</italic> &#x0003C; &#x0003D; <italic>s</italic> <bold>then</bold> (<italic>s</italic>:number of shallow stages)</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>6: &#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;<italic>H</italic><sup><italic>k</italic></sup>&#x02190;MHSA<sup><italic>k</italic></sup>(<italic>H</italic><sup><italic>k</italic>&#x02212;1</sup>)</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>7: &#x000A0;&#x000A0;&#x000A0;<bold>else</bold></monospace></td></tr>
<tr><td align="left" valign="top"><monospace>8: &#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;<inline-formula><mml:math id="M3"><mml:msubsup><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02190;</mml:mo><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">MHSA</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">if</mml:mtext></mml:mstyle><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn><mml:mo>=</mml:mo><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">else</mml:mtext></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></monospace></td></tr>
<tr><td align="left" valign="top"><monospace>9: &#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;<inline-formula><mml:math id="M4"><mml:msubsup><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02190;</mml:mo><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">MHSA</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">if</mml:mtext></mml:mstyle><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn><mml:mo>=</mml:mo><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">else</mml:mtext></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></monospace></td></tr>
<tr><td align="left" valign="top"><monospace>10: &#x000A0;&#x000A0;&#x000A0;<italic>k</italic> &#x0003D; <italic>k</italic>&#x0002B;1</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>11: <bold>end for</bold></monospace></td></tr>
<tr><td align="left" valign="top"><monospace>12: <inline-formula><mml:math id="M5"><mml:msub><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>u</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">F2M</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></monospace></td></tr>
<tr><td align="left" valign="top"><monospace>13: &#x00176;&#x02190;AvgPool &#x0002B; FC(<italic>Z</italic><sub><italic>aug</italic></sub>)</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>14: <italic>loss</italic> &#x0003D; Loss_function(<italic>Y</italic>, &#x00176;)</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>15: &#x003B8;&#x02190;&#x02212;&#x02207;<sub>&#x003B8;</sub>(<italic>loss</italic>)</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>16: <bold>until</bold> convergence</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>17: <bold>return</bold> &#x00176;&#x02208;&#x0211D;<sup><italic>n</italic></sup></monospace></td></tr>
</tbody>
</table>
</table-wrap>
 <p>The key factors of our proposed method for effective AD diagnosis include: (1) how to encode dynamic positions for inputs of different resolutions; (2) how to extract local and global features in a uniform manner using convolution and self-attention operations for subsequent feature alignment; and (3) how to effectively fuse local-global context features. These solutions will be elaborated in Sections 3.2&#x02013;3.4.</p></sec>
<sec>
<title>3.2 Dynamic embedding block</title>
<p>Dynamic convolution can adjust the size of the convolution kernel according to the different sizes of the input data, effectively reducing the parameter count of the model. Unlike traditional convolutions that only use fixed kernel sizes, this type of operator also improves the robustness of convolution nerual nerworks, as the model can more quickly focus on discriminative features. Before executing the multi-head attention in each stage, the DEB uses a regular convolution operation to non-overlappingly divide the feature map of the original input into many patches (kernel_size, stride = patch_size). As shown in <xref ref-type="fig" rid="F3">Figure 3</xref>, before and after this regular convolution operation, the DEB introduces a dynamic convolutional layer to guide the subsequent multi-head module to focus on more meaningful features:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>H</mml:mi><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">DEB</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where DEB represents a single dynamic embedding block, consisting of two dynamic convolution layers and a vanilla convolution layer with a stride of 2 or 4 for downsampling. We introduce this block to integrate of all tokens before feeding them to the multi-headed self-attentive module. Such a design combining dynamic convolution has two benefits. First, the dynamic convolution operation is very friendly to varying input resolutions. Second, dynamic convolution is light-weight, which can largely alleviate the over-fitting problem encountered because of the small amount of data.</p>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>Detailed structure of the proposed DEB. The kernel size of 3D dynamic convolution is 1, while the normal 3D convolution with patch size as the kernel size and stride is used to downsample at a specific ratio. This structure enables the block to be more adaptive and reduces the number of parameters as the input image size decreases.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fneur-16-1626922-g0003.tif">
<alt-text>Diagram of a Dynamic Embedding Block showing a sequence of processes: Dynamic Conv 3D with GELU and batch normalization, Conv 3D with stride equal to patch size, another Dynamic Conv 3D with GELU and batch normalization, followed by Flatten, Layer Norm, and Reshape. Inputs are X and outputs are H.</alt-text>
</graphic>
</fig></sec>
<sec>
<title>3.3 Parallel local-global feature extraction</title>
<p>Positional encoding introduces relative position relationships between multiple tokens, which is essential for ViTs to capture sequential information in the sequence. Traditionally, absolute positional encoding was first introduced into ViTs (<xref ref-type="bibr" rid="B35">35</xref>, <xref ref-type="bibr" rid="B36">36</xref>), and this style is not friendly for different input resolutions. And relative positional encoding also does not always perform well due to its heavy computational burden (<xref ref-type="bibr" rid="B38">38</xref>). To improve efficiency, several recent works introducing convolutional positional encoding have been proposed to flexibly embed position relations (<xref ref-type="bibr" rid="B39">39</xref>, <xref ref-type="bibr" rid="B40">40</xref>), and we follow them.</p>
<p>As previously discussed, convolution operations are effective at capturing local contextual relationships, while self-attentive mechanisms excel at capturing global dependencies. Therefore, we propose a two-branch structure that concurrently extracts local and global context features, with one branch utilizing convolution operations and the other branch calculating attention distributions. This design enables efficient and effective hierarchical representation learning of local-global context features through multi-stage stacking. Additionally, for feature alignment, we implement both branches as unified multi-head styles, i.e., local MHSA: MHSA<sub><italic>l</italic></sub> and global MHSA: MHSA<sub><italic>g</italic></sub>,</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E3"><label>(3)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mi>W</mml:mi><mml:mo>*</mml:mo><mml:mtext class="textrm" mathvariant="normal">Concat</mml:mtext><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Given a series of tokens <italic>H</italic><sub><italic>n</italic></sub>, the token relation aggregators <italic>A</italic><sub><italic>n</italic></sub> capture the dependencies between them. Then, these relations are concatenated in the channel dimension, where <italic>W</italic> represents the learnable parameter matrix.</p>
<p><italic>1) Local MHSA:</italic> benefiting from the fact that convolution neural networks are specialized in focusing on local detailed features within a small region, MHSA<sub><italic>l</italic></sub> is designed as a pure convolutional structure without introducing self-attention operations, as described in <xref ref-type="fig" rid="F4">Figure 4</xref>. In particular, unlike the previous convolutional blocks, MHSA<sub><italic>l</italic></sub> follows a transformer-like style. We extract features from various &#x0201C;heads&#x0201D; using group convolution, and aggregate multi-head features in the channel dimension. Concretely, given a series of anchor tokens {<sub><italic>H</italic><sub><italic>i</italic></sub>}1:<italic>n</italic></sub> obtained by group convolution, the MHSA<sub><italic>l</italic></sub> learns the affinities between them by a local convolution operation in a small neighborhood &#x003A9;<sup><italic>d</italic> &#x000D7; <italic>h</italic> &#x000D7; <italic>w</italic></sup>:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msubsup><mml:mo>*</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mi>j</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mtext>&#x003A9;</mml:mtext></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>h</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>w</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M10"><mml:msubsup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>h</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>w</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> is the learnable convolutional kernel parameter matrix, and (<italic>i</italic>&#x02212;<italic>j</italic>) denotes the relative position between raw tokens <italic>H</italic><sub><italic>i</italic></sub> and <italic>H</italic><sub><italic>j</italic></sub>. By doing so, each head corresponds to each channel of the feature map, enabling us to extract local discriminative features in sMRI within a limited receptive field while preserving the spatial structure of the feature map.</p>
<fig position="float" id="F4">
<label>Figure 4</label>
<caption><p>The detailed structure diagram of MHSA<sub><italic>l</italic></sub>. Multi-head feature extraction is achieved by grouped convolution to extract local features, followed by point convolution for feature aggregation. Similar to traditional ViTs, we add residual connections for the model to trace back to previous features. MHSA<sub><italic>g</italic></sub>, on the other hand, extracts and aggregates global features through fully connected layers.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fneur-16-1626922-g0004.tif">
<alt-text>Diagram illustrating a neural network architecture. It starts with convolutional positional encoding, followed by batch normalization (BN). Grouped convolution transforms input, it&#x00027;s concatenated, then batch normalized again. A feedforward network (FFN) follows with linear transformations, a GELU activation, and ends with addition operations connecting back to earlier inputs.</alt-text>
</graphic>
</fig>
<p><italic>2) Global MHSA:</italic> self-attention is a natural method for capturing long-range dependencies between features and constructing discriminative global representations. In Vision Transformers (ViTs), a multi-head structure is frequently employed to process sequence information. Each &#x0201C;head&#x0201D; learns independent feature mappings and, ultimately, obtains a global view via linear aggregation:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M11"><mml:mrow><mml:msub><mml:mi>H</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:msub><mml:mi>Q</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>H</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mi>T</mml:mi></mml:msup><mml:msub><mml:mi>K</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>H</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:msup><mml:mi>j</mml:mi><mml:mo>&#x00027;</mml:mo></mml:msup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mi>&#x003A9;</mml:mi><mml:mrow><mml:mi>D</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:msub><mml:mi>Q</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>H</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mi>T</mml:mi></mml:msup><mml:msub><mml:mi>K</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:mo>&#x00027;</mml:mo></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:mstyle></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula>
<p>where <italic>Q</italic><sub><italic>n</italic></sub>(&#x000B7;) and <italic>K</italic><sub><italic>n</italic></sub>(&#x000B7;) are chosen as normal fully connected transformations and the above equation is a standard <italic>softmax</italic> operator. Note that <italic>j</italic>&#x02032;&#x02208;&#x003A9;<sup><italic>D</italic> &#x000D7; <italic>H</italic> &#x000D7; <italic>W</italic></sup> belongs to the global, signaling the concern of the MHSA<sub><italic>g</italic></sub> for global dependencies. To process the spatial dimensions of sMRI (depth <italic>D</italic>, height <italic>H</italic>, and width <italic>W</italic>), we convert them into a one-dimensional tensor and input it to the fully connected layer. Although traditionally this operator introduces significant computational burden, our MHSA<sub><italic>g</italic></sub> is located at the post-stage of the network, where it processes feature maps that have been downsampled through multiple stages. This allows for a balance between computational efficiency and accuracy.</p>
<p>Local-global context features in sMRI are aggregated in multiple stages through parallel feature extraction using MHSA<sub><italic>l</italic></sub> and MHSA<sub><italic>g</italic></sub>. Similarly to conventional ViTs, we introduce a feed-forward network (FFN) after each MHSA block. Our FFN has a specific architecture consisting of two linear layers wrapped around a nonlinear activation function (GELU). The first linear layer expands the channel dimension by the ratio of 4, while the next linear layer reduces it back to its original level. This operation facilitates the further filtering of features and improves the nonlinear expression of the model.</p>
<p>Considering the excessively large dimensions of the feature maps in the first two stages, it imposes a significant computational burden on the calculation of self-attention. Therefore, in the initial two stages of PlgFormer, the computation of MHSA adopts a locally based convolutional feature extraction module. Subsequently, a parallel local-global context extraction structure is employed to independently extract local features and aggregate global representations.</p></sec>
<sec>
<title>3.4 Feature fusion module</title>
<p>The F2M introduces a simple gating mechanism to enable adaptive feature selection, as depicted in <xref ref-type="fig" rid="F5">Figure 5</xref>. For local feature maps, a convolution with a larger kernel size of 5 &#x000D7; 5 &#x000D7; 5 is applied to focus on the more global features (the size of feature maps is 5 &#x000D7; 6 &#x000D7; 5 at this stage). Conversely, for global feature maps, point-by-point convolution is employed to preserve the global features of interest to self-attention. Finally, <italic>Z</italic><sub><italic>l</italic></sub> is passed through a sigmoid activation function, resulting in a mapping to values between 0 and 1 that controls the flow of local-global context features:</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>*</mml:mo><mml:msubsup><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E7"><label>(7)</label><mml:math id="M15"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>*</mml:mo><mml:msubsup><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E8"><label>(8)</label><mml:math id="M16"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>u</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>*</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>*</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>It&#x00027;s worth noting that, in the computation of multi-head self-attention, we drew inspiration from Zhang et al. (<xref ref-type="bibr" rid="B14">14</xref>) and introduced residual connections for the MHSA<sub><italic>g</italic></sub> module, while this design consideration was not applied to MHSA<sub><italic>l</italic></sub>. This choice is primarily made because residual connections assist deep networks in rapidly backpropagating shallow features. The convolutional module, having fewer network layers, does not necessitate the introduction of residual connections, as gradients can quickly propagate to the lower layers without them. However, for the globally self-attention operation based on fully connected layers, residual connections play a crucial role in facilitating better information propagation. They help alleviate the issue of gradient decay, ensuring improved preservation and dissemination of vital global information during global modeling. Here, <italic>W</italic><sub><italic>l</italic></sub> and <italic>W</italic><sub><italic>g</italic></sub> represent the learnable convolutional kernel parameter matrices, and &#x003C3;(&#x000B7;) denoting the sigmoid activation function. The resulting feature set <italic>Z</italic><sub><italic>aug</italic></sub> contains both global features aggregated by self-attention and local features extracted through convolution. We obtain <italic>Z</italic><sub><italic>aug</italic></sub> as a one-hot tensor following global average pooling of aggregated spatial features and projection using a fully connected layer.</p>
<fig position="float" id="F5">
<label>Figure 5</label>
<caption><p>The detailed structure of the F2M.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fneur-16-1626922-g0005.tif">
<alt-text>Diagram of a neural network architecture with two convolutional layers. Inputs Hgm and Hlm pass through convolution layers. Outputs Zg and Zl undergo sigmoid and Hadamard product operations. The results combine to generate Zaug.</alt-text>
</graphic>
</fig>
</sec></sec>
<sec id="s4">
<title>4 Experiments</title>
<p>In this section, we first present the two datasets used in this study in Section 4.1. Then, we present the specific experimental setup and evaluation metrics in Section 4.2. Subsequently, we introduce the comparison experiments with other existing methods in Section 4.3, and the ablation studies to validate the key components of the proposed model in Section 4.4.</p>
<sec>
<title>4.1 Datasets</title>
<p>In this study, we employed two independent datasets to validate the performance of our proposed method, i.e., the publicly available Alzheimer&#x00027;s Disease Neuroimaging Initiative (ADNI) dataset and the privately available Xuanwu Neuroimaging (XWNI) dataset.</p>
<p><italic>1) ADNI dataset:</italic> the ADNI is a large-scale, multicenter research study that aims to identify clinical, imaging, genetic, and biochemical biomarkers for the early detection and tracking of AD (<xref ref-type="bibr" rid="B41">41</xref>). This dataset has been extensively used in AD research, including studies on disease progression, diagnosis, and treatment. The availability of longitudinal data from multiple modalities makes it an invaluable resource for developing and evaluating machine learning algorithms for AD detection, prediction, and diagnosis. In this study, we utilized T1-weighted MRI scans from the ADNI dataset, which consists of various types of data, including clinical assessments, neuropsychological tests, MRI, PET, and genetic data. We only select data from ADNI1 dataset, and all sMRI data are acquired using a 1.5T MRI scanner. The ADNI dataset used in our study includes 1,779 samples, comprising 546 cases of NC (283 females, 263 males, 76.5 &#x000B1; 5.1 years), 840 cases of mild cognitive impairment (MCI; 336 females, 504 males, 75.5 &#x000B1; 7.1 years), and 393 cases of AD (198 females, 195 males, 75.3 &#x000B1; 7.6 years). The demographics of participants in ADNI dataset are shown in <xref ref-type="table" rid="T1">Table 1</xref>. Moreover, we use <xref ref-type="table" rid="T2">Table 2</xref> to show our training-testing splits number of ADNI datasets.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Demographics of participants in ADNI and XWNI datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left" colspan="3"></th>
<th valign="top" align="center"><bold>Gender (F/M)</bold></th>
<th valign="top" align="center"><bold>Age (years)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Datasets</td>
<td valign="top" align="left">ADNI (<italic>n</italic> = 1,779)</td>
<td valign="top" align="left">CN (<italic>n</italic> = 546)</td>
<td valign="top" align="center">283/263</td>
<td valign="top" align="center">76.5 &#x000B1; 5.1</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">MCI (<italic>n</italic> = 840)</td>
<td valign="top" align="center">336/504</td>
<td valign="top" align="center">75.5 &#x000B1; 7.1</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">AD (<italic>n</italic> = 393)</td>
<td valign="top" align="center">198/195</td>
<td valign="top" align="center">75.3 &#x000B1; 7.6</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">XWNI (<italic>n</italic> = 711)</td>
<td valign="top" align="left">CN (<italic>n</italic> = 515)</td>
<td valign="top" align="center">291/222<sup>&#x0002A;</sup></td>
<td valign="top" align="center">64.8 &#x000B1; 24.8</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">MCI (<italic>n</italic> = 47)</td>
<td valign="top" align="center">28/18<sup>&#x0002A;</sup></td>
<td valign="top" align="center">62.2 &#x000B1; 21.2</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">AD (<italic>n</italic> = 149)</td>
<td valign="top" align="center">97/52</td>
<td valign="top" align="center">70.4 &#x000B1; 30.6</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>F and M refer to female and male, respectively. Gender information is missing for 2 CN and 1 MCI subjects in the XWNI dataset and is marked with an asterisk (*).</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Training and testing split numbers of ADNI and XWNI datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Datasets</bold></th>
<th valign="top" align="center" colspan="6"><bold>ADNI</bold></th>
<th valign="top" align="center" colspan="6"><bold>XWNI</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td valign="top" align="center"><bold>Tasks</bold></td>
<td valign="top" align="center" colspan="2"><bold>CN vs. AD</bold></td>
<td valign="top" align="center" colspan="2"><bold>CN vs. MCI</bold></td>
<td valign="top" align="center" colspan="2"><bold>CN vs. MCI vs. AD</bold></td>
<td valign="top" align="center" colspan="2"><bold>CN vs. AD</bold></td>
<td valign="top" align="center" colspan="2"><bold>CN vs. MCI</bold></td>
<td valign="top" align="center" colspan="2"><bold>CN vs. MCI vs. AD</bold></td>
</tr>
 <tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="center"><bold>Train</bold></td>
<td valign="top" align="center"><bold>Test</bold></td>
<td valign="top" align="center"><bold>Train</bold></td>
<td valign="top" align="center"><bold>Test</bold></td>
<td valign="top" align="center"><bold>Train</bold></td>
<td valign="top" align="center"><bold>Test</bold></td>
<td valign="top" align="center"><bold>Train</bold></td>
<td valign="top" align="center"><bold>Test</bold></td>
<td valign="top" align="center"><bold>Train</bold></td>
<td valign="top" align="center"><bold>Test</bold></td>
<td valign="top" align="center"><bold>Train</bold></td>
<td valign="top" align="center"><bold>Test</bold></td>
</tr> <tr>
<td valign="top" align="left">CN</td>
<td valign="top" align="center">363</td>
<td valign="top" align="center">183</td>
<td valign="top" align="center">474</td>
<td valign="top" align="center">72</td>
<td valign="top" align="center">474</td>
<td valign="top" align="center">72</td>
<td valign="top" align="center">331</td>
<td valign="top" align="center">204</td>
<td valign="top" align="center">66</td>
<td valign="top" align="center">33</td>
<td valign="top" align="center">126</td>
<td valign="top" align="center">63</td>
</tr> <tr>
<td valign="top" align="left">AD</td>
<td valign="top" align="center">330</td>
<td valign="top" align="center">64</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">330</td>
<td valign="top" align="center">63</td>
<td valign="top" align="center">100</td>
<td valign="top" align="center">49</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">100</td>
<td valign="top" align="center">49</td>
</tr> <tr>
<td valign="top" align="left">MCI</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">486</td>
<td valign="top" align="center">354</td>
<td valign="top" align="center">486</td>
<td valign="top" align="center">93</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">30</td>
<td valign="top" align="center">17</td>
<td valign="top" align="center">30</td>
<td valign="top" align="center">17</td>
</tr></tbody>
</table>
</table-wrap>
<p><italic>2) XWNI dataset:</italic> the XWNI dataset was obtained from Xuanwu Hospital Capital Medical University, located in Beijing, China.<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref> This dataset includes sMRI data from patients diagnosed with AD, MCI, and NC. It stands out due to its large sample size and high-quality sMRI images. The sMRI data were acquired using a 3.0T MRI scanner and were preprocessed to improve image quality. The dataset contains both raw and preprocessed sMRI data, including skull-stripped and segmented images. A total of 711 samples were collected, including 515 cases of CN, 47 cases of MCI, and 149 cases of AD. The XWNI database is anticipated to be a valuable resource for researchers involved in AD diagnosis and related studies. The demographics of participants in XWNI dataset are shown in <xref ref-type="table" rid="T1">Table 1</xref>. Moreover, we use <xref ref-type="table" rid="T2">Table 2</xref> to show our training-testing splits number of XWNI datasets.</p>
<p><italic>3) Preprocessing:</italic> MNI152_T1 is a commonly used standard template for brain MRI image analysis and research, featuring a standard spatial coordinate system and brain structure information. Considering spatial resolution as a vital parameter affecting image quality and resolution, we initially paired the T1-weighted sMRI with the MNI152_T1_1mm template to obtain a more precise description of brain structure. Moreover, to avoid the computational burden from excessive sMRI data dimensionality, we cropped all images to <italic>D</italic> &#x000D7; <italic>H</italic> &#x000D7; <italic>W</italic>:148 &#x000D7; 192 &#x000D7; 156. These preprocessing methods can enhance the quality and effectiveness of sMRI data and lay the groundwork for subsequent analysis and research. <xref ref-type="fig" rid="F6">Figure 6</xref> illustrates the difference between sMRI images before and after preprocessing, indicating that the preprocessed images better capture the structural organization of brain tissue and demonstrate improved image quality.</p>
<fig position="float" id="F6">
<label>Figure 6</label>
<caption><p>Comparison of sMRI before and after preprocessing. <bold>(a&#x02013;c)</bold> Are raw sMRI from the axial view, the coronal view, and the sagittal view, respectively; <bold>(d&#x02013;f)</bold> are preprocessed brain images from the corresponding view that have been cropped after matching to the MNI152_T1_1mm template.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fneur-16-1626922-g0006.tif">
<alt-text>Six MRI brain scans showing different perspectives. (a) Axial view, (b) sagittal view, (c) coronal view, (d) axial view with clearer detail, (e) sagittal view with enhanced detail, (f) coronal view highlighting brain structure intricacies.</alt-text>
</graphic>
</fig>
<p>We utilized the medical image processing tool, MONAI, to transform and augment the dataset to adapt to the model input and enhance the robustness of model training, as described in <xref ref-type="table" rid="T3">Table 3</xref>. In the table, &#x0201C;Augmentation&#x0201D; refers to the functions encapsulated in MONAI, and &#x0201C;Values&#x0201D; refers to the specific parameters corresponding to the functions.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>The dataset was transformed and augmented using MONAI.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Dataset split</bold></th>
<th valign="top" align="left"><bold>Augmentation</bold></th>
<th valign="top" align="left"><bold>Values</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Training set</td>
<td valign="top" align="left">Rand spatial cropd</td>
<td valign="top" align="left">roi_size = [144, 176, 144]</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Histogram normalized</td>
<td valign="top" align="left">-</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Normalize intensityd</td>
<td valign="top" align="left">-</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Rand flipd</td>
<td valign="top" align="left">prob = 0.2 (dim:depth, heigth and width)</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Rand scale intensityd</td>
<td valign="top" align="left">factors = 0.1, prob = 1.0</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Rand shift intensityd</td>
<td valign="top" align="left">factors = 0.1, prob = 1.0</td>
</tr> <tr>
<td valign="top" align="left">Test set</td>
<td valign="top" align="left">Center spatial cropd</td>
<td valign="top" align="left">roi_size = [144, 176, 144]</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Histogram normalized</td>
<td valign="top" align="left">-</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Normalize intensityd</td>
<td valign="top" align="left">-</td>
</tr></tbody>
</table>
</table-wrap></sec>
<sec>
<title>4.2 Experimental setup and evaluation metrics</title>
<p>The proposed method was implemented using Python 3.7.0 and PyTorch 1.10.0. The GPU we used is one P40 with a memory of 24GB. We trained the network end-to-end using AdamW optimizer and optimize under the supervision of the cross-entropy loss function, without additional data for pre-training. PlgFormer consists of four stages, with 2, 2, 3, and 4 Multi-Head Self-Attention (MHSA) mechanisms allocated to each stage respectively. The DEB module uses 4 dynamic convolution kernels to enhance feature representation, an attention hidden ratio of 0.25 to balance performance and computational efficiency, and a temperature parameter initialized at 34 and decreased by 3 during training to progressively sharpen the attention distribution. The initial learning rate was set to 5e-4 and decayed with a cosine annealing strategy. To overcome early optimization difficulties, we performed a linear warm-up for the first 100 epochs, while the total training epochs were 400. Considering the extremely unbalanced number of samples contained in each category of the dataset, especially in the XWNI dataset, we used weighted random sampling to balance the number of samples contained in each category when loading the dataset to pursue better diagnostic performances. We conducted dichotomous (AD vs. CN and MCI vs. CN) and trichotomous (AD vs. MCI vs. CN) classifications to comprehensively evaluate the diagnostic performance of our model with varying degrees of cognitive impairment.</p>
<p>For model evaluation, we select a comprehensive set of metrics including accuracy (ACC), recall (REC), precision (PRE), F1-score (F1), and specificity (SPE). The calculation of each metric is as follows:</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M17"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">ACC</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E10"><label>(10)</label><mml:math id="M18"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">REC</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E11"><label>(11)</label><mml:math id="M19"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">PRE</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E12"><label>(12)</label><mml:math id="M20"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">F1</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mtext class="textrm" mathvariant="normal">REC</mml:mtext><mml:mo>&#x000D7;</mml:mo><mml:mtext class="textrm" mathvariant="normal">PRE</mml:mtext></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">REC</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:mtext class="textrm" mathvariant="normal">PRE</mml:mtext></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E13"><label>(13)</label><mml:math id="M21"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">SPE</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>TP, FP, TN</italic>, and <italic>FN</italic> denote true positive, false positive, true negative, and false negative, respectively. In practice, REC denotes the probability that a positive sample in the data set is correctly discriminated, and PRE denotes the proportion of all examples diagnosed as positive by the model that are in fact positive. Obviously, REC and PRE are numerically a pair of mutually exclusive indicators, so we calculate the F1-score to evaluate the performance of the model in a comprehensive way. And SPE indicates the ability of the model to determine negative samples. In addition, we also report the Area Under the Curve (AUC), which is the area under the Receiver Operating Characteristic Curve (ROC) curve, to measure the performance of the dichotomous model. is equal to 0.5, the model is equivalent to a random guess.</p></sec>
<sec>
<title>4.3 Comparison experiments</title>
<p>We compared our proposed method with several existing classification approaches fine-tuned on large-scale multimodal medical imaging datasets, including Med3D-ResNet-10, Med3D-ResNet-18, and Med3D-ResNet-34 (<xref ref-type="bibr" rid="B42">42</xref>). Additionally, we extended the original ViT model (<xref ref-type="bibr" rid="B35">35</xref>) to support 3D medical image classification, referred to as ViT-ori in this study. We also included methods specifically designed for Alzheimer&#x00027;s disease classification, covering both CNN-based and Transformer-based architectures, such as DA-MIDL (<xref ref-type="bibr" rid="B24">24</xref>), AMSNet (<xref ref-type="bibr" rid="B25">25</xref>), ResAttNet-10 (<xref ref-type="bibr" rid="B26">26</xref>), ResAttNet-18 (<xref ref-type="bibr" rid="B26">26</xref>), ViT-for-AD (<xref ref-type="bibr" rid="B43">43</xref>), and MCNEL (<xref ref-type="bibr" rid="B44">44</xref>). To comprehensively evaluate the performance of our approach, we conducted three sets of experiments on both datasets: (1) binary classification between AD and CN, (2) binary classification between MCI and CN, and (3) multi-class classification among AD, MCI, and CN subjects.</p>
<p><italic>1) Binary classification on AD and CN subjects:</italic> accurately identifying AD plays a crucial role in clinical diagnosis as well as computer-aided diagnosis. AD is characterized by brain atrophy, gray matter atrophy, reduced brain tissue, white matter damage, enlarged sulcal gyrus, and ventricles, which can be easily detected by physicians or computers on sMRI. In this study, we employed PlgFormer to set the embedded dimension of each stage to 16, 32, 64, and 128, and each head dimension to 16. The weight decay, learning rate, and batch size were set to 5e-4, 2e-4, and 16, respectively. We conducted comparison experiments on ADNI and XWNI datasets, and the results are presented in <xref ref-type="table" rid="T4">Table 4</xref>.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Quantitative comparison of our proposed PlgFormer and other existing methods for AD and CN binary classification on ADNI and XWNI datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold>ACC</bold></th>
<th valign="top" align="center"><bold>REC</bold></th>
<th valign="top" align="center"><bold>PRE</bold></th>
<th valign="top" align="center"><bold>F1</bold></th>
<th valign="top" align="center"><bold>SPE</bold></th>
<th valign="top" align="center"><bold>AUC</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1;">
<td valign="top" align="left" colspan="7"><bold>ADNI dataset</bold></td>
</tr> <tr>
<td valign="top" align="left">Med3D-ResNet-10 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.8780</td>
<td valign="top" align="center">0.9365</td>
<td valign="top" align="center">0.6941</td>
<td valign="top" align="center">0.7973</td>
<td valign="top" align="center">0.8579</td>
<td valign="top" align="center">0.8972</td>
</tr> <tr>
<td valign="top" align="left">Med3D-ResNet-18 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.8943</td>
<td valign="top" align="center">0.9206</td>
<td valign="top" align="center">0.7342</td>
<td valign="top" align="center">0.8169</td>
<td valign="top" align="center">0.8852</td>
<td valign="top" align="center">0.9029</td>
</tr> <tr>
<td valign="top" align="left">Med3D-ResNet-34 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.8984</td>
<td valign="top" align="center">0.9048</td>
<td valign="top" align="center">0.7500</td>
<td valign="top" align="center">0.8201</td>
<td valign="top" align="center">0.8962</td>
<td valign="top" align="center">0.9005</td>
</tr> <tr>
<td valign="top" align="left">DA-MIDL (<xref ref-type="bibr" rid="B24">24</xref>)</td>
<td valign="top" align="center">0.9106</td>
<td valign="top" align="center">0.7619</td>
<td valign="top" align="center">0.8727</td>
<td valign="top" align="center">0.8136</td>
<td valign="top" align="center">0.9617</td>
<td valign="top" align="center">0.8618</td>
</tr> <tr>
<td valign="top" align="left">AMSNet (<xref ref-type="bibr" rid="B25">25</xref>)</td>
<td valign="top" align="center">0.8902</td>
<td valign="top" align="center">0.9206</td>
<td valign="top" align="center">0.7250</td>
<td valign="top" align="center">0.8112</td>
<td valign="top" align="center">0.8798</td>
<td valign="top" align="center">0.9002</td>
</tr> <tr>
<td valign="top" align="left">ResAttNet-10 (<xref ref-type="bibr" rid="B26">26</xref>)</td>
<td valign="top" align="center">0.9065</td>
<td valign="top" align="center">0.8730</td>
<td valign="top" align="center">0.7857</td>
<td valign="top" align="center">0.8271</td>
<td valign="top" align="center">0.9180</td>
<td valign="top" align="center">0.8955</td>
</tr> <tr>
<td valign="top" align="left">ResAttNet-18 (<xref ref-type="bibr" rid="B26">26</xref>)</td>
<td valign="top" align="center">0.8821</td>
<td valign="top" align="center">0.7302</td>
<td valign="top" align="center">0.7931</td>
<td valign="top" align="center">0.7603</td>
<td valign="top" align="center">0.9344</td>
<td valign="top" align="center">0.8323</td>
</tr> <tr>
<td valign="top" align="left">ViT-ori (<xref ref-type="bibr" rid="B35">35</xref>)</td>
<td valign="top" align="center">0.6870</td>
<td valign="top" align="center">0.8095</td>
<td valign="top" align="center">0.4397</td>
<td valign="top" align="center">0.5698</td>
<td valign="top" align="center">0.6448</td>
<td valign="top" align="center">0.7272</td>
</tr> <tr>
<td valign="top" align="left">ViT-for-AD (<xref ref-type="bibr" rid="B43">43</xref>)</td>
<td valign="top" align="center">0.9000</td>
<td valign="top" align="center"><bold>1.0000</bold></td>
<td valign="top" align="center">0.8750</td>
<td valign="top" align="center"><bold>0.9333</bold></td>
<td valign="top" align="center">0.6667</td>
<td valign="top" align="center">0.8335</td>
</tr> <tr>
<td valign="top" align="left">MCNEL (<xref ref-type="bibr" rid="B44">44</xref>)</td>
<td valign="top" align="center">0.8998</td>
<td valign="top" align="center">0.8750</td>
<td valign="top" align="center">0.8750</td>
<td valign="top" align="center">0.8750</td>
<td valign="top" align="center">0.9167</td>
<td valign="top" align="center">0.8958</td>
</tr> <tr>
<td valign="top" align="left"><bold>PlgFormer (ours)</bold></td>
<td valign="top" align="center"><bold>0.9431</bold></td>
<td valign="top" align="center">0.8730</td>
<td valign="top" align="center"><bold>0.9016</bold></td>
<td valign="top" align="center">0.8871</td>
<td valign="top" align="center"><bold>0.9672</bold></td>
<td valign="top" align="center"><bold>0.9201</bold></td>
</tr> <tr style="background-color:#dee1e1;">
<td valign="top" align="left" colspan="7"><bold>XWNI dataset</bold></td>
</tr> <tr>
<td valign="top" align="left">Med3D-ResNet-10 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.8933</td>
<td valign="top" align="center">0.8367</td>
<td valign="top" align="center">0.6833</td>
<td valign="top" align="center">0.7523</td>
<td valign="top" align="center">0.9069</td>
<td valign="top" align="center">0.8769</td>
</tr> <tr>
<td valign="top" align="left">Med3D-ResNet-18 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.9012</td>
<td valign="top" align="center">0.8163</td>
<td valign="top" align="center">0.7143</td>
<td valign="top" align="center">0.7872</td>
<td valign="top" align="center">0.9216</td>
<td valign="top" align="center">0.8718</td>
</tr> <tr>
<td valign="top" align="left">Med3D-ResNet-34 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.8814</td>
<td valign="top" align="center">0.8367</td>
<td valign="top" align="center">0.6508</td>
<td valign="top" align="center">0.7321</td>
<td valign="top" align="center">0.8922</td>
<td valign="top" align="center">0.8648</td>
</tr> <tr>
<td valign="top" align="left">DA-MIDL (<xref ref-type="bibr" rid="B24">24</xref>)</td>
<td valign="top" align="center">0.9091</td>
<td valign="top" align="center">0.7347</td>
<td valign="top" align="center">0.7826</td>
<td valign="top" align="center">0.7579</td>
<td valign="top" align="center"><bold>0.9510</bold></td>
<td valign="top" align="center">0.8428</td>
</tr> <tr>
<td valign="top" align="left">AMSNet (<xref ref-type="bibr" rid="B25">25</xref>)</td>
<td valign="top" align="center">0.9209</td>
<td valign="top" align="center">0.8163</td>
<td valign="top" align="center">0.7843</td>
<td valign="top" align="center">0.8000</td>
<td valign="top" align="center">0.9461</td>
<td valign="top" align="center">0.8812</td>
</tr> <tr>
<td valign="top" align="left">ResAttNet-10 (<xref ref-type="bibr" rid="B26">26</xref>)</td>
<td valign="top" align="center">0.9091</td>
<td valign="top" align="center">0.7959</td>
<td valign="top" align="center">0.7501</td>
<td valign="top" align="center">0.7723</td>
<td valign="top" align="center">0.9363</td>
<td valign="top" align="center">0.8781</td>
</tr> <tr>
<td valign="top" align="left">ResAttNet-18 (<xref ref-type="bibr" rid="B26">26</xref>)</td>
<td valign="top" align="center">0.9130</td>
<td valign="top" align="center">0.8776</td>
<td valign="top" align="center">0.7288</td>
<td valign="top" align="center">0.7963</td>
<td valign="top" align="center">0.9216</td>
<td valign="top" align="center">0.8996</td>
</tr> <tr>
<td valign="top" align="left">ViT-ori (<xref ref-type="bibr" rid="B35">35</xref>)</td>
<td valign="top" align="center">0.8735</td>
<td valign="top" align="center">0.6226</td>
<td valign="top" align="center">0.6889</td>
<td valign="top" align="center">0.6596</td>
<td valign="top" align="center">0.9314</td>
<td valign="top" align="center">0.7820</td>
</tr> <tr>
<td valign="top" align="left">ViT-for-AD (<xref ref-type="bibr" rid="B43">43</xref>)</td>
<td valign="top" align="center">0.8824</td>
<td valign="top" align="center">0.9000</td>
<td valign="top" align="center"><bold>0.9000</bold></td>
<td valign="top" align="center">0.9333</td>
<td valign="top" align="center">0.9167</td>
<td valign="top" align="center">0.8958</td>
</tr> <tr>
<td valign="top" align="left">MCNEL (<xref ref-type="bibr" rid="B44">44</xref>)</td>
<td valign="top" align="center">0.9105</td>
<td valign="top" align="center">0.9050</td>
<td valign="top" align="center">0.8700</td>
<td valign="top" align="center"><bold>0.9400</bold></td>
<td valign="top" align="center">0.9300</td>
<td valign="top" align="center">0.9100</td>
</tr> <tr>
<td valign="top" align="left"><bold>PlgFormer (ours)</bold></td>
<td valign="top" align="center"><bold>0.9407</bold></td>
<td valign="top" align="center"><bold>0.9184</bold></td>
<td valign="top" align="center">0.8036</td>
<td valign="top" align="center">0.8517</td>
<td valign="top" align="center">0.9461</td>
<td valign="top" align="center"><bold>0.9322</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values indicate the best value in the current column.</p>
</table-wrap-foot>
</table-wrap>
<p>As elaborated in <xref ref-type="table" rid="T4">Table 4</xref>, our proposed PlgFormer achieves superior performance on ADNI and XWNI datasets in most cases. PlgFormer outperforms other methods in most of the evaluation metrics on both datasets, with ACC = 0.9431, PRE = 0.9016, F1 = 0.8871, SPE = 0.9672, and AUC = 0.9201 on ADNI dataset and ACC = 0.9407, REC = 0.9184, PRE = 0.8036, F1 = 0.8517, and AUC = 0.9021 on XWNI dataset. Although PlgFormer did not achieve the best performance in some metrics, such as REC on ADNI dataset and SPE on XWNI dataset, it notably outperformed other existing methods on several metrics. Notably, PlgFormer achieved the highest accuracy, suggesting its excellent performance in diagnosing AD patients. These results demonstrate the effectiveness and potential of using PlgFormer for binary classification in AD and CN, which can aid in early diagnosis and intervention for AD.</p>
<p><italic>2) Binary classification on MCI and CN subjects:</italic> identifying individuals with MCI is a crucial task for early intervention in AD, especially since there are currently no well-established treatment options for AD. However, using sMRI alone to perform this task is challenging, as the brain regions of patients with MCI usually do not undergo significant morphological changes. In this study, we set the embedding dimension of each stage of PlgFormer to 16, 32, 64, 96, while keeping the other hyperparameters the same as those for the AD and CN binary classification. The results of a detailed comparison with other existing methods are presented in <xref ref-type="table" rid="T5">Table 5</xref>.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Quantitative comparison of our proposed PlgFormer and other existing methods for MCI and CN binary classification on ADNI and XWNI datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold>ACC</bold></th>
<th valign="top" align="center"><bold>REC</bold></th>
<th valign="top" align="center"><bold>PRE</bold></th>
<th valign="top" align="center"><bold>F1</bold></th>
<th valign="top" align="center"><bold>SPE</bold></th>
<th valign="top" align="center"><bold>AUC</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1;">
<td valign="top" align="left" colspan="7"><bold>ADNI dataset</bold></td>
</tr> <tr>
<td valign="top" align="left">Med3D-ResNet-10 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.7160</td>
<td valign="top" align="center">0.7260</td>
<td valign="top" align="center">0.9146</td>
<td valign="top" align="center">0.8094</td>
<td valign="top" align="center">0.6667</td>
<td valign="top" align="center">0.6963</td>
</tr> <tr>
<td valign="top" align="left">Med3D-ResNet-18 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.7958</td>
<td valign="top" align="center">0.8192</td>
<td valign="top" align="center">0.9265</td>
<td valign="top" align="center">0.8696</td>
<td valign="top" align="center">0.6806</td>
<td valign="top" align="center">0.7499</td>
</tr> <tr>
<td valign="top" align="left">Med3D-ResNet-34 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.7981</td>
<td valign="top" align="center"><bold>0.8531</bold></td>
<td valign="top" align="center">0.8978</td>
<td valign="top" align="center">0.8751</td>
<td valign="top" align="center">0.5278</td>
<td valign="top" align="center">0.6904</td>
</tr> <tr>
<td valign="top" align="left">DA-MIDL (<xref ref-type="bibr" rid="B24">24</xref>)</td>
<td valign="top" align="center">0.7887</td>
<td valign="top" align="center">0.8192</td>
<td valign="top" align="center">0.9177</td>
<td valign="top" align="center">0.8657</td>
<td valign="top" align="center"><bold>0.8194</bold></td>
<td valign="top" align="center">0.7077</td>
</tr> <tr>
<td valign="top" align="left">AMSNet (<xref ref-type="bibr" rid="B25">25</xref>)</td>
<td valign="top" align="center">0.7840</td>
<td valign="top" align="center">0.8362</td>
<td valign="top" align="center">0.8970</td>
<td valign="top" align="center">0.8655</td>
<td valign="top" align="center">0.5278</td>
<td valign="top" align="center">0.6820</td>
</tr> <tr>
<td valign="top" align="left">ResAttNet-10 (<xref ref-type="bibr" rid="B26">26</xref>)</td>
<td valign="top" align="center">0.7864</td>
<td valign="top" align="center">0.8305</td>
<td valign="top" align="center">0.9046</td>
<td valign="top" align="center">0.8660</td>
<td valign="top" align="center">0.5694</td>
<td valign="top" align="center">0.7000</td>
</tr> <tr>
<td valign="top" align="left">ResAttNet-18 (<xref ref-type="bibr" rid="B26">26</xref>)</td>
<td valign="top" align="center">0.7934</td>
<td valign="top" align="center">0.8390</td>
<td valign="top" align="center">0.9055</td>
<td valign="top" align="center">0.8710</td>
<td valign="top" align="center">0.5694</td>
<td valign="top" align="center">0.7042</td>
</tr> <tr>
<td valign="top" align="left">ViT-for-AD (<xref ref-type="bibr" rid="B43">43</xref>)</td>
<td valign="top" align="center">0.7959</td>
<td valign="top" align="center">0.6250</td>
<td valign="top" align="center">0.7143</td>
<td valign="top" align="center">0.6667</td>
<td valign="top" align="center">0.8788</td>
<td valign="top" align="center">0.7368</td>
</tr> <tr>
<td valign="top" align="left">MCNEL (<xref ref-type="bibr" rid="B44">44</xref>)</td>
<td valign="top" align="center">0.8000</td>
<td valign="top" align="center">0.7500</td>
<td valign="top" align="center">0.7500</td>
<td valign="top" align="center">0.7500</td>
<td valign="top" align="center">0.8333</td>
<td valign="top" align="center">0.7107</td>
</tr> <tr>
<td valign="top" align="left"><bold>PlgFormer (ours)</bold></td>
<td valign="top" align="center"><bold>0.8216</bold></td>
<td valign="top" align="center">0.8503</td>
<td valign="top" align="center"><bold>0.9290</bold></td>
<td valign="top" align="center"><bold>0.8879</bold></td>
<td valign="top" align="center">0.6806</td>
<td valign="top" align="center"><bold>0.7654</bold></td>
</tr> <tr style="background-color:#dee1e1;">
<td valign="top" align="left" colspan="7"><bold>XWNI dataset</bold></td>
</tr> <tr>
<td valign="top" align="left">Med3D-ResNet-10 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.7800</td>
<td valign="top" align="center">0.8824</td>
<td valign="top" align="center">0.6250</td>
<td valign="top" align="center">0.7317</td>
<td valign="top" align="center">0.7273</td>
<td valign="top" align="center">0.7796</td>
</tr> <tr>
<td valign="top" align="left">Med3D-ResNet-18 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.8000</td>
<td valign="top" align="center">0.7647</td>
<td valign="top" align="center">0.6842</td>
<td valign="top" align="center">0.7222</td>
<td valign="top" align="center">0.8182</td>
<td valign="top" align="center">0.7832</td>
</tr> <tr>
<td valign="top" align="left">Med3D-ResNet-34 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.8200</td>
<td valign="top" align="center">0.7059</td>
<td valign="top" align="center">0.7500</td>
<td valign="top" align="center">0.7273</td>
<td valign="top" align="center">0.8788</td>
<td valign="top" align="center">0.7905</td>
</tr> <tr>
<td valign="top" align="left">DA-MIDL (<xref ref-type="bibr" rid="B24">24</xref>)</td>
<td valign="top" align="center">0.8000</td>
<td valign="top" align="center">0.7059</td>
<td valign="top" align="center">0.7059</td>
<td valign="top" align="center">0.7059</td>
<td valign="top" align="center">0.8485</td>
<td valign="top" align="center">0.7745</td>
</tr> <tr>
<td valign="top" align="left">AMSNet (<xref ref-type="bibr" rid="B25">25</xref>)</td>
<td valign="top" align="center">0.8000</td>
<td valign="top" align="center">0.8235</td>
<td valign="top" align="center">0.6667</td>
<td valign="top" align="center">0.7368</td>
<td valign="top" align="center">0.7879</td>
<td valign="top" align="center">0.8057</td>
</tr> <tr>
<td valign="top" align="left">ResAttNet-10 (<xref ref-type="bibr" rid="B26">26</xref>)</td>
<td valign="top" align="center">0.8000</td>
<td valign="top" align="center">0.8824</td>
<td valign="top" align="center">0.6522</td>
<td valign="top" align="center">0.7500</td>
<td valign="top" align="center">0.7576</td>
<td valign="top" align="center">0.8200</td>
</tr> <tr>
<td valign="top" align="left">ResAttNet-18 (<xref ref-type="bibr" rid="B26">26</xref>)</td>
<td valign="top" align="center">0.7800</td>
<td valign="top" align="center"><bold>0.9412</bold></td>
<td valign="top" align="center">0.6154</td>
<td valign="top" align="center">0.7442</td>
<td valign="top" align="center">0.6970</td>
<td valign="top" align="center">0.8191</td>
</tr> <tr>
<td valign="top" align="left">ViT-for-AD (<xref ref-type="bibr" rid="B43">43</xref>)</td>
<td valign="top" align="center">0.8235</td>
<td valign="top" align="center">0.9000</td>
<td valign="top" align="center"><bold>0.8182</bold></td>
<td valign="top" align="center">0.8571</td>
<td valign="top" align="center">0.7143</td>
<td valign="top" align="center">0.8285</td>
</tr> <tr>
<td valign="top" align="left">MCNEL (<xref ref-type="bibr" rid="B44">44</xref>)</td>
<td valign="top" align="center">0.8367</td>
<td valign="top" align="center">0.6875</td>
<td valign="top" align="center">0.7851</td>
<td valign="top" align="center">0.7333</td>
<td valign="top" align="center"><bold>0.9090</bold></td>
<td valign="top" align="center">0.8421</td>
</tr> <tr>
<td valign="top" align="left"><bold>PlgFormer (ours)</bold></td>
<td valign="top" align="center"><bold>0.8600</bold></td>
<td valign="top" align="center">0.8235</td>
<td valign="top" align="center">0.7778</td>
<td valign="top" align="center"><bold>0.8000</bold></td>
<td valign="top" align="center">0.8788</td>
<td valign="top" align="center"><bold>0.8512</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values indicate the best value in the current column.</p>
</table-wrap-foot>
</table-wrap>
<p>The results presented in <xref ref-type="table" rid="T5">Table 5</xref> demonstrate that our proposed PlgFormer outperforms other existing methods on both ADNI and XWNI datasets, with the highest metrics achieved in most cases. For example, on the ADNI dataset, our method achieved a PRE of 0.9290 and an AUC of 0.7654, while on the XWNI dataset, our method achieved a PRE of 0.7778 and an AUC of 0.8512. It is worth noting that MCI patients exhibit less significant structural alterations in brain regions on sMRI compared to AD patients. Consequently, the binary performance of the MCI and CN subjects was not as good as that of the AD and CN subjects on both data sets. As illustrated in <xref ref-type="table" rid="T5">Table 5</xref>, our PlgFormer achieves notable performance in F1 scores on both datasets, thus demonstrating its ability to distinguish between MCI and CN subjects.</p>
<p><italic>3) Triple classification on AD, MCI, and CN subjects:</italic> performing a triple classification experiment using sMRI on AD, MCI, and CN subjects is of great significance for computer-aided AD diagnosis. Currently, there is no effective cure for AD, underscoring the importance of early definitive detection to enable early intervention and treatment. MCI, as a transitional stage between normal aging and dementia, significantly increases the risk of developing AD. Therefore, accurately diagnosing and differentiating individuals</p>
<p>with MCI and AD from those with normal cognitive function (CN) is crucial to facilitate the early detection and management of AD.</p>
<p><xref ref-type="table" rid="T6">Table 6</xref> presents quantitative comparisons, revealing that our proposed PlgFormer outperforms other existing methods, particularly in the recognition of AD subjects, notably in the ADNI dataset. We attribute this to the flexibility of PlgFormer in addressing both local and global requirements, making it more sensitive to significant structural changes in brain regions. In general, our method demonstrated remarkable performance on most metrics, providing compelling evidence of its potential use in clinical practice for computer-aided diagnosis. Furthermore, our review of existing studies utilizing sMRI for AD diagnosis reveals that limited attention has been given to the three-way classification of AD, MCI, and CN. We acknowledge the inherent difficulty of this task, as current models often struggle to capture the subtle and discriminative features present in sMRI. Nevertheless, given its substantial clinical relevance, we strongly encourage future research to further explore and address this challenging problem.</p>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Quantitative comparison of our proposed PlgFormer and other existing methods for AD MCI and CN triple classification on ADNI and XWNI datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th/>
<th valign="top" align="center" colspan="4"><bold>AD</bold></th>
<th valign="top" align="center" colspan="4"><bold>MCI</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="center"><bold>ACC</bold></td>
<td valign="top" align="center"><bold>REC</bold></td>
<td valign="top" align="center"><bold>PRE</bold></td>
<td valign="top" align="center"><bold>F1</bold></td>
<td valign="top" align="center"><bold>SPE</bold></td>
<td valign="top" align="center"><bold>REC</bold></td>
<td valign="top" align="center"><bold>PRE</bold></td>
<td valign="top" align="center"><bold>F1</bold></td>
<td valign="top" align="center"><bold>SPE</bold></td>
</tr> <tr style="background-color:#dee1e1;">
<td valign="top" align="left" colspan="10"><bold>ADNI dataset</bold></td>
</tr> <tr>
<td valign="top" align="left">Med3D-ResNet-10(<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.5864</td>
<td valign="top" align="center">0.4000</td>
<td valign="top" align="center">0.2424</td>
<td valign="top" align="center">0.3019</td>
<td valign="top" align="center">0.8111</td>
<td valign="top" align="center">0.6036</td>
<td valign="top" align="center">0.7913</td>
<td valign="top" align="center">0.6848</td>
<td valign="top" align="center">0.5726</td>
</tr> <tr>
<td valign="top" align="left">Med3D-ResNet-18 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.5908</td>
<td valign="top" align="center">0.3000</td>
<td valign="top" align="center">0.2222</td>
<td valign="top" align="center">0.2553</td>
<td valign="top" align="center">0.8413</td>
<td valign="top" align="center">0.6126</td>
<td valign="top" align="center">0.7969</td>
<td valign="top" align="center">0.6927</td>
<td valign="top" align="center">0.5806</td>
</tr> <tr>
<td valign="top" align="left">Med3D-ResNet-34 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.5996</td>
<td valign="top" align="center">0.4333</td>
<td valign="top" align="center">0.3377</td>
<td valign="top" align="center">0.3796</td>
<td valign="top" align="center">0.8715</td>
<td valign="top" align="center">0.6006</td>
<td valign="top" align="center"><bold>0.8197</bold></td>
<td valign="top" align="center">0.6932</td>
<td valign="top" align="center">0.6452</td>
</tr> <tr>
<td valign="top" align="left">DA-MIDL (<xref ref-type="bibr" rid="B24">24</xref>)</td>
<td valign="top" align="center">0.6053</td>
<td valign="top" align="center">0.3968</td>
<td valign="top" align="center">0.2778</td>
<td valign="top" align="center">0.3268</td>
<td valign="top" align="center">0.8474</td>
<td valign="top" align="center"><bold>0.6582</bold></td>
<td valign="top" align="center">0.7664</td>
<td valign="top" align="center"><bold>0.7082</bold></td>
<td valign="top" align="center">0.4741</td>
</tr> <tr>
<td valign="top" align="left">AMSNet (<xref ref-type="bibr" rid="B25">25</xref>)</td>
<td valign="top" align="center">0.6012</td>
<td valign="top" align="center">0.4603</td>
<td valign="top" align="center">0.2929</td>
<td valign="top" align="center">0.3580</td>
<td valign="top" align="center">0.8357</td>
<td valign="top" align="center">0.6384</td>
<td valign="top" align="center">0.7740</td>
<td valign="top" align="center">0.6997</td>
<td valign="top" align="center">0.5111</td>
</tr> <tr>
<td valign="top" align="left">ResAttNet-10 (<xref ref-type="bibr" rid="B26">26</xref>)</td>
<td valign="top" align="center">0.5667</td>
<td valign="top" align="center">0.3167</td>
<td valign="top" align="center">0.2879</td>
<td valign="top" align="center">0.3016</td>
<td valign="top" align="center"><bold>0.8816</bold></td>
<td valign="top" align="center">0.6036</td>
<td valign="top" align="center">0.7614</td>
<td valign="top" align="center">0.6734</td>
<td valign="top" align="center">0.4919</td>
</tr> <tr>
<td valign="top" align="left">ResAttNet-18 (<xref ref-type="bibr" rid="B26">26</xref>)</td>
<td valign="top" align="center">0.5886</td>
<td valign="top" align="center">0.2833</td>
<td valign="top" align="center">0.2537</td>
<td valign="top" align="center">0.2677</td>
<td valign="top" align="center">0.8741</td>
<td valign="top" align="center">0.6366</td>
<td valign="top" align="center">0.7823</td>
<td valign="top" align="center">0.7020</td>
<td valign="top" align="center">0.5242</td>
</tr> <tr>
<td valign="top" align="left">ViT-for-AD (<xref ref-type="bibr" rid="B43">43</xref>)</td>
<td valign="top" align="center">0.5882</td>
<td valign="top" align="center">0.4000</td>
<td valign="top" align="center">0.4444</td>
<td valign="top" align="center">0.4211</td>
<td valign="top" align="center">0.8750</td>
<td valign="top" align="center">0.6364</td>
<td valign="top" align="center">0.7000</td>
<td valign="top" align="center">0.6667</td>
<td valign="top" align="center">0.4286</td>
</tr> <tr>
<td valign="top" align="left">MCNEL (<xref ref-type="bibr" rid="B44">44</xref>)</td>
<td valign="top" align="center">0.6105</td>
<td valign="top" align="center">0.5800</td>
<td valign="top" align="center">0.4500</td>
<td valign="top" align="center">0.5100</td>
<td valign="top" align="center">0.8600</td>
<td valign="top" align="center">0.6400</td>
<td valign="top" align="center">0.7300</td>
<td valign="top" align="center">0.6850</td>
<td valign="top" align="center">0.5800</td>
</tr> <tr>
<td valign="top" align="left"><bold>PlgFormer (ours)</bold></td>
<td valign="top" align="center"><bold>0.6228</bold></td>
<td valign="top" align="center"><bold>0.7619</bold></td>
<td valign="top" align="center"><bold>0.5161</bold></td>
<td valign="top" align="center"><bold>0.6154</bold></td>
<td valign="top" align="center">0.7273</td>
<td valign="top" align="center">0.3871</td>
<td valign="top" align="center">0.6545</td>
<td valign="top" align="center">0.4865</td>
<td valign="top" align="center"><bold>0.8593</bold></td>
</tr> <tr style="background-color:#dee1e1;">
<td valign="top" align="left" colspan="10"><bold>XWNI dataset</bold></td>
</tr> <tr>
<td valign="top" align="left">3D-ResNet-10 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.8047</td>
<td valign="top" align="center">0.7347</td>
<td valign="top" align="center">0.8372</td>
<td valign="top" align="center">0.7826</td>
<td valign="top" align="center">0.9114</td>
<td valign="top" align="center">0.4706</td>
<td valign="top" align="center">0.6154</td>
<td valign="top" align="center">0.5333</td>
<td valign="top" align="center">0.9550</td>
</tr> <tr>
<td valign="top" align="left">3D-ResNet-18 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.8125</td>
<td valign="top" align="center">0.8367</td>
<td valign="top" align="center">0.8913</td>
<td valign="top" align="center">0.8632</td>
<td valign="top" align="center">0.9367</td>
<td valign="top" align="center"><bold>0.6471</bold></td>
<td valign="top" align="center">0.4231</td>
<td valign="top" align="center">0.5116</td>
<td valign="top" align="center">0.8649</td>
</tr> <tr>
<td valign="top" align="left">3D-ResNet-34 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">0.8203</td>
<td valign="top" align="center">0.7143</td>
<td valign="top" align="center">0.8750</td>
<td valign="top" align="center">0.7865</td>
<td valign="top" align="center">0.9367</td>
<td valign="top" align="center">0.5882</td>
<td valign="top" align="center">0.7143</td>
<td valign="top" align="center">0.6452</td>
<td valign="top" align="center">0.9640</td>
</tr> <tr>
<td valign="top" align="left">DA-MIDL (<xref ref-type="bibr" rid="B24">24</xref>)</td>
<td valign="top" align="center">0.8438</td>
<td valign="top" align="center">0.7755</td>
<td valign="top" align="center">0.8636</td>
<td valign="top" align="center">0.8172</td>
<td valign="top" align="center">0.9241</td>
<td valign="top" align="center">0.5882</td>
<td valign="top" align="center">0.7143</td>
<td valign="top" align="center">0.6452</td>
<td valign="top" align="center">0.9640</td>
</tr> <tr>
<td valign="top" align="left">AMSNet (<xref ref-type="bibr" rid="B25">25</xref>)</td>
<td valign="top" align="center">0.8281</td>
<td valign="top" align="center">0.7551</td>
<td valign="top" align="center">0.8605</td>
<td valign="top" align="center">0.8043</td>
<td valign="top" align="center">0.9241</td>
<td valign="top" align="center">0.5294</td>
<td valign="top" align="center">0.6000</td>
<td valign="top" align="center">0.5625</td>
<td valign="top" align="center">0.9459</td>
</tr> <tr>
<td valign="top" align="left">ResAttNet-10 (<xref ref-type="bibr" rid="B26">26</xref>)</td>
<td valign="top" align="center">0.8047</td>
<td valign="top" align="center">0.7755</td>
<td valign="top" align="center">0.8261</td>
<td valign="top" align="center">0.8000</td>
<td valign="top" align="center">0.8987</td>
<td valign="top" align="center">0.5882</td>
<td valign="top" align="center">0.6250</td>
<td valign="top" align="center">0.6061</td>
<td valign="top" align="center">0.9459</td>
</tr> <tr>
<td valign="top" align="left">ResAttNet-18 (<xref ref-type="bibr" rid="B26">26</xref>)</td>
<td valign="top" align="center">0.8281</td>
<td valign="top" align="center">0.7551</td>
<td valign="top" align="center">0.8605</td>
<td valign="top" align="center">0.8043</td>
<td valign="top" align="center">0.9241</td>
<td valign="top" align="center">0.5294</td>
<td valign="top" align="center">0.6923</td>
<td valign="top" align="center">0.6000</td>
<td valign="top" align="center">0.9640</td>
</tr> <tr>
<td valign="top" align="left">ViT-for-AD (<xref ref-type="bibr" rid="B43">43</xref>)</td>
<td valign="top" align="center">0.8359</td>
<td valign="top" align="center">0.7857</td>
<td valign="top" align="center">0.8696</td>
<td valign="top" align="center">0.8020</td>
<td valign="top" align="center">0.9300</td>
<td valign="top" align="center">0.5588</td>
<td valign="top" align="center">0.6667</td>
<td valign="top" align="center">0.6061</td>
<td valign="top" align="center">0.9550</td>
</tr> <tr>
<td valign="top" align="left">MCNEL (<xref ref-type="bibr" rid="B44">44</xref>)</td>
<td valign="top" align="center">0.8500</td>
<td valign="top" align="center">0.8250</td>
<td valign="top" align="center">0.8800</td>
<td valign="top" align="center">0.7700</td>
<td valign="top" align="center">0.9350</td>
<td valign="top" align="center">0.6200</td>
<td valign="top" align="center">0.8000</td>
<td valign="top" align="center"><bold>0.6850</bold></td>
<td valign="top" align="center">0.9700</td>
</tr> <tr>
<td valign="top" align="left"><bold>PlgFormer (ours)</bold></td>
<td valign="top" align="center"><bold>0.8672</bold></td>
<td valign="top" align="center"><bold>0.8571</bold></td>
<td valign="top" align="center"><bold>0.9130</bold></td>
<td valign="top" align="center"><bold>0.8842</bold></td>
<td valign="top" align="center"><bold>0.9494</bold></td>
<td valign="top" align="center">0.5833</td>
<td valign="top" align="center"><bold>1.0000</bold></td>
<td valign="top" align="center">0.5833</td>
<td valign="top" align="center"><bold>1.0000</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values indicate the best value in the current column.</p>
</table-wrap-foot>
</table-wrap></sec>
<sec>
<title>4.4 Ablation studies</title>
<p>In this subsection, we conducted ablation studies on AD and CN binary classification on ADNI dataset to evaluate the impact of various key components in PlgFormer on model representation capacity. Similarly, we selected ACC, REC, PRE, F1, SPE, and AUC as evaluation metrics. The corresponding results are presented in <xref ref-type="table" rid="T7">Table 7</xref>, <italic>L</italic> and <italic>G</italic> respectively represent MHSA<sub><italic>l</italic></sub> and MHSA<sub><italic>g</italic></sub>, &#x02713;denotes leaving the corresponding component in place, whereas &#x000D7; denotes replacing it with another multi-head self-attention module.</p>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>Ablation studies on individual components of the proposed PlgFormer.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold>F2M</bold></th>
<th valign="top" align="center"><bold>DEB</bold></th>
<th valign="top" align="center"><bold><italic>L</italic></bold></th>
<th valign="top" align="center"><bold><italic>G</italic></bold></th>
<th valign="top" align="center"><bold>ACC</bold></th>
<th valign="top" align="center"><bold>REC</bold></th>
<th valign="top" align="center"><bold>PRE</bold></th>
<th valign="top" align="center"><bold>F1</bold></th>
<th valign="top" align="center"><bold>SPE</bold></th>
<th valign="top" align="center"><bold>AUC</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Med3D-ResNet-34 (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">0.8984</td>
<td valign="top" align="center">0.9048</td>
<td valign="top" align="center">0.7500</td>
<td valign="top" align="center">0.8201</td>
<td valign="top" align="center">0.8962</td>
<td valign="top" align="center">0.9005</td>
</tr> <tr>
<td valign="top" align="left">DA-MIDL (<xref ref-type="bibr" rid="B24">24</xref>)</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">0.9106</td>
<td valign="top" align="center">0.7619</td>
<td valign="top" align="center">0.8727</td>
<td valign="top" align="center">0.8136</td>
<td valign="top" align="center">0.9617</td>
<td valign="top" align="center">0.8618</td>
</tr> <tr>
<td valign="top" align="left">ResAttNet-10 (<xref ref-type="bibr" rid="B26">26</xref>)</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">0.9065</td>
<td valign="top" align="center">0.8730</td>
<td valign="top" align="center">0.7857</td>
<td valign="top" align="center">0.8271</td>
<td valign="top" align="center">0.9180</td>
<td valign="top" align="center">0.8955</td>
</tr> <tr>
<td valign="top" align="left">PlgFormer</td>
<td valign="top" align="center">&#x000D7;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">0.8821</td>
<td valign="top" align="center">0.8571</td>
<td valign="top" align="center">0.7297</td>
<td valign="top" align="center">0.7883</td>
<td valign="top" align="center">0.8907</td>
<td valign="top" align="center">0.8739</td>
</tr> <tr>
<td valign="top" align="left">PlgFormer</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x000D7;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">0.9146</td>
<td valign="top" align="center">0.7778</td>
<td valign="top" align="center">0.8750</td>
<td valign="top" align="center">0.8235</td>
<td valign="top" align="center">0.9617</td>
<td valign="top" align="center">0.8698</td>
</tr> <tr>
<td valign="top" align="left">PlgFormer</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x000D7;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">0.7480</td>
<td valign="top" align="center"><bold>1.0000</bold></td>
<td valign="top" align="center">0.5040</td>
<td valign="top" align="center">0.6702</td>
<td valign="top" align="center">0.6612</td>
<td valign="top" align="center">0.8306</td>
</tr> <tr>
<td valign="top" align="left">PlgFormer</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x000D7;</td>
<td valign="top" align="center">0.8984</td>
<td valign="top" align="center">0.8413</td>
<td valign="top" align="center">0.7794</td>
<td valign="top" align="center">0.8092</td>
<td valign="top" align="center">0.9180</td>
<td valign="top" align="center">0.8797</td>
</tr> <tr>
<td valign="top" align="left">PlgFormer</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center"><bold>0.9431</bold></td>
<td valign="top" align="center">0.8730</td>
<td valign="top" align="center"><bold>0.9016</bold></td>
<td valign="top" align="center"><bold>0.8871</bold></td>
<td valign="top" align="center"><bold>0.9672</bold></td>
<td valign="top" align="center"><bold>0.9201</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>&#x02713;Denotes leaving the corresponding component in place whereas &#x000D7; denotes removing it. All experiments are conducted on ADNI dataset to distinguish AD and CN subjects. The bold values indicate the best value in the current column.</p>
</table-wrap-foot>
</table-wrap>
<p>According to the results presented in <xref ref-type="table" rid="T7">Table 7</xref>, it is evident that convolutional operations play a crucial role in the PlgFormer model we designed. When MHSA<sub><italic>l</italic></sub> was replaced with MHSA<sub><italic>g</italic></sub>, the performance of the model dropped significantly (ACC decreased from 0.9431 to 0.7480). We attribute this phenomenon to the small sample size of our dataset, as pure self-attention modules require a large amount of data to demonstrate their effectiveness. We also observed a slight drop in performance when MHSA<sub><italic>g</italic></sub> was replaced with a local MHSA<sub><italic>l</italic></sub>, suggesting that global features are still necessary for AD classification using sMRI. This conclusion was supported by the fact that attention modules were embedded in the DA-MIDL and ResAttNet-10 architectures. Furthermore, when we removed the designed DEB, the performance slightly decreased, indicating that encoding image patch sequences dynamically is a meaningful operation. In addition, F2M is also an important and effective module for feature fusion, as replacing it with a simple concatenation caused a decrease in all evaluation metrics.</p>
<p>We evaluated the effects of dynamic convolution on overfitting and generalization, as shown in <xref ref-type="table" rid="T8">Table 8</xref>. The results indicate that when dynamic convolution is used, the model demonstrates similar training errors but reduces validation errors, highlighting its capability to mitigate overfitting. Furthermore, dynamic convolution leads to lower errors in the testing set of the XWNI dataset, suggesting improved generalization on other datasets. It is worth noting that we conducted experiments across 5 independent runs with different random seeds and reported the mean and standard deviation of the final validation loss, in order to evaluate the training stability of DEB. The results show that the DEB-enhanced model achieves a lower average loss with reduced variance, demonstrating improved robustness and stability.</p>
<table-wrap position="float" id="T8">
<label>Table 8</label>
<caption><p>Ablation studies of DEB.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold>DEB</bold></th>
<th valign="top" align="center"><bold>loss_train</bold></th>
<th valign="top" align="center"><bold>loss_val</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">PlgFormer</td>
<td valign="top" align="center">&#x000D7;</td>
<td valign="top" align="center">0.0573 &#x000B1; 0.0137</td>
<td valign="top" align="center">0.7807 &#x000B1; 0.0894</td>
</tr> <tr>
<td valign="top" align="left">PlgFormer</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">0.0611 &#x000B1; 0.0088</td>
<td valign="top" align="center">0.4760 &#x000B1; 0.0836</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>All experiments are conducted on ADNI dataset to distinguish AD and CN subjects. loss_train is the loss value in the training data, loss_val represents the loss value in the validation data.</p>
</table-wrap-foot>
</table-wrap></sec>
<sec>
<title>4.5 Visualization</title>
<p>To provide human physicians with reliable and accurate computer-aided diagnostic results, we employed Grad-CAM (<xref ref-type="bibr" rid="B45">45</xref>) to generate sMRI slice heat maps in sagittal, axial, and coronal planes, as illustrated in <xref ref-type="fig" rid="F7">Figure 7</xref>. To produce high-resolution heat maps that are easy to interpret, we applied 3D Grad-CAM at a lower layer with a resolution of 36  &#x000D7;  44  &#x000D7;  36 (<italic>D</italic> &#x000D7; <italic>H</italic> &#x000D7; <italic>W</italic>). In the process of visualization, the reshaped tensor of dimensions 36  &#x000D7;  44  &#x000D7;  36 was restored to its original resolution to facilitate a more detailed examination of features attended to by the neural network. The Grad-CAM procedure commenced with the loading of a pre-trained model alongside the original sMRI input. Subsequently, an intermediate layer&#x00027;s output and its gradients with respect to the output were selectively chosen. The visual representation of the feature maps derived from this output was then obtained. Following this, the impact of the gradients on the output of the target layer was quantified, yielding weight factors indicative of the significance of the feature maps. These weights were employed to project the importance of the feature maps back onto the input image. Finally, the application of these weights to the target layer&#x00027;s output resulted in the generation of a heatmap. This heatmap provides insights into which regions of the input image hold critical information for the model&#x00027;s predictive capacity.</p>
<fig position="float" id="F7">
<label>Figure 7</label>
<caption><p>Salient maps of sagittal, axial and coronal sMRI slices generated by Grad-CAM. The visualization results allow to observe the features of network interest.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fneur-16-1626922-g0007.tif">
<alt-text>Medical image grid showing brain scans from different views: sagittal, axial, and coronal. Each view includes an MRI, Weight CAM, and overlay image. The overlay images exhibit varying intensity, with a color scale indicating values from 0.00 to 0.40. </alt-text>
</graphic>
</fig>
<p>We have drawn GradCAM using six patients with AD, as illustrated in <xref ref-type="fig" rid="F7">Figure 7</xref>, all of whom are from the test set of the ADNI dataset. Additionally, We have analyzed the visualization results for these patients using the ICBM 152 template. First, we aligned the sMRI data of AD patients with the template, and then analyzed the brain regions corresponding to the areas of interest (high-luminance areas) of our model. After analysis, we found that the brain regions that our model focuses on include: Posterior Cingulate, Precuneus, Isthmus Cingulate, and Lateral Ventricle. We have consulted with doctors who diagnose AD clinically, and found that the Posterior Cingulate and Precuneus regions are consistent with the areas that doctors focus on during clinical diagnosis. This further confirms that the features extracted by our method are not only meaningful for deep neural network models but also provide credible diagnoses of AD via sMRI for human physicians (<xref ref-type="bibr" rid="B46">46</xref>).</p>
<p>Moreover, the visualization results obtained using Grad-CAM empower physicians to better understand and interpret the classification decision of our model. By providing a visual reference, physicians can easily validate the reasoning behind a diagnosis and identify any potential shortcomings or biases in the model. This serves as a valuable tool for improving the interpretability and transparency of Computer-aided diagnosis (CAD), and ultimately helps build trust between physicians and machine learning models.</p></sec></sec>
<sec sec-type="conclusions" id="s5">
<title>5 Conclusions</title>
<p>Using sMRI for computer-aided diagnosis is significant for early detection and timely intervention of AD. In this paper, we propose PlgFormer, a unified and parallel approach that combines CNNs and pure self-attention mechanisms to extract local-global context features in sMRI with discriminative value for AD diagnosis. Our designed DEB introduces dynamic convolutions that adaptively adjust the kernel size based on the input size, while our designed F2M adaptively fuses the extracted local and global features through a gating mechanism. On publicly available ADNI and privately held XWNI datasets, our PlgFormer achieved state-of-the-art performance compared to existing methods in AD vs. CN binary classification, MCI vs. CN binary classification, and AD vs. MCI vs. CN triple classification tasks. Saliency maps generated by Grad-CAM confirmed that our proposed method can help human experts identify lesions quickly in sMRI. Further studies could investigate the potential of PlgFormer on other datasets and explore its application in other areas of medical image analysis. Our research anticipates practical clinical applications.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The datasets presented in this article are not readily available because the XWNI dataset used in this study is not publicly available due to patient privacy concerns and institutional data sharing restrictions. Access to the dataset is limited by the policies of Xuanwu Hospital Capital Medical University and requires specific institutional approvals. Requests to access the datasets should be directed to Zhixiong Li, <email>865818683&#x00040;qq.com</email>.</p>
</sec>
<sec sec-type="ethics-statement" id="s7">
<title>Ethics statement</title>
<p>The studies involving humans were approved by Ethics Committee of Xuanwu Hospital Capital Medical University. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>GW: Conceptualization, Formal analysis, Investigation, Methodology, Validation, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. YL: Data curation, Investigation, Writing &#x02013; review &#x00026; editing. ZZ: Data curation, Formal analysis, Software, Writing &#x02013; original draft. SA: Project administration, Validation, Writing &#x02013; review &#x00026; editing. XC: Formal analysis, Methodology, Writing &#x02013; review &#x00026; editing. YJ: Writing &#x02013; review &#x00026; editing. ZS: Resources, Writing &#x02013; review &#x00026; editing. GC: Resources, Writing &#x02013; review &#x00026; editing. MZ: Resources, Writing &#x02013; review &#x00026; editing. ZL: Resources, Writing &#x02013; review &#x00026; editing. FY: Project administration, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. The authors gratefully acknowledge the support of the National Key Research and Development Program of China (Grant No. 2023YFC3603601), the National Natural Science Foundation of China (Grant No. 82001773), the Medical Science Research Project of Hebei Province (20221842), and the Construction Project of Academician Cooperation Key Unit of Hebei Province.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>XC and YJ were employed by JD Health International Inc. The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p></sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p></sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn0001"><p><sup>1</sup>This study was performed in line with the principles of the Declaration of Helsinki. Ethical Approval was granted by the Ethics Committee of Xuanwu Hospital Capital Medical University (No. 2017046). All subjects signed informed consent forms.</p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Association</surname> <given-names>A</given-names></name></person-group>. <article-title>2023 Alzheimer&#x00027;s disease facts and figures</article-title>. <source>Alzheimers Dement</source>. (<year>2023</year>) <volume>19</volume>:<fpage>1598</fpage>&#x02013;<lpage>695</lpage>. <pub-id pub-id-type="doi">10.1002/alz.13016</pub-id><pub-id pub-id-type="pmid">36918389</pub-id></citation></ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Breijyeh</surname> <given-names>Z</given-names></name> <name><surname>Karaman</surname> <given-names>R</given-names></name></person-group>. <article-title>Comprehensive review on Alzheimer&#x00027;s disease: causes and treatment</article-title>. <source>Molecules</source>. (<year>2020</year>) <volume>25</volume>:<fpage>5789</fpage>. <pub-id pub-id-type="doi">10.3390/molecules25245789</pub-id><pub-id pub-id-type="pmid">33302541</pub-id></citation></ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Weller</surname> <given-names>J</given-names></name> <name><surname>Budson</surname> <given-names>A</given-names></name></person-group>. <article-title>Current understanding of Alzheimer&#x00027;s disease diagnosis and treatment</article-title>. <source>F1000Research</source>. (<year>2018</year>) 7:F1000-Faculty. <pub-id pub-id-type="doi">10.12688/f1000research.14506.1</pub-id><pub-id pub-id-type="pmid">30135715</pub-id></citation></ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Frisoni</surname> <given-names>GB</given-names></name> <name><surname>Fox</surname> <given-names>NC</given-names></name> <name><surname>Jack Jr</surname> <given-names>CR</given-names></name> <name><surname>Scheltens</surname> <given-names>P</given-names></name> <name><surname>Thompson</surname> <given-names>PM</given-names></name></person-group>. <article-title>The clinical use of structural MRI in Alzheimer disease</article-title>. <source>Nat Rev Neurol</source>. (<year>2010</year>) <volume>6</volume>:<fpage>67</fpage>&#x02013;<lpage>77</lpage>. <pub-id pub-id-type="doi">10.1038/nrneurol.2009.215</pub-id><pub-id pub-id-type="pmid">20139996</pub-id></citation></ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jack</surname> <given-names>Jr PRCXYCea C R</given-names></name></person-group>. <article-title>Prediction of AD with MRI-based hippocampal volume in mild cognitive impairment</article-title>. <source>Neurology</source>. (<year>1999</year>) <volume>52</volume>:<fpage>1397</fpage>. <pub-id pub-id-type="doi">10.1212/WNL.52.7.1397</pub-id><pub-id pub-id-type="pmid">10227624</pub-id></citation></ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Convit</surname> <given-names>A</given-names></name> <name><surname>De Leon</surname> <given-names>M</given-names></name> <name><surname>Tarshish</surname> <given-names>C</given-names></name> <name><surname>De Santi</surname> <given-names>S</given-names></name> <name><surname>Tsui</surname> <given-names>W</given-names></name> <name><surname>Rusinek</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>Specific hippocampal volume reductions in individuals at risk for Alzheimer&#x00027;s disease</article-title>. <source>Neurobiol Aging</source>. (<year>1997</year>) <volume>18</volume>:<fpage>131</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1016/S0197-4580(97)00001-8</pub-id><pub-id pub-id-type="pmid">9258889</pub-id></citation></ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Apostolova</surname> <given-names>LG</given-names></name> <name><surname>Dinov</surname> <given-names>ID</given-names></name> <name><surname>Dutton</surname> <given-names>RA</given-names></name> <name><surname>Hayashi</surname> <given-names>KM</given-names></name> <name><surname>Toga</surname> <given-names>AW</given-names></name> <name><surname>Cummings</surname> <given-names>JL</given-names></name> <etal/></person-group>. <article-title>3D comparison of hippocampal atrophy in amnestic mild cognitive impairment and Alzheimer&#x00027;s disease</article-title>. <source>Brain</source>. (<year>2006</year>) <volume>129</volume>:<fpage>2867</fpage>&#x02013;<lpage>73</lpage>. <pub-id pub-id-type="doi">10.1093/brain/awl274</pub-id><pub-id pub-id-type="pmid">17018552</pub-id></citation></ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hinrichs</surname> <given-names>C</given-names></name> <name><surname>Singh</surname> <given-names>V</given-names></name> <name><surname>Mukherjee</surname> <given-names>L</given-names></name> <name><surname>Xu</surname> <given-names>G</given-names></name> <name><surname>Chung</surname> <given-names>MK</given-names></name> <name><surname>Johnson</surname> <given-names>SC</given-names></name> <etal/></person-group>. <article-title>Spatially augmented LPboosting for AD classification with evaluations on the ADNI dataset</article-title>. <source>Neuroimage</source>. (<year>2009</year>) <volume>48</volume>:<fpage>138</fpage>&#x02013;<lpage>49</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2009.05.056</pub-id><pub-id pub-id-type="pmid">19481161</pub-id></citation></ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kao</surname> <given-names>YH</given-names></name> <name><surname>Chou</surname> <given-names>MC</given-names></name> <name><surname>Chen</surname> <given-names>CH</given-names></name> <name><surname>Yang</surname> <given-names>YH</given-names></name></person-group>. <article-title>White matter changes in patients with Alzheimer&#x00027;s disease and associated factors</article-title>. <source>J Clin Med</source>. (<year>2019</year>) <volume>8</volume>:<fpage>167</fpage>. <pub-id pub-id-type="doi">10.3390/jcm8020167</pub-id><pub-id pub-id-type="pmid">30717182</pub-id></citation></ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vounou</surname> <given-names>M</given-names></name> <name><surname>Janousova</surname> <given-names>E</given-names></name> <name><surname>Wolz</surname> <given-names>R</given-names></name> <name><surname>Stein</surname> <given-names>JL</given-names></name> <name><surname>Thompson</surname> <given-names>PM</given-names></name> <name><surname>Rueckert</surname> <given-names>D</given-names></name> <etal/></person-group>. <article-title>Sparse reduced-rank regression detects genetic associations with voxel-wise longitudinal phenotypes in Alzheimer&#x00027;s disease</article-title>. <source>Neuroimage</source>. (<year>2012</year>) <volume>60</volume>:<fpage>700</fpage>&#x02013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2011.12.029</pub-id><pub-id pub-id-type="pmid">22209813</pub-id></citation></ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>D</given-names></name> <name><surname>Wang</surname> <given-names>Y</given-names></name> <name><surname>Zhou</surname> <given-names>L</given-names></name> <name><surname>Yuan</surname> <given-names>H</given-names></name> <name><surname>Shen</surname> <given-names>D</given-names></name> <name><surname>Initiative</surname> <given-names>ADN</given-names></name> <etal/></person-group>. <article-title>Multimodal classification of Alzheimer&#x00027;s disease and mild cognitive impairment</article-title>. <source>Neuroimage</source>. (<year>2011</year>) <volume>55</volume>:<fpage>856</fpage>&#x02013;<lpage>67</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2011.01.008</pub-id><pub-id pub-id-type="pmid">21236349</pub-id></citation></ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>M</given-names></name> <name><surname>Zhang</surname> <given-names>D</given-names></name> <name><surname>Shen</surname> <given-names>D</given-names></name></person-group>. <article-title>Relationship induced multi-template learning for diagnosis of Alzheimer&#x00027;s disease and mild cognitive impairment</article-title>. <source>IEEE Trans Med Imaging</source>. (<year>2016</year>) <volume>35</volume>:<fpage>1463</fpage>&#x02013;<lpage>74</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2016.2515021</pub-id><pub-id pub-id-type="pmid">26742127</pub-id></citation></ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qiu</surname> <given-names>S</given-names></name> <name><surname>Joshi</surname> <given-names>PS</given-names></name> <name><surname>Miller</surname> <given-names>MI</given-names></name> <name><surname>Xue</surname> <given-names>C</given-names></name> <name><surname>Zhou</surname> <given-names>X</given-names></name> <name><surname>Karjadi</surname> <given-names>C</given-names></name> <etal/></person-group>. <article-title>Development and validation of an interpretable deep learning framework for Alzheimer&#x00027;s disease classification</article-title>. <source>Brain</source>. (<year>2020</year>) <volume>143</volume>:<fpage>1920</fpage>&#x02013;<lpage>33</lpage>. <pub-id pub-id-type="doi">10.1093/brain/awaa137</pub-id><pub-id pub-id-type="pmid">32357201</pub-id></citation></ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>X</given-names></name> <name><surname>Han</surname> <given-names>L</given-names></name> <name><surname>Han</surname> <given-names>L</given-names></name> <name><surname>Chen</surname> <given-names>H</given-names></name> <name><surname>Dancey</surname> <given-names>D</given-names></name> <name><surname>Zhang</surname> <given-names>D</given-names></name></person-group>. <article-title>MRI-PatchNet: a novel efficient explainable patch-based deep learning network for Alzheimer&#x00027;s disease diagnosis with Structural MRI</article-title>. <source>IEEE Access</source>. (<year>2023</year>) <volume>11</volume>:<fpage>108603</fpage>&#x02013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2023.3321220</pub-id></citation>
</ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ashburner</surname> <given-names>J</given-names></name> <name><surname>Friston</surname> <given-names>KJ</given-names></name></person-group>. <article-title>Voxel-based morphometry&#x02014;the methods</article-title>. <source>Neuroimage</source>. (<year>2000</year>) <volume>11</volume>:<fpage>805</fpage>&#x02013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.1006/nimg.2000.0582</pub-id><pub-id pub-id-type="pmid">10860804</pub-id></citation></ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kl&#x000F6;ppel</surname> <given-names>S</given-names></name> <name><surname>Stonnington</surname> <given-names>CM</given-names></name> <name><surname>Chu</surname> <given-names>C</given-names></name> <name><surname>Draganski</surname> <given-names>B</given-names></name> <name><surname>Scahill</surname> <given-names>RI</given-names></name> <name><surname>Rohrer</surname> <given-names>JD</given-names></name> <etal/></person-group>. <article-title>Automatic classification of MR scans in Alzheimer&#x00027;s disease</article-title>. Brain. (<year>2008</year>) <volume>131</volume>:<fpage>681</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1093/brain/awm319</pub-id><pub-id pub-id-type="pmid">18202106</pub-id></citation></ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fan</surname> <given-names>Y</given-names></name> <name><surname>Shen</surname> <given-names>D</given-names></name> <name><surname>Gur</surname> <given-names>RC</given-names></name> <name><surname>Gur</surname> <given-names>RE</given-names></name> <name><surname>Davatzikos</surname> <given-names>C</given-names></name></person-group>. <article-title>COMPARE: classification of morphological patterns using adaptive regional elements</article-title>. <source>IEEE Trans Med Imaging</source>. (<year>2006</year>) <volume>26</volume>:<fpage>93</fpage>&#x02013;<lpage>105</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2006.886812</pub-id><pub-id pub-id-type="pmid">17243588</pub-id></citation></ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cao</surname> <given-names>P</given-names></name> <name><surname>Liu</surname> <given-names>X</given-names></name> <name><surname>Yang</surname> <given-names>J</given-names></name> <name><surname>Zhao</surname> <given-names>D</given-names></name> <name><surname>Huang</surname> <given-names>M</given-names></name> <name><surname>Zhang</surname> <given-names>J</given-names></name> <etal/></person-group>. <article-title>Nonlinearity-aware based dimensionality reduction and over-sampling for AD/MCI classification from MRI measures</article-title>. <source>Comput Biol Med</source>. (<year>2017</year>) <volume>91</volume>:<fpage>21</fpage>&#x02013;<lpage>37</lpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2017.10.002</pub-id><pub-id pub-id-type="pmid">29031664</pub-id></citation></ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abuhmed</surname> <given-names>T</given-names></name> <name><surname>El-Sappagh</surname> <given-names>S</given-names></name> <name><surname>Alonso</surname> <given-names>JM</given-names></name></person-group>. <article-title>Robust hybrid deep learning models for Alzheimer&#x00027;s progression detection</article-title>. <source>Knowl Based Syst</source>. (<year>2021</year>) <volume>213</volume>:<fpage>106688</fpage>. <pub-id pub-id-type="doi">10.1016/j.knosys.2020.106688</pub-id></citation>
</ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krizhevsky</surname> <given-names>A</given-names></name> <name><surname>Sutskever</surname> <given-names>I</given-names></name> <name><surname>Hinton</surname> <given-names>GE</given-names></name></person-group>. <article-title>ImageNet classification with deep convolutional neural networks</article-title>. <source>Commun ACM</source>. (<year>2017</year>) <volume>60</volume>:<fpage>84</fpage>&#x02013;<lpage>90</lpage>. <pub-id pub-id-type="doi">10.1145/3065386</pub-id></citation>
</ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Simonyan</surname> <given-names>K</given-names></name> <name><surname>Zisserman</surname> <given-names>A</given-names></name></person-group>. <article-title>Very deep convolutional networks for large-scale image recognition</article-title>. <source>arXiv preprint arXiv:14091556.</source> (<year>2014</year>). <pub-id pub-id-type="doi">10.48550/arXiv.1409.1556</pub-id></citation>
</ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K</given-names></name> <name><surname>Zhang</surname> <given-names>X</given-names></name> <name><surname>Ren</surname> <given-names>S</given-names></name> <name><surname>Sun</surname> <given-names>J</given-names></name></person-group>. <article-title>Deep residual learning for image recognition</article-title>. In: <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>. <publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE</publisher-name> (<year>2016</year>). p. <fpage>770</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id></citation>
</ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lian</surname> <given-names>C</given-names></name> <name><surname>Liu</surname> <given-names>M</given-names></name> <name><surname>Zhang</surname> <given-names>J</given-names></name> <name><surname>Shen</surname> <given-names>D</given-names></name></person-group>. <article-title>Hierarchical fully convolutional network for joint atrophy localization and Alzheimer&#x00027;s disease diagnosis using structural MRI</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. (<year>2018</year>) <volume>42</volume>:<fpage>880</fpage>&#x02013;<lpage>93</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2018.2889096</pub-id><pub-id pub-id-type="pmid">30582529</pub-id></citation></ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>W</given-names></name> <name><surname>Sun</surname> <given-names>L</given-names></name> <name><surname>Huang</surname> <given-names>J</given-names></name> <name><surname>Han</surname> <given-names>L</given-names></name> <name><surname>Zhang</surname> <given-names>D</given-names></name></person-group>. <article-title>Dual attention multi-instance deep learning for Alzheimer&#x00027;s disease diagnosis with structural MRI</article-title>. <source>IEEE Trans Med Imaging</source>. (<year>2021</year>) <volume>40</volume>:<fpage>2354</fpage>&#x02013;<lpage>66</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2021.3077079</pub-id><pub-id pub-id-type="pmid">33939609</pub-id></citation></ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>Y</given-names></name> <name><surname>Zhou</surname> <given-names>Y</given-names></name> <name><surname>Zeng</surname> <given-names>W</given-names></name> <name><surname>Qian</surname> <given-names>Q</given-names></name> <name><surname>Song</surname> <given-names>M</given-names></name></person-group>. <article-title>An attention-based 3D CNN with multi-scale integration block for Alzheimer&#x00027;s disease classification</article-title>. <source>IEEE J Biomed Health Inform</source>. (<year>2022</year>) <volume>26</volume>:<fpage>5665</fpage>&#x02013;<lpage>73</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2022.3197331</pub-id><pub-id pub-id-type="pmid">35939481</pub-id></citation></ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>X</given-names></name> <name><surname>Han</surname> <given-names>L</given-names></name> <name><surname>Zhu</surname> <given-names>W</given-names></name> <name><surname>Sun</surname> <given-names>L</given-names></name> <name><surname>Zhang</surname> <given-names>D</given-names></name></person-group>. <article-title>An explainable 3D residual self-attention deep neural network for joint atrophy localization and Alzheimer&#x00027;s disease diagnosis using structural MRI</article-title>. <source>IEEE J Biomed Health Inform</source>. (<year>2021</year>) <volume>26</volume>:<fpage>5289</fpage>&#x02013;<lpage>97</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2021.3066832</pub-id><pub-id pub-id-type="pmid">33735087</pub-id></citation></ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ashtari-Majlan</surname> <given-names>M</given-names></name> <name><surname>Seifi</surname> <given-names>A</given-names></name> <name><surname>Dehshibi</surname> <given-names>MM</given-names></name></person-group>. <article-title>A multi-stream convolutional neural network for classification of progressive MCI in Alzheimer&#x00027;s disease using structural MRI images</article-title>. <source>IEEE J Biomed Health Inform</source>. (<year>2022</year>) <volume>26</volume>:<fpage>3918</fpage>&#x02013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2022.3155705</pub-id><pub-id pub-id-type="pmid">35239494</pub-id></citation></ref>
<ref id="B28">
<label>28.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>J</given-names></name> <name><surname>He</surname> <given-names>X</given-names></name> <name><surname>Qing</surname> <given-names>L</given-names></name> <name><surname>Chen</surname> <given-names>X</given-names></name> <name><surname>Liu</surname> <given-names>Y</given-names></name> <name><surname>Chen</surname> <given-names>H</given-names></name></person-group>. <article-title>Multi-relation graph convolutional network for Alzheimer&#x00027;s disease diagnosis using structural MRI</article-title>. <source>Knowl Based Syst</source>. (<year>2023</year>) <volume>270</volume>:<fpage>110546</fpage>. <pub-id pub-id-type="doi">10.1016/j.knosys.2023.110546</pub-id></citation>
</ref>
<ref id="B29">
<label>29.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Billones</surname> <given-names>CD</given-names></name> <name><surname>Demetria</surname> <given-names>OJLD</given-names></name> <name><surname>Hostallero</surname> <given-names>DED</given-names></name> <name><surname>Naval</surname> <given-names>PC</given-names></name></person-group>. <article-title>DemNet: a convolutional neural network for the detection of Alzheimer&#x00027;s disease and mild cognitive impairment</article-title>. In: <source>2016 IEEE Region 10 Conference (TENCON)</source>. <publisher-loc>Singapore</publisher-loc>: <publisher-name>IEEE</publisher-name> (<year>2016</year>). p. <fpage>3724</fpage>&#x02013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1109/TENCON.2016.7848755</pub-id></citation>
</ref>
<ref id="B30">
<label>30.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H</given-names></name> <name><surname>Habes</surname> <given-names>M</given-names></name> <name><surname>Fan</surname> <given-names>Y</given-names></name></person-group>. <article-title>Deep ordinal ranking for multi-category diagnosis of Alzheimer&#x00027;s disease using hippocampal MRI data</article-title>. <source>arXiv preprint arXiv:170901599</source>. (<year>2017</year>). <pub-id pub-id-type="doi">10.48550/arXiv.1709.01599</pub-id></citation>
</ref>
<ref id="B31">
<label>31.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Z</given-names></name> <name><surname>Lu</surname> <given-names>H</given-names></name> <name><surname>Pan</surname> <given-names>X</given-names></name> <name><surname>Xu</surname> <given-names>M</given-names></name> <name><surname>Lan</surname> <given-names>R</given-names></name> <name><surname>Luo</surname> <given-names>X</given-names></name></person-group>. <article-title>Diagnosis of Alzheimer&#x00027;s disease via an attention-based multi-scale convolutional neural network</article-title>. <source>Knowl-Based Syst</source>. (<year>2022</year>) <volume>238</volume>:<fpage>107942</fpage>. <pub-id pub-id-type="doi">10.1016/j.knosys.2021.107942</pub-id></citation>
</ref>
<ref id="B32">
<label>32.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>J</given-names></name> <name><surname>Zheng</surname> <given-names>B</given-names></name> <name><surname>Gao</surname> <given-names>A</given-names></name> <name><surname>Feng</surname> <given-names>X</given-names></name> <name><surname>Liang</surname> <given-names>D</given-names></name> <name><surname>Long</surname> <given-names>X</given-names></name> <etal/></person-group>. <article-title>3D densely connected convolution neural network with connection-wise attention mechanism for Alzheimer&#x00027;s disease classification</article-title>. <source>Magn Reson Imaging</source>. (<year>2021</year>) <volume>78</volume>:<fpage>119</fpage>&#x02013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1016/j.mri.2021.02.001</pub-id><pub-id pub-id-type="pmid">33588019</pub-id></citation></ref>
<ref id="B33">
<label>33.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Jin</surname> <given-names>D</given-names></name> <name><surname>Xu</surname> <given-names>J</given-names></name> <name><surname>Zhao</surname> <given-names>K</given-names></name> <name><surname>Hu</surname> <given-names>F</given-names></name> <name><surname>Yang</surname> <given-names>Z</given-names></name> <name><surname>Liu</surname> <given-names>B</given-names></name> <etal/></person-group>. <article-title>Attention-based 3D convolutional network for Alzheimer&#x00027;s disease diagnosis and biomarkers exploration</article-title>. In: <source>2019 IEEE 16Th International Symposium on biomedical imaging (ISBI 2019)</source>. <publisher-loc>Venice</publisher-loc>: <publisher-name>IEEE</publisher-name> (<year>2019</year>). p. <fpage>1047</fpage>&#x02013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.1109/ISBI.2019.8759455</pub-id></citation>
</ref>
<ref id="B34">
<label>34.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaswani</surname> <given-names>A</given-names></name> <name><surname>Shazeer</surname> <given-names>N</given-names></name> <name><surname>Parmar</surname> <given-names>N</given-names></name> <name><surname>Uszkoreit</surname> <given-names>J</given-names></name> <name><surname>Jones</surname> <given-names>L</given-names></name> <name><surname>Gomez</surname> <given-names>AN</given-names></name> <etal/></person-group>. <article-title>Attention is all you need</article-title>. <source>Adv Neural Inf Process Syst</source>. (<year>2017</year>) <volume>30</volume>:<fpage>6000</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.5555/3295222.3295349</pub-id></citation>
</ref>
<ref id="B35">
<label>35.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dosovitskiy</surname> <given-names>A</given-names></name> <name><surname>Beyer</surname> <given-names>L</given-names></name> <name><surname>Kolesnikov</surname> <given-names>A</given-names></name> <name><surname>Weissenborn</surname> <given-names>D</given-names></name> <name><surname>Zhai</surname> <given-names>X</given-names></name> <name><surname>Unterthiner</surname> <given-names>T</given-names></name> <etal/></person-group>. <article-title>An image is worth 16 &#x000D7; 16 words: transformers for image recognition at scale</article-title>. <source>arXiv preprint arXiv:201011929</source>. (<year>2020</year>). <pub-id pub-id-type="doi">10.48550/arXiv.2010.11929</pub-id></citation>
</ref>
<ref id="B36">
<label>36.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Z</given-names></name> <name><surname>Lin</surname> <given-names>Y</given-names></name> <name><surname>Cao</surname> <given-names>Y</given-names></name> <name><surname>Hu</surname> <given-names>H</given-names></name> <name><surname>Wei</surname> <given-names>Y</given-names></name> <name><surname>Zhang</surname> <given-names>Z</given-names></name> <etal/></person-group>. <article-title>Swin transformer: hierarchical vision transformer using shifted windows</article-title>. In: <source>Proceedings of the IEEE/CVF International Conference on Computer Vision</source>. <publisher-loc>Montreal, QC</publisher-loc>: <publisher-name>IEEE</publisher-name> (<year>2021</year>). p. <fpage>10012</fpage>&#x02013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV48922.2021.00986</pub-id></citation>
</ref>
<ref id="B37">
<label>37.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Jang</surname> <given-names>J</given-names></name> <name><surname>Hwang</surname> <given-names>D</given-names></name></person-group>. <article-title>M3T: three-dimensional medical image classifier using multi-plane and multi-slice transformer</article-title>. In: <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>. <publisher-loc>New Orleans, LA</publisher-loc>: <publisher-name>IEEE</publisher-name> (<year>2022</year>). p. <fpage>20718</fpage>&#x02013;<lpage>29</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.02006</pub-id></citation>
</ref>
<ref id="B38">
<label>38.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chu</surname> <given-names>X</given-names></name> <name><surname>Zhang</surname> <given-names>B</given-names></name> <name><surname>Tian</surname> <given-names>Z</given-names></name> <name><surname>Wei</surname> <given-names>X</given-names></name> <name><surname>Xia</surname> <given-names>H</given-names></name></person-group>. <article-title>Do we really need explicit position encodings for vision transformers</article-title>. <source>arXiv preprint arXiv:210210882</source>. (<year>2021</year>). <pub-id pub-id-type="doi">10.48550/arXiv.2102.10882</pub-id></citation>
</ref>
<ref id="B39">
<label>39.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chu</surname> <given-names>X</given-names></name> <name><surname>Tian</surname> <given-names>Z</given-names></name> <name><surname>Wang</surname> <given-names>Y</given-names></name> <name><surname>Zhang</surname> <given-names>B</given-names></name> <name><surname>Ren</surname> <given-names>H</given-names></name> <name><surname>Wei</surname> <given-names>X</given-names></name> <etal/></person-group>. <article-title>Twins: revisiting the design of spatial attention in vision transformers</article-title>. <source>Adv Neural Inf Process Syst</source>. (<year>2021</year>) <volume>34</volume>:<fpage>9355</fpage>&#x02013;<lpage>66</lpage>. <pub-id pub-id-type="doi">10.5555/3540261.3540977</pub-id></citation>
</ref>
<ref id="B40">
<label>40.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Dong</surname> <given-names>X</given-names></name> <name><surname>Bao</surname> <given-names>J</given-names></name> <name><surname>Chen</surname> <given-names>D</given-names></name> <name><surname>Zhang</surname> <given-names>W</given-names></name> <name><surname>Yu</surname> <given-names>N</given-names></name> <name><surname>Yuan</surname> <given-names>L</given-names></name> <etal/></person-group>. <article-title>CSWin transformer: a general vision transformer backbone with cross-shaped windows</article-title>. In: <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>. <publisher-loc>New Orleans, LA</publisher-loc>: <publisher-name>IEEE</publisher-name> (<year>2022</year>). p. <fpage>12124</fpage>&#x02013;<lpage>34</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.01181</pub-id></citation>
</ref>
<ref id="B41">
<label>41.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jack Jr</surname> <given-names>CR</given-names></name> <name><surname>Bernstein</surname> <given-names>MA</given-names></name> <name><surname>Fox</surname> <given-names>NC</given-names></name> <name><surname>Thompson</surname> <given-names>P</given-names></name> <name><surname>Alexander</surname> <given-names>G</given-names></name> <name><surname>Harvey</surname> <given-names>D</given-names></name> <etal/></person-group>. <article-title>The Alzheimer&#x00027;s disease neuroimaging initiative (ADNI): MRI methods</article-title>. <source>J Magn Reson Imaging</source>. (<year>2008</year>) <volume>27</volume>:<fpage>685</fpage>&#x02013;<lpage>91</lpage>. <pub-id pub-id-type="doi">10.1002/jmri.21049</pub-id><pub-id pub-id-type="pmid">18302232</pub-id></citation></ref>
<ref id="B42">
<label>42.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>S</given-names></name> <name><surname>Ma</surname> <given-names>K</given-names></name> <name><surname>Zheng</surname> <given-names>Y</given-names></name></person-group>. <article-title>Med3D: transfer learning for 3D medical image analysis</article-title>. <source>arXiv preprint arXiv:190400625</source>. (<year>2019</year>). <pub-id pub-id-type="doi">10.48550/arXiv.1904.00625</pub-id></citation>
</ref>
<ref id="B43">
<label>43.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kunanbayev</surname> <given-names>K</given-names></name> <name><surname>Shen</surname> <given-names>V</given-names></name> <name><surname>Kim</surname> <given-names>DS</given-names></name></person-group>. <article-title>Training ViT with limited data for Alzheimer&#x00027;s disease classification: an empirical study</article-title>. In: <source>Proceedings of Medical Image Computing and Computer Assisted Intervention-MICCAI 2024. Vol. LNCS 15012</source>. Springer Nature Switzerland: New York (<year>2024</year>). <pub-id pub-id-type="doi">10.1007/978-3-031-72390-2_32</pub-id></citation>
</ref>
<ref id="B44">
<label>44.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yan</surname> <given-names>F</given-names></name> <name><surname>Peng</surname> <given-names>L</given-names></name> <name><surname>Dong</surname> <given-names>F</given-names></name> <name><surname>Hirota</surname> <given-names>K</given-names></name></person-group>. <article-title>MCNEL: a multi-scale convolutional network and ensemble learning for Alzheimer&#x00027;s disease diagnosis</article-title>. <source>Comput Methods Programs Biomed</source>. (<year>2025</year>) <volume>264</volume>:<fpage>108703</fpage>. <pub-id pub-id-type="doi">10.1016/j.cmpb.2025.108703</pub-id><pub-id pub-id-type="pmid">40081198</pub-id></citation></ref>
<ref id="B45">
<label>45.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Selvaraju</surname> <given-names>RR</given-names></name> <name><surname>Cogswell</surname> <given-names>M</given-names></name> <name><surname>Das</surname> <given-names>A</given-names></name> <name><surname>Vedantam</surname> <given-names>R</given-names></name> <name><surname>Parikh</surname> <given-names>D</given-names></name> <name><surname>Batra</surname> <given-names>D</given-names></name></person-group>. <article-title>Grad-cam: visual explanations from deep networks via gradient-based localization</article-title>. In: <source>Proceedings of the IEEE International Conference on Computer Vision</source>. <publisher-loc>Venice</publisher-loc>: <publisher-name>IEEE</publisher-name> (<year>2017</year>). p. <fpage>618</fpage>&#x02013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV.2017.74</pub-id></citation>
</ref>
<ref id="B46">
<label>46.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ott</surname> <given-names>BR</given-names></name> <name><surname>Cohen</surname> <given-names>RA</given-names></name> <name><surname>Gongvatana</surname> <given-names>A</given-names></name> <name><surname>Okonkwo</surname> <given-names>OC</given-names></name> <name><surname>Johanson</surname> <given-names>CE</given-names></name> <name><surname>Stopa</surname> <given-names>EG</given-names></name> <etal/></person-group>. <article-title>Brain ventricular volume and cerebrospinal fluid biomarkers of Alzheimer&#x00027;s disease</article-title>. <source>J Alzheimers Dis</source>. (<year>2010</year>) <volume>20</volume>:<fpage>647</fpage>&#x02013;<lpage>57</lpage>. <pub-id pub-id-type="doi">10.3233/JAD-2010-1406</pub-id><pub-id pub-id-type="pmid">20182051</pub-id></citation></ref>
</ref-list>
</back>
</article>