<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurosci.</journal-id>
<journal-title>Frontiers in Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-453X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnins.2025.1637291</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A lightweight triple-modal fusion network for progressive mild cognitive impairment prediction in Alzheimer&#x00027;s disease</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Shen</surname> <given-names>Xiangyu</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3053750/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Hu</surname> <given-names>Xiangyang</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Renfeng</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Fu</surname> <given-names>Yunzhan</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3168299/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Xu</surname> <given-names>Jiamin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Lyu</surname> <given-names>Degang</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Xie</surname> <given-names>Hongbiao</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3168383/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Shi</surname> <given-names>Deen</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3168416/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Shi</surname> <given-names>Changsheng</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Li</surname> <given-names>Lisi</given-names></name>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Gao</surname> <given-names>Yuantong</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3168612/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Hangzhou Dianzi University</institution>, <addr-line>Hangzhou</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Laboratory Medicine, Shandong Provincial Hospital Affiliated to Shandong First Medical University</institution>, <addr-line>Jinan</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Radiation Oncology, Shenzhen People&#x00027;s Hospital (The Second Clinical Medical College, Ji&#x00027;nan University, The First Affiliated Hospital of Southern University of Science and Technology)</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>School of Information Technology, Zhejiang Institute of Economics and Trade</institution>, <addr-line>Hangzhou</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>Department of Radiology, The Third Affiliated Hospital of Wenzhou Medical University</institution>, <addr-line>Wenzhou</addr-line>, <country>China</country></aff>
<aff id="aff6"><sup>6</sup><institution>Department of Interventional Vascular Surgery, The Third Affiliated Hospital of Wenzhou Medical University</institution>, <addr-line>Wenzhou</addr-line>, <country>China</country></aff>
<aff id="aff7"><sup>7</sup><institution>Shenzhen Hospital of Southern Medical University</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Ahmed Elazab, Shenzhen University, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Duolin Wang, University of Missouri, United States</p>
<p>Shuai Zeng, University of Missouri, United States</p>
<p>Shah Muhammad Azmat Ullah, Khulna University of Engineering &#x00026; Technology, Bangladesh</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Lisi Li <email>leelisy&#x00040;163.com</email></corresp>
<corresp id="c002">Yuantong Gao <email>gaoyuantonggyt&#x00040;126.com</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>26</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>19</volume>
<elocation-id>1637291</elocation-id>
<history>
<date date-type="received">
<day>30</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Shen, Hu, Zhang, Fu, Xu, Lyu, Xie, Shi, Shi, Li and Gao.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Shen, Hu, Zhang, Fu, Xu, Lyu, Xie, Shi, Shi, Li and Gao</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>As a progressive neurodegeneration, Alzheimer&#x00027;s disease (AD) represents the primary etiology of dementia among the elderly. Early identification of individuals with mild cognitive impairment (MCI) who are likely to convert to AD is essential for timely diagnosis and therapeutic intervention. Although multimodal neuroimaging and clinical data provide complementary information, existing fusion models often face challenges such as high computational complexity and limited interpretability.</p>
</sec>
<sec>
<title>Methods</title>
<p>To address these limitations, we introduce TriLightNet, an innovative lightweight triple-modal fusion network designed to integrate structural MRI, functional PET, and clinical tabular data for predicting MCI-to-AD conversion. TriLightNet incorporates a hybrid backbone that combines Kolmogorov-Arnold Networks with PoolFormer for efficient feature extraction. Additionally, it introduces a Hybrid Block Attention Module to capture subtle interactions between image and clinical features and employs a MultiModal Cascaded Attention mechanism to enable progressive and efficient fusion across the modalities. These components work together to streamline multimodal data integration while preserving meaningful insights.</p>
</sec>
<sec>
<title>Results</title>
<p>Extensive experiments conducted on the Alzheimer&#x00027;s Disease Neuroimaging Initiative (ADNI) dataset demonstrate the effectiveness of TriLightNet, showcasing superior performance compared to state-of-the-art methods. Specifically, the model achieves an accuracy of 81.25%, an AUROC of 0.8146, and an F1-score of 69.39%, all while maintaining reduced computational costs.</p>
</sec>
<sec>
<title>Discussion</title>
<p>Furthermore, its interpretability was validated using the Integrated Gradients method, which revealed clinically relevant brain regions contributing to the predictions, enhancing its potential for meaningful clinical application. Our code is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/sunyzhi55/TriLightNet">https://github.com/sunyzhi55/TriLightNet</ext-link>.</p>
</sec></abstract>
<kwd-group>
<kwd>Alzheimer&#x00027;s disease</kwd>
<kwd>mild cognitive impairment</kwd>
<kwd>triple-modal fusion</kwd>
<kwd>lightweight neural network</kwd>
<kwd>attention mechanism</kwd>
<kwd>integrated gradients</kwd>
</kwd-group>
<counts>
<fig-count count="6"/>
<table-count count="4"/>
<equation-count count="22"/>
<ref-count count="49"/>
<page-count count="14"/>
<word-count count="9255"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Brain Imaging Methods</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Alzheimer&#x00027;s disease (AD) is a progressive neurodegenerative disorder characterized by irreversible cognitive decline, significantly impacting the daily functioning of affected individuals (<xref ref-type="bibr" rid="B44">Wen et al., 2020</xref>; <xref ref-type="bibr" rid="B10">Francesconi et al., 2025</xref>). The Alzheimer&#x00027;s Association reports that over 50 million people globally are affected by AD, with projections indicating that the number of patients in the United States will more than double by 2050 (<xref ref-type="bibr" rid="B32">Moradi et al., 2015</xref>). As AD advances, individuals often suffer from severe deterioration in cognitive functions, including memory loss, language difficulties, and impaired reasoning (<xref ref-type="bibr" rid="B14">Grigas et al., 2024</xref>). The prodromal stages of AD, categorized by the severity of cognitive decline, include subjective cognitive decline (SCD) and mild cognitive impairment (MCI) (<xref ref-type="bibr" rid="B7">Elazab et al., 2024</xref>). Patients in the MCI stage already show noticeable cognitive deficits. Currently, there is no effective treatment to reverse AD or MCI, and only a limited number of medications can alleviate symptoms. In clinical practice, progressive mild cognitive impairment (pMCI) refers to patients who are likely to convert to AD within &#x0007E;3 years, whereas stable MCI (sMCI) describes those whose cognitive condition remains unchanged during the same period (<xref ref-type="bibr" rid="B12">Gaser et al., 2013</xref>). Accurately predicting pMCI is crucial for early intervention, which can delay the onset of AD.</p>
<p>Advances in neuroimaging have significantly enhanced our ability to gather detailed anatomical and functional information about the brain using techniques such as magnetic resonance imaging (MRI) and positron emission tomography (PET). MRI offers high-resolution structural details, distinguishing between gray and white matter, while PET detects functional changes in brain metabolism, providing insights into neurodegenerative processes. Leveraging this data, computer-aided diagnostic (CAD) systems have been developed to aid in the early differentiation between pMCI and sMCI (<xref ref-type="bibr" rid="B8">El-Gamal et al., 2021</xref>). Alongside imaging data, clinical information, including demographic details, laboratory test results, and neurological assessments, offers valuable insights into AD. Building on this wealth of information, machine learning (ML) techniques have been widely applied to analyze clinical data related to AD. For instance, <xref ref-type="bibr" rid="B32">Moradi et al. (2015)</xref> utilized a random forest (RF) classifier to predict the early conversion from MCI to AD. Similarly, <xref ref-type="bibr" rid="B31">Mathew et al. (2016)</xref> integrated principal component analysis, discrete wavelet transform, and support vector machines for AD classification. While these traditional ML approaches have shown promise, they heavily depend on handcrafted features, which require considerable domain expertise and are often influenced by subjective interpretations.</p>
<p>Recently, deep learning techniques such as ResNet (<xref ref-type="bibr" rid="B17">He et al., 2016</xref>), EfficientNet (<xref ref-type="bibr" rid="B39">Tan and Le, 2019</xref>), and Vision Transformer (ViT) (<xref ref-type="bibr" rid="B5">Dosovitskiy et al., 2020</xref>) have been increasingly applied to AD diagnostic tasks. <xref ref-type="bibr" rid="B41">Wang et al. (2024)</xref> introduced the HOPE framework, which leverages MRI features from various disease stages to predict the conversion from MCI to AD with promising results. (<xref ref-type="bibr" rid="B48">Zhang et al. 2025a</xref>) developed a spatiotemporal transformer-based approach for constructing asynchronous functional brain networks. Several studies have shown significant progress in MRI-based AD diagnosis (<xref ref-type="bibr" rid="B1">Atitallah et al., 2024</xref>; <xref ref-type="bibr" rid="B23">Lei et al., 2024</xref>; <xref ref-type="bibr" rid="B20">Jabason et al., 2025</xref>; <xref ref-type="bibr" rid="B16">Haq et al., 2025</xref>). However, relying solely on MRI does not capture crucial metabolic information in the brain, underscoring the limitations of single-modality approaches. To address these shortcomings, <xref ref-type="bibr" rid="B21">Kang et al. (2023)</xref> introduced the Visual Attribute Prompt Learning (VAPL) to integrate MRI with clinical tabular data, while <xref ref-type="bibr" rid="B6">Duenias et al. (2025)</xref> proposed the hyperfusion framework, which combines medical imaging with clinical features. The fusion of MRI and PET has also emerged as a common multimodal strategy, providing a comprehensive understanding of brain pathology through structural and functional imaging. <xref ref-type="bibr" rid="B26">Li et al. (2025)</xref> presented the Diamond framework based on ViT, using dual attention mechanisms to model inter-modal similarities. Other methods, such as MMGPL (<xref ref-type="bibr" rid="B34">Peng et al., 2024</xref>) and MDL (<xref ref-type="bibr" rid="B36">Qiu et al., 2024</xref>), have also achieved competitive performance. Moreover, the problem of missing modalities is commonly encountered in real-world scenarios, and previous studies have also explored this issue (<xref ref-type="bibr" rid="B29">Liu et al., 2022</xref>; <xref ref-type="bibr" rid="B19">Hu et al., 2025</xref>). Our work is currently conducted under the setting of complete modalities, and addressing modality missing will be a key direction for future research.</p>
<p>Many current models face challenges in effectively integrating more than two modalities, complicating comprehensive diagnoses that require structural imaging, functional imaging, and clinical data. To tackle this issue, <xref ref-type="bibr" rid="B25">Li et al. (2023)</xref> introduced the IMF framework, which enhances inter-modal interactions through a two-stage fusion design. The Modality-Flexible Framework offers adaptive diagnosis using diverse clinical data (<xref ref-type="bibr" rid="B49">Zhang et al., 2025b</xref>), while the longitudinal prediction method incorporates modality uncertainty to boost robustness (<xref ref-type="bibr" rid="B3">Dao et al., 2025</xref>). Innovative approaches have also emerged, such as the adversarial learning (<xref ref-type="bibr" rid="B2">Bayta&#x0015F;, 2024</xref>) and the flexible Mixture-of-Experts architecture (<xref ref-type="bibr" rid="B47">Yun et al., 2024</xref>). Despite these advancements, current multimodal fusion methods often rely heavily on deep convolutional layers and repetitive attention mechanisms, resulting in high computational overhead and increased model complexity. Streamlining these processes remains a critical area of research to enhance efficiency without sacrificing diagnostic accuracy.</p>
<p>To overcome the limitations of current models, we propose TriLightNet, a novel and lightweight triple-modal framework designed for predicting the conversion from MCI to AD. This model effectively integrates structural MRI (sMRI), Fluorodeoxyglucose PET (FDG-PET), and clinical tabular data while maintaining low computational demands. The main contributions of this work are summarized as follows:</p>
<list list-type="bullet">
<list-item><p>We develop a new backbone network, blending the representational power of Kolmogorov-Arnold Networks (KAN) (<xref ref-type="bibr" rid="B30">Liu et al., 2024</xref>) with the efficiency of PoolFormer (<xref ref-type="bibr" rid="B46">Yu et al., 2022</xref>). This fusion supports compact and robust feature extraction from clinical data, crucial for medical applications where computational resources may be limited.</p></list-item>
<list-item><p>We present the Hybrid Block Attention Module (HBAM), designed for AD diagnostic tasks, capturing intricate interactions between imaging modalities and clinical tabular variables. This module allows the model to take into account both spatial brain patterns and essential clinical indicators such as cognitive scores and patient history.</p></list-item>
<list-item><p>We propose the MultiModal Cascaded Attention (MMCA), a scalable and memory-efficient fusion strategy inspired by Cascaded Group Attention (CGA) (<xref ref-type="bibr" rid="B28">Liu et al., 2023</xref>). This mechanism progressively aggregates multimodal features, enhancing cross-modal synergy and addressing challenges related to modality imbalance, often encountered in real-world AD datasets.</p></list-item>
<list-item><p>We perform comprehensive experiments using benchmark AD datasets, including comparative experiments, ablation studies, and interpretability analyses. The results show that TriLightNet not only surpasses existing multimodal baselines in accuracy and efficiency but also offers clinically meaningful visualizations that can facilitate medical decision-making.</p></list-item>
</list>
</sec>
<sec id="s2">
<title>2 Related work</title>
<sec>
<title>2.1 Neural network backbones</title>
<p>Traditional image encoders, such as convolutional neural networks (CNNs) and Vision Transformer (ViT) (<xref ref-type="bibr" rid="B5">Dosovitskiy et al., 2020</xref>), have achieved remarkable performance across a wide range of visual tasks. While CNNs like ResNet are known for their efficiency and effectiveness, ViT typically needs higher computational resources and larger model sizes (<xref ref-type="bibr" rid="B17">He et al., 2016</xref>). To address these limitations, <xref ref-type="bibr" rid="B46">Yu et al. (2022)</xref> proposed PoolFormer, which eschews attention mechanisms in favor of spatial pooling and global token mixing, achieving competitive performance with reduced computational demands.</p>
<p>Recently, <xref ref-type="bibr" rid="B30">Liu et al. (2024)</xref> introduced the KAN, a novel architecture based on the Kolmogorov-Arnold representation theorem. KAN replaces the linear transformations in traditional multilayer perceptron (MLP) with learnable spline-based functions, offering a flexible and interpretable modeling framework with strong approximation capabilities, particularly beneficial in low-data scenarios and for capturing complex nonlinear relationships. Several studies have explored KAN in Alzheimer&#x00027;s disease. For example, <xref ref-type="bibr" rid="B40">Verma et al. (2025)</xref> combined KAN with Visual Geometry Group (VGG) for Alzheimer&#x00027;s disease prediction and achieved promising results. Others have integrated KAN with graph neural networks to further enhance modeling capabilities (<xref ref-type="bibr" rid="B43">Wang, 2025</xref>; <xref ref-type="bibr" rid="B4">Ding et al., 2025</xref>). In addition to its role as a classifier, KAN has recently been applied to tabular feature extraction tasks and demonstrated notable effectiveness (<xref ref-type="bibr" rid="B35">Poeta et al., 2024</xref>; <xref ref-type="bibr" rid="B9">Eslamian et al., 2025</xref>; <xref ref-type="bibr" rid="B11">Gao et al., 2024</xref>).</p>
<p>Building on these advancements, we propose a hybrid neural backbone that combines the architectural efficiency of PoolFormer with the functional expressiveness and interpretability of KAN. Specifically, our design utilizes PoolFormer&#x00027;s spatial feature aggregation capabilities and enhances representation learning with KAN-based modules, providing a robust and efficient framework for processing clinical tabular data.</p>
</sec>
<sec>
<title>2.2 Attention mechanism development</title>
<p>In visual tasks, traditional attention mechanisms include Squeeze and Excitation Networks (SENet) (<xref ref-type="bibr" rid="B18">Hu et al., 2018</xref>) and Convolutional Block Attention Modules (CBAM) (<xref ref-type="bibr" rid="B45">Woo et al., 2018</xref>). In particular, CBAM applies channel-wise attention and spatial attention sequentially to enhance feature representations by highlighting useful features and diminishing less important ones. More recently, <xref ref-type="bibr" rid="B28">Liu et al. (2023)</xref> proposed the CGA mechanism, a memory-efficient attention scheme that divides input tokens into separate groups and performs cascaded cross-group attention, greatly reducing computational complexity while preserving expressive power .</p>
<p>For AD analysis, attention mechanisms have also been explored. For instance, <xref ref-type="bibr" rid="B26">Li et al. (2025)</xref> combined self-attention and bi-attention to effectively fuse features from MRI and PET modalities. Similarly, <xref ref-type="bibr" rid="B21">Kang et al. (2023)</xref> employed cross-attention to integrate MRI data with clinical features. However, most existing approaches do not adequately consider the unique structural characteristics of each modality, particularly the disparity between high-dimensional imaging data and low-dimensional tabular data. Moreover, the high computational cost of attention is also a problem.</p>
<p>To address these limitations, we introduce two innovative attention modules designed for multimodal AD analysis. The first is HBAM, which extends the CBAM architecture to facilitate bidirectional attention between image and tabular modalities. The second is the MMCA, which adapts the CGA mechanism to support efficient and effective cross-modal interaction. Together, these modules form the backbone of our proposed framework, enabling precise and computationally efficient fusion of heterogeneous medical data.</p>
</sec>
<sec>
<title>2.3 Model interpretability techniques</title>
<p>Interpretability is essential in medical image analysis. Among various explainability methods (<xref ref-type="bibr" rid="B22">Kohlbrenner et al., 2020</xref>; <xref ref-type="bibr" rid="B37">Selvaraju et al., 2017</xref>; <xref ref-type="bibr" rid="B38">Sundararajan et al., 2017</xref>), Integrated Gradients (IG) (<xref ref-type="bibr" rid="B38">Sundararajan et al., 2017</xref>) has been widely adopted due to its solid theoretical foundations and practical effectiveness, which stands out as a path-based attribution method that overcomes the limitations of standard gradient-based explanations, such as saturation and noise. In the application of neuroscience in dementia detection, previous studies have successfully employed the IG attribution method (<xref ref-type="bibr" rid="B15">Gryshchuk et al., 2025</xref>; <xref ref-type="bibr" rid="B42">Wang et al., 2023</xref>). Following this line of research, we adopt IG in our study to investigate feature contributions across multimodal inputs.</p>
<p>For a given model <italic>F</italic>, an input <italic><bold>x</bold></italic>, and a baseline input <italic><bold>x</bold></italic><bold>&#x02032;</bold> (often a zero vector), IG calculates the attribution for each input dimension as the path integral of the gradients along a straight-line path from the baseline to the actual input:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>I</mml:mi><mml:mi>G</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>x</mml:mi></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>x</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>x</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x003B1;</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mstyle><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mi>F</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>x</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>x</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>x</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>x</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:mfrac><mml:mi>d</mml:mi><mml:mi>&#x003B1;</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic><bold>IG</bold></italic><sub><italic>i</italic></sub>(<italic><bold>x</bold></italic>) represents the attribution of the <italic>i</italic>-th feature. Essentially, IG quantifies each input feature&#x00027;s contribution to the change in the model output from the baseline to the actual input. This approach is particularly suitable for high-dimensional medical data, providing pixel-level attributions for image modalities and feature-level attributions for clinical inputs.</p>
</sec>
</sec>
<sec sec-type="materials and methods" id="s3">
<title>3 Materials and methods</title>
<sec>
<title>3.1 Materials</title>
<p>This study utilizes data from the Alzheimer&#x00027;s disease Neuroimaging Initiative (ADNI), drawing specifically from the ADNI-1 and ADNI-2 datasets (<xref ref-type="bibr" rid="B33">Mueller et al., 2005</xref>). The ADNI cohort includes individuals diagnosed with AD, MCI, and Normal Controls (NC). To avoid duplication, subjects appearing in both datasets were excluded from ADNI-2. We used T1-weighted sMRI, FDG-PET imaging, and clinical data, categorizing subjects into pMCI and sMCI groups. Demographic information is detailed in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Characteristics of the datasets used in experiments.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left" rowspan="2"><bold>Variable</bold></th>
<th valign="top" align="center" colspan="4"><bold>ADNI1 (Mueller et al.</bold>, <xref ref-type="bibr" rid="B33"><bold>2005</bold></xref><bold>)</bold></th>
<th valign="top" align="center" colspan="4"><bold>ADNI2 (Mueller et al.</bold>, <xref ref-type="bibr" rid="B33"><bold>2005</bold></xref><bold>)</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="center"><bold>AD</bold></th>
<th valign="top" align="center"><bold>pMCI</bold></th>
<th valign="top" align="center"><bold>sMCI</bold></th>
<th valign="top" align="center"><bold>NC</bold></th>
<th valign="top" align="center"><bold>AD</bold></th>
<th valign="top" align="center"><bold>pMCI</bold></th>
<th valign="top" align="center"><bold>sMCI</bold></th>
<th valign="top" align="center"><bold>NC</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Number (M/F)</td>
<td valign="top" align="center">88/83</td>
<td valign="top" align="center">90/61</td>
<td valign="top" align="center">136/72</td>
<td valign="top" align="center">103/104</td>
<td valign="top" align="center">89/67</td>
<td valign="top" align="center">43/38</td>
<td valign="top" align="center">156/125</td>
<td valign="top" align="center">132/165</td>
</tr>
<tr>
<td valign="top" align="left">Age</td>
<td valign="top" align="center">75.35 &#x000B1; 7.47</td>
<td valign="top" align="center">74.63 &#x000B1; 7.18</td>
<td valign="top" align="center">74.75 &#x000B1; 7.63</td>
<td valign="top" align="center">75.92 &#x000B1; 5.12</td>
<td valign="top" align="center">74.75 &#x000B1; 8.09</td>
<td valign="top" align="center">72.60 &#x000B1; 7.27</td>
<td valign="top" align="center">71.29 &#x000B1; 7.43</td>
<td valign="top" align="center">72.80 &#x000B1; 6.01</td>
</tr>
<tr>
<td valign="top" align="left">Education</td>
<td valign="top" align="center">14.64 &#x000B1; 3.19</td>
<td valign="top" align="center">15.66 &#x000B1; 2.92</td>
<td valign="top" align="center">15.61 &#x000B1; 3.11</td>
<td valign="top" align="center">15.91 &#x000B1; 2.87</td>
<td valign="top" align="center">15.72 &#x000B1; 2.75</td>
<td valign="top" align="center">16.29 &#x000B1; 2.55</td>
<td valign="top" align="center">16.31 &#x000B1; 2.61</td>
<td valign="top" align="center">16.61 &#x000B1; 2.5</td>
</tr>
<tr>
<td valign="top" align="left">CDR-SB</td>
<td valign="top" align="center">4.32 &#x000B1; 1.58</td>
<td valign="top" align="center">1.85 &#x000B1; 0.98</td>
<td valign="top" align="center">1.38 &#x000B1; 0.75</td>
<td valign="top" align="center">0.03 &#x000B1; 0.12</td>
<td valign="top" align="center">4.51 &#x000B1; 1.67</td>
<td valign="top" align="center">2.18 &#x000B1; 0.95</td>
<td valign="top" align="center">1.33 &#x000B1; 0.82</td>
<td valign="top" align="center">0.04 &#x000B1; 0.15</td>
</tr>
<tr>
<td valign="top" align="left">MMSE</td>
<td valign="top" align="center">23.23 &#x000B1; 2.03</td>
<td valign="top" align="center">26.59 &#x000B1; 1.7</td>
<td valign="top" align="center">27.33 &#x000B1; 1.77</td>
<td valign="top" align="center">29.14 &#x000B1; 0.98</td>
<td valign="top" align="center">23.12 &#x000B1; 2.07</td>
<td valign="top" align="center">27.1 &#x000B1; 1.82</td>
<td valign="top" align="center">28.21 &#x000B1; 1.63</td>
<td valign="top" align="center">28.99 &#x000B1; 1.26</td>
</tr></tbody>
</table>
</table-wrap>
<p>Given the challenges of missing PET data and class imbalance between pMCI and sMCI in both datasets, we focused on subjects with complete multimodal data, comprising MRI, PET, and clinical features. Therefore, we merge the ADNI-1 and ADNI-2 cohorts into a single dataset. The finalized dataset comprises 512 subjects: 149 pMCI and 363 sMCI. The dataset was randomly split into training and testing sets with a 4:1 ratio. The training set included 119 pMCI and 290 sMCI. The testing set comprised 30 pMCI and 73 sMCI.</p>
<p>All MRI images underwent preprocessing, including intensity normalization, skull stripping, and normalization to Montreal Neurological Institute (MNI) space. FDG-PET images were similarly processed through intensity normalization, normalization to MNI space, and co-registration with MRI images. All images were resized to a resolution of 96 &#x000D7; 128 &#x000D7; 96. For clinical data, similar to previous studies (<xref ref-type="bibr" rid="B19">Hu et al., 2025</xref>; <xref ref-type="bibr" rid="B6">Duenias et al., 2025</xref>; <xref ref-type="bibr" rid="B21">Kang et al., 2023</xref>) and based on clinical experience from doctors, we selected seven features. Among them, age, gender, and education belong to demographic attributes, while ApoE4 status, phosphorylated tau 181 (P-tau 181) and total tau (T-tau) are categorized as cerebrospinal fluid (CSF) biomarkers. The last attribute is a composite measure derived from 18F-fluorodeoxyglucose (FDG) and florbetapir (AV45) PET scans. It is worth noting that cognitive scores were excluded because they are directly related to the diagnosis of Alzheimer&#x00027;s disease and thus were excluded to avoid introducing bias into the prediction of disease progression.</p>
</sec>
<sec>
<title>3.2 Methodology</title>
<p>This section details the architecture of TriLightNet, our lightweight triple-modal fusion network for predicting MCI-to-AD progression. We first describe the feature extraction from MRI, PET, and clinical data. For imaging modalities, ResNet is employed to capture spatial features, and the tabular encoder combines KAN with PoolFormer to enable efficient and expressive representation learning from clinical variables. Subsequently, we introduce the HBAM for nuanced image-clinical feature interaction, followed by the MMCA for efficient progressive fusion of all modalities. Finally, we discuss the loss function used for model training.</p>
<sec>
<title>3.2.1 Feature extraction</title>
<p>In our study, we utilize ResNet as the encoder for medical imaging data such as MRI and PET scans to extract deep spatial features effectively. For clinical tabular data, we developed a specialized encoder that combines the KAN module with the PoolFormer architecture. The KAN module provides a flexible structure and high computational efficiency, and PoolFormer replaces the self-attention mechanisms typically found in traditional Transformer with simple pooling operations.</p>
<p>A general KAN network consists of <italic>L</italic> layers. Given an input <inline-formula><mml:math id="M2"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>x</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula>, its output is defined as:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">KAN</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>x</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mo>&#x003A6;</mml:mo></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x025CB;</mml:mo><mml:mo>&#x022EF;</mml:mo><mml:mo>&#x025CB;</mml:mo><mml:msub><mml:mrow><mml:mo>&#x003A6;</mml:mo></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x025CB;</mml:mo><mml:msub><mml:mrow><mml:mo>&#x003A6;</mml:mo></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>x</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where each &#x003A6;<sub><italic>l</italic></sub> denotes a nonlinear transformation. Typically, B-spline curves are used as the nonlinear activation functions due to their ability to precisely approximate low-dimensional functions, thereby enhancing network accuracy. A B-spline function is defined as a piecewise polynomial expressed as a linear combination of basis functions:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>S</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>x</mml:mi></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>c</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>B</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>x</mml:mi></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>n</italic> is the number of control points, <italic>t</italic> is the knot vector, <italic><bold>c</bold></italic><sub><italic>i</italic></sub> are the control point coefficients, and <italic><bold>B</bold></italic><sub><italic>i</italic></sub>(<italic>x</italic>) are the basis functions.</p>
<p>Let the original clinical input be denoted as <inline-formula><mml:math id="M5"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>I</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula>. In our implementation, the feature extraction process for clinical data involves two main stages. First, we use a single layer KAN module with SiLU-based spline activations to map the input <italic><bold>I</bold></italic><sub><italic>cli</italic></sub> from its original dimension to a higher-dimensional latent feature space. This step generates an initial feature representation, which we denote as <italic><bold>H</bold></italic><sub>0</sub>:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>H</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">KAN</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>I</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Subsequently, this initial representation <italic><bold>H</bold></italic><sub>0</sub> is fed into a sequence of <italic>M</italic> stacked PoolFormer blocks to further refine the features. Each PoolFormerBlock consists of an average pooling operation and an MLP, both integrated with residual connections and layer normalization. The transformation within the <italic>m</italic>-th block (for <italic>m</italic> &#x0003D; 1, &#x02026;, <italic>M</italic>) is defined as:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>H</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>H</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mtext class="textrm" mathvariant="normal">AvgPool</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext class="textrm" mathvariant="normal">LayerNorm</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>H</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E6"><label>(6)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>H</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>H</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:mtext class="textrm" mathvariant="normal">MLP</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext class="textrm" mathvariant="normal">LayerNorm</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>H</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic><bold>H</bold></italic><sub><italic>m</italic>&#x02212;1</sub> is the input to the <italic>m</italic>-th block and <italic><bold>H</bold></italic><sub><italic>m</italic></sub> is its output.</p>
</sec>
<sec>
<title>3.2.2 Hybrid Block Attention Module</title>
<p>We introduce HBAM, which builds upon the CBAM by extending its attention mechanism to effectively integrate multimodal inputs, specifically image and clinical features. As depicted in <xref ref-type="fig" rid="F1">Figure 1A</xref>, HBAM processes both image and tabular features. Let <inline-formula><mml:math id="M9"><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:mi>D</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> represent the feature tensor extracted from an image encoder, and <inline-formula><mml:math id="M10"><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> denotes the vectorized representation of clinical tabular data. The objective is to refine <italic><bold>F</bold></italic><sub><italic>img</italic></sub> by leveraging complementary information from <italic><bold>F</bold></italic><sub><italic>cli</italic></sub>, thereby enhancing the overall feature representation.</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>The overall architecture of TriLightNet, including <bold>(A)</bold> Hybrid Block Attention Module and <bold>(D)</bold> MultiModal Cascaded Attention. Among these, Hybrid Block Attention Module is composed of two sub-modules: <bold>(B)</bold> Hybrid Channel Attention Module, responsible for adaptively assigning weights to channel-level interactions between table and image features; and <bold>(C)</bold> Spatial Attention Module, which captures and enhances local spatial dependencies.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1637291-g0001.tif">
<alt-text>Diagram illustrating the total framework of TriLightNet. It combines MRI, PET, and clinical data through separate encoders and hybrid block attention modules. The process includes cross attention, concatenation, and projection, culminating in classification as pMCI or sMCI. Various components like hybrid channel attention and spatial attention modules refine the features. The legend explains symbols for operations such as addition, multiplication, and sigmoid functions.</alt-text>
</graphic>
</fig>
<sec>
<title>3.2.2.1 Hybrid Channel Attention Module</title>
<p>The Hybrid Channel Attention Module (HCAM) is an essential part of HBAM, specifically crafted to merge information from both image and clinical features at the channel level. As depicted in <xref ref-type="fig" rid="F1">Figure 1B</xref>, HCAM refines feature representation by utilizing complementary data from these two modalities.</p>
<p>Initially, the image feature tensor <italic><bold>F</bold></italic><sub><italic>img</italic></sub> and the clinical feature vector <italic><bold>F</bold></italic><sub><italic>cli</italic></sub> are processed. The clinical feature is embedded into a format compatible with the image feature using a linear embedding layer:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M11"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>E</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">Embed</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where Embed(&#x000B7;) is a linear layer that projects <italic><bold>F</bold></italic><sub><italic>cli</italic></sub> from <inline-formula><mml:math id="M12"><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> to <inline-formula><mml:math id="M13"><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>.</p>
<p>Subsequently, Global Average Pooling (GAP) and Global Max Pooling (GMP) are applied to the image feature tensor to yield two vectors that capture different aspects of the image feature:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>v</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">GAP</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">GMP</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>These pooled vectors are passed through a shared MLP to generate channel attention weights:</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M15"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>M</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>v</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">MLP</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>v</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>M</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">MLP</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The <italic><bold>M</bold></italic><sub><italic>avg</italic></sub>, <italic><bold>M</bold></italic><sub><italic>max</italic></sub>, and <italic><bold>E</bold></italic><sub><italic>cli</italic></sub> are combined using element-wise addition and normalization to produce the hybrid channel attention map <italic><bold>M</bold></italic><sub><italic>hc</italic></sub>:</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M16"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>M</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>M</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>v</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>M</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>E</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003C3; is the sigmoid function.</p>
<p>Finally, the hybrid channel attention map <italic><bold>M</bold></italic><sub><italic>hc</italic></sub> is applied to the original image feature tensor <italic><bold>F</bold></italic><sub><italic>img</italic></sub> through channel-wise multiplication:</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M17"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>M</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02297;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x02297; indicates channel-wise multiplication.</p>
<p>This design allows HCAM to effectively blend information from both image and tabular features, enhancing the feature representation by highlighting informative channels and suppressing less useful ones. The refined feature tensor <italic><bold>F</bold></italic><sub><italic>hc</italic></sub> is then forwarded to the Spatial Attention Module for further enhancement.</p>
</sec>
<sec>
<title>3.2.2.2 Spatial attention module</title>
<p>Following the refinement at the channel level, we employ a spatial attention module akin to that in CBAM to further emphasize important spatial regions, as shown in <xref ref-type="fig" rid="F1">Figure 1C</xref>. The spatial attention map, <inline-formula><mml:math id="M18"><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>M</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mi>D</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, is created by concatenating channel-wise average and max pooled features, which are then processed through a convolutional layer:</p>
<disp-formula id="E12"><label>(12)</label><mml:math id="M19"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>M</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Concat</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext class="textrm" mathvariant="normal">GAP</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mtext class="textrm" mathvariant="normal">GMP</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003C3; is the sigmoid function. The final output feature map is calculated as:</p>
<disp-formula id="E13"><label>(13)</label><mml:math id="M20"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>M</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02297;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Compared to the original CBAM, our HBAM offers a structured approach to integrating tabular data into the attention mechanism, enhancing the robustness and semantic relevance of feature refinement in a hybrid-modality context.</p>
</sec>
</sec>
<sec>
<title>3.2.3 Multimodal cascaded attention</title>
<p>To capture complex, hierarchical dependencies between modalities in multimodal learning, we extend CGA into the multimodal domain with the MMCA module, illustrated in <xref ref-type="fig" rid="F1">Figure 1D</xref>. The feature outputs from the HBAM modules, <italic><bold>F</bold></italic><sub><italic>mri&#x00026;cli</italic></sub> and <italic><bold>F</bold></italic><sub><italic>pet&#x00026;cli</italic></sub>, are processed by the MMCA module to produce two modality-enhanced outputs. These are then concatenated and passed through a classification head for the final prediction. <xref ref-type="fig" rid="F2">Figure 2</xref> provides further insight into the internal workflow and structure of MMCA.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>The specific architecture of MultiModal Cascaded Attention. Inputs from two modalities are first divided into <italic>N</italic> groups for independent cross-attention. Then the attention outputs from all groups are concatenated and projected. Two projected features are finally concatenated to form the final representation.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1637291-g0002.tif">
<alt-text>Diagram of a neural network model showing inputs processed in parallel groups. &#x0201C;Input1&#x0201D; and &#x0201C;Input2&#x0201D; are divided into groups, each passing through linear and attention layers. The outputs are combined via &#x0201C;Concat &#x00026; Projection&#x0201D; before final concatenation, leading to the &#x0201C;Output.&#x0201D; Arrows indicate the flow of data.</alt-text>
</graphic>
</fig>
<p>Given two input modalities, <italic><bold>X</bold></italic> and <italic><bold>Y</bold></italic>, each of shape &#x0211D;<sup><italic>B</italic>&#x000D7;<italic>L</italic>&#x000D7;<italic>C</italic></sup>, where <italic>B</italic> is the batch size, <italic>L</italic> is the sequence length, and <italic>C</italic> is the number of channels, MMCA first reshapes these inputs to &#x0211D;<sup><italic>B</italic>&#x000D7;<italic>C</italic>&#x000D7;<italic>L</italic></sup> and splits them into <italic>N</italic> groups, corresponding to the number of attention groups. For each group <italic>i</italic> in {1, &#x02026;, <italic>N</italic>}, a cascaded computation occurs, where the feature representation of the current group builds on features from previous groups, refining attention hierarchically.</p>
<p>For each group, three projections (query, key, value) are derived from the grouped features of both modalities. Let <inline-formula><mml:math id="M21"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>q</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>k</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>v</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> denote the query, key, and value for modality <italic><bold>X</bold></italic>, and <inline-formula><mml:math id="M22"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>q</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>k</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>v</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> for modality <italic><bold>Y</bold></italic>. Cross-attention is performed bidirectionally, where features from one modality attend to the keys and values of the other:</p>
<disp-formula id="E14"><label>(14)</label><mml:math id="M23"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">Attention</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>q</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>k</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>v</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">Attention</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>q</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>k</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>v</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The attention operation incorporates a learnable position-aware bias term <inline-formula><mml:math id="M24"><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>b</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>L</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> to enhance spatial sensitivity. The attention score is calculated as:</p>
<disp-formula id="E15"><label>(15)</label><mml:math id="M25"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">Attention</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>q</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>k</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>v</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">Softmax</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>q</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>k</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>d</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>b</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>v</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic><bold>d</bold></italic><sub><italic>k</italic><sub><italic>i</italic></sub></sub> is the dimension of the key vectors. Specifically, the Softmax transforms attention logits into a probability distribution across all input positions, which is defined as:</p>
<disp-formula id="E16"><label>(16)</label><mml:math id="M26"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">Softmax</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>z</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>z</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="false"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>z</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic><bold>z</bold></italic><sub><italic>i</italic></sub> denotes the attention logit corresponding to position <italic>i</italic>, and <italic>L</italic> is the total number of positions.</p>
<p>The outputs from all groups are concatenated along the channel dimension and projected to fuse the attended features:</p>
<disp-formula id="E17"><label>(17)</label><mml:math id="M27"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>Z</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Proj</mml:mtext></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Concat</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mstyle></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mstyle></mml:mrow><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>Z</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Proj</mml:mtext></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Concat</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mstyle></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mstyle></mml:mrow><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M28"><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>Z</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>Z</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>L</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>C</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> are the fused representations for modalities <italic><bold>X</bold></italic> and <italic><bold>Y</bold></italic>, respectively.</p>
<p>Finally, <italic><bold>Z</bold></italic><sub><italic>x</italic></sub> and <italic><bold>Z</bold></italic><sub><italic>y</italic></sub> are concatenated along the last dimension to form the final fused representation <inline-formula><mml:math id="M29"><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>Z</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>f</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>L</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mn>2</mml:mn><mml:mi>C</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>:</p>
<disp-formula id="E18"><label>(18)</label><mml:math id="M30"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>Z</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>f</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">Concat</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>Z</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>Z</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The MMCA module offers significant advantages. Its cascaded structure facilitates progressive cross-modal feature fusion, with early attention groups guiding subsequent ones. Moreover, the bidirectional design ensures both modalities are symmetrically enhanced by the other&#x00027;s information, fostering balanced and robust fusion in multimodal contexts.</p>
</sec>
<sec>
<title>3.2.4 Loss function</title>
<p>In order to address class imbalance within our dataset, we utilize the focal loss function (<xref ref-type="bibr" rid="B27">Lin et al., 2017</xref>). In classification tasks, the cross-entropy (ce) loss is traditionally employed, defined as:</p>
<disp-formula id="E19"><label>(19)</label><mml:math id="M31"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>y</italic>&#x02208;{0, 1} signifies the ground-truth label, while <italic>p</italic>&#x02208;[0, 1] indicates the predicted probability for the positive class. This conventional approach assigns equal weight to both positive and negative samples. Consequently, models often exhibit a bias toward the majority class when faced with class imbalance.</p>
<p>To alleviate this issue, we have adopted the focal loss, which refines the cross-entropy loss by incorporating a dynamic modulating factor. This factor reduces the weight of easily classified samples and concentrates learning on challenging, misclassified instances. The focal loss <inline-formula><mml:math id="M32"><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mi>f</mml:mi><mml:mi>o</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is expressed as:</p>
<disp-formula id="E20"><label>(20)</label><mml:math id="M33"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mi>f</mml:mi><mml:mi>o</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B1;&#x02208;[0, 1] serves to balance the significance between positive and negative samples, while &#x003B3;&#x0003E;0 functions as a focusing parameter, regulating the degree to which easy examples are down-weighted.</p>
</sec>
</sec>
</sec>
<sec sec-type="results" id="s4">
<title>4 Results</title>
<sec>
<title>4.1 Implementation details</title>
<p>All experiments were conducted using PyTorch version 2.6.0 alongside CUDA 11.8, running on a single NVIDIA V100 32GB GPU. The model underwent training for 200 epochs with a batch size of 8, allowing for efficient data management. For optimizing model parameters, the Adam optimizer was utilized, with the learning rate fixed at 0.0001 to facilitate precise adjustments during training. Given the relatively small size of the datasets, we adopted several strategies to reduce sampling bias and alleviate overfitting. First, we performed five-fold cross-validation to ensure robust evaluation of the model&#x00027;s performance. Secondly, we used a cosine scheduler with a hyperparameter <italic>T</italic><sub><italic>max</italic></sub> set to 50, allowing for dynamic adjustment of the learning rate throughout the training process and enhancing the model&#x00027;s adaptation capabilities. Additionally, we incorporated an early stopping strategy with a patience value of 50, which effectively prevented overfitting by halting training when the validation loss ceased to improve.</p>
<p>For evaluation, we employ eight metrics to assess both classification performance and model efficiency: Accuracy, Sensitivity, Precision, Area Under the Receiver Operating Characteristic Curve (AUROC), F1-score, Balanced Accuracy, number of parameters (Params), and floating point operations (FLOPs). Given the dataset&#x00027;s class imbalance, we emphasize F1-score and Balanced Accuracy for fairer evaluation. Balanced Accuracy mitigates bias from uneven class distributions and is better suited for imbalanced binary tasks, which is defined as:</p>
<disp-formula id="E21"><label>(21)</label><mml:math id="M34"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>B</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>c</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>A</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mtext>TP</mml:mtext></mml:mrow><mml:mrow><mml:mtext>TP</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:mtext>FN</mml:mtext></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:mfrac><mml:mrow><mml:mtext>TN</mml:mtext></mml:mrow><mml:mrow><mml:mtext>TN</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:mtext>FP</mml:mtext></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where TP, TN, FP, and FN represent the numbers of true positives, true negatives, false positives, and false negatives, respectively.</p>
</sec>
<sec>
<title>4.2 Comparative experiments</title>
<p>We comprehensively evaluate our framework against representative multimodal approaches under three fusion scenarios. For MRI and PET bimodal fusion, we compare with ResNet&#x0002B;Concat (<xref ref-type="bibr" rid="B17">He et al., 2016</xref>), ViT&#x0002B;Concat (<xref ref-type="bibr" rid="B5">Dosovitskiy et al., 2020</xref>), nnMamba&#x0002B;Concat (<xref ref-type="bibr" rid="B13">Gong et al., 2025</xref>), Diamond (<xref ref-type="bibr" rid="B26">Li et al., 2025</xref>), and MDL (<xref ref-type="bibr" rid="B36">Qiu et al., 2024</xref>). For MRI and clinical fusion, methods include VAPL (<xref ref-type="bibr" rid="B21">Kang et al., 2023</xref>) and HyperFusionNet (<xref ref-type="bibr" rid="B6">Duenias et al., 2025</xref>). In the trimodal setting integrating MRI, PET, and clinical data, we compare with IMF (<xref ref-type="bibr" rid="B25">Li et al., 2023</xref>), HFBSurv (<xref ref-type="bibr" rid="B24">Li et al., 2022</xref>), ITCFN (<xref ref-type="bibr" rid="B19">Hu et al., 2025</xref>), and MultimodalADNet (<xref ref-type="bibr" rid="B49">Zhang et al., 2025b</xref>). Detailed comparative metrics are presented in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Comparison of various methods on the pMCI vs. sMCI classification task based on five-fold cross-validation.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left" rowspan="2"><bold>Method</bold></th>
<th valign="top" align="center" colspan="3"><bold>Modality</bold></th>
<th valign="top" align="center" colspan="8"><bold>Performance metrics</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="center"><bold>M</bold></th>
<th valign="top" align="center"><bold>P</bold></th>
<th valign="top" align="center"><bold>C</bold></th>
<th valign="top" align="center"><bold>Accuracy (%)</bold> &#x02191;</th>
<th valign="top" align="center"><bold>Sensitivity (%)</bold> &#x02191;</th>
<th valign="top" align="center"><bold>Precision (%)</bold> &#x02191;</th>
<th valign="top" align="center"><bold>AUROC</bold> &#x02191;</th>
<th valign="top" align="center"><bold>F1-score (%)</bold> &#x02191;</th>
<th valign="top" align="center"><bold>Balanced accuracy (%)</bold> &#x02191;</th>
<th valign="top" align="center"><bold>Params (M)</bold> &#x02193;</th>
<th valign="top" align="center"><bold>FLOPs (G)</bold> &#x02193;</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">ResNet&#x0002B;Concat (<xref ref-type="bibr" rid="B17">He et al., 2016</xref>)</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td/>
<td valign="top" align="center">73.75 &#x000B1; 3.79</td>
<td valign="top" align="center"><inline-formula><mml:math id="M35"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>75.04</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>9.16</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center">53.81 &#x000B1; 5.43</td>
<td valign="top" align="center">0.7699 &#x000B1; 0.0276</td>
<td valign="top" align="center">62.33 &#x000B1; 4.95</td>
<td valign="top" align="center">73.26 &#x000B1; 6.15</td>
<td valign="top" align="center">66.952</td>
<td valign="top" align="center">70.924</td>
</tr>
<tr>
<td valign="top" align="left">ViT&#x0002B;Concat (<xref ref-type="bibr" rid="B5">Dosovitskiy et al., 2020</xref>)</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td/>
<td valign="top" align="center">71.75 &#x000B1; 3.12</td>
<td valign="top" align="center">29.09 &#x000B1; 18.08</td>
<td valign="top" align="center">42.38 &#x000B1; 22.15</td>
<td valign="top" align="center">0.5523 &#x000B1; 0.0777</td>
<td valign="top" align="center">33.23 &#x000B1; 18.82</td>
<td valign="top" align="center">59.12 &#x000B1; 6.03</td>
<td valign="top" align="center">20.853</td>
<td valign="top" align="center">20.853</td>
</tr>
<tr>
<td valign="top" align="left">nnMamba&#x0002B;Concat (<xref ref-type="bibr" rid="B13">Gong et al., 2025</xref>)</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td/>
<td valign="top" align="center">73.75 &#x000B1; 5.04</td>
<td valign="top" align="center">51.31 &#x000B1; 16.36</td>
<td valign="top" align="center">57.81 &#x000B1; 9.18</td>
<td valign="top" align="center">0.6978 &#x000B1; 0.0376</td>
<td valign="top" align="center">51.75 &#x000B1; 4.10</td>
<td valign="top" align="center">66.71 &#x000B1; 2.17</td>
<td valign="top" align="center">26.042</td>
<td valign="top" align="center">48.626</td>
</tr>
<tr>
<td valign="top" align="left">Diamond (<xref ref-type="bibr" rid="B26">Li et al., 2025</xref>)</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td/>
<td valign="top" align="center"><inline-formula><mml:math id="M36"><mml:mrow><mml:mstyle mathcolor="#1f3bff"><mml:mn>77.76</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>1.72</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center">48.73 &#x000B1; 4.09</td>
<td valign="top" align="center"><inline-formula><mml:math id="M37"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>67.13</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>6.79</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center">0.7257 &#x000B1; 0.0304</td>
<td valign="top" align="center">56.02 &#x000B1; 1.24</td>
<td valign="top" align="center">69.19 &#x000B1; 0.72</td>
<td valign="top" align="center">23.504</td>
<td valign="top" align="center">97.638</td>
</tr>
<tr>
<td valign="top" align="left">MDL (<xref ref-type="bibr" rid="B36">Qiu et al., 2024</xref>)</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td/>
<td valign="top" align="center">71.50 &#x000B1; 7.39</td>
<td valign="top" align="center">54.51 &#x000B1; 17.94</td>
<td valign="top" align="center">57.29 &#x000B1; 13.57</td>
<td valign="top" align="center">0.7130 &#x000B1; 0.0321</td>
<td valign="top" align="center">51.65 &#x000B1; 3.54</td>
<td valign="top" align="center">66.44 &#x000B1; 1.98</td>
<td valign="top" align="center"><inline-formula><mml:math id="M38"><mml:mrow><mml:mstyle mathcolor="#1f3bff"><mml:mn>10.707</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M39"><mml:mrow><mml:mstyle mathcolor="#1f3bff"><mml:mn>19.243</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td valign="top" align="left">VAPL (<xref ref-type="bibr" rid="B21">Kang et al., 2023</xref>)</td>
<td valign="top" align="center">&#x02713;</td>
<td/>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">69.25 &#x000B1; 4.00</td>
<td valign="top" align="center">42.25 &#x000B1; 15.78</td>
<td valign="top" align="center">48.27 &#x000B1; 6.03</td>
<td valign="top" align="center">0.6701 &#x000B1; 0.0680</td>
<td valign="top" align="center">42.92 &#x000B1; 10.81</td>
<td valign="top" align="center">61.46 &#x000B1; 5.26</td>
<td valign="top" align="center">63.504</td>
<td valign="top" align="center">40.350</td>
</tr>
<tr>
<td valign="top" align="left">HyperFusionNet (<xref ref-type="bibr" rid="B6">Duenias et al., 2025</xref>)</td>
<td valign="top" align="center">&#x02713;</td>
<td/>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">75.50 &#x000B1; 2.69</td>
<td valign="top" align="center">54.24 &#x000B1; 8.25</td>
<td valign="top" align="center">59.96 &#x000B1; 6.92</td>
<td valign="top" align="center">0.7330 &#x000B1; 0.0175</td>
<td valign="top" align="center">56.05 &#x000B1; 2.76</td>
<td valign="top" align="center">69.20 &#x000B1; 1.50</td>
<td valign="top" align="center">15.402</td>
<td valign="top" align="center">47.750</td>
</tr>
<tr>
<td valign="top" align="left">IMF (<xref ref-type="bibr" rid="B25">Li et al., 2023</xref>)</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">77.75 &#x000B1; 3.98</td>
<td valign="top" align="center">70.07 &#x000B1; 9.52</td>
<td valign="top" align="center">61.81 &#x000B1; 10.27</td>
<td valign="top" align="center"><inline-formula><mml:math id="M40"><mml:mrow><mml:mstyle mathcolor="#1f3bff"><mml:mn>0.7946</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>0.0316</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M41"><mml:mrow><mml:mstyle mathcolor="#1f3bff"><mml:mn>64.34</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>2.01</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M42"><mml:mrow><mml:mstyle mathcolor="#1f3bff"><mml:mn>75.39</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>1.41</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center">67.843</td>
<td valign="top" align="center">70.925</td>
</tr>
<tr>
<td valign="top" align="left">HFBsurv (<xref ref-type="bibr" rid="B24">Li et al., 2022</xref>)</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">75.00 &#x000B1; 3.26</td>
<td valign="top" align="center">71.73 &#x000B1; 5.70</td>
<td valign="top" align="center">56.12 &#x000B1; 5.24</td>
<td valign="top" align="center">0.7552 &#x000B1; 0.0371</td>
<td valign="top" align="center">62.54 &#x000B1; 2.17</td>
<td valign="top" align="center">74.10 &#x000B1; 1.37</td>
<td valign="top" align="center">34.123</td>
<td valign="top" align="center">141.849</td>
</tr>
<tr>
<td valign="top" align="left">ITCFN (<xref ref-type="bibr" rid="B19">Hu et al., 2025</xref>)</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">75.50 &#x000B1; 3.76</td>
<td valign="top" align="center">73.31 &#x000B1; 3.80</td>
<td valign="top" align="center">56.40 &#x000B1; 6.03</td>
<td valign="top" align="center">0.7750 &#x000B1; 0.0580</td>
<td valign="top" align="center">63.58 &#x000B1; 4.44</td>
<td valign="top" align="center">74.87 &#x000B1; 3.23</td>
<td valign="top" align="center">71.305</td>
<td valign="top" align="center">71.098</td>
</tr>
<tr>
<td valign="top" align="left">MultimodalADNet (<xref ref-type="bibr" rid="B49">Zhang et al., 2025b</xref>)</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">73.25 &#x000B1; 4.23</td>
<td valign="top" align="center">63.93 &#x000B1; 17.73</td>
<td valign="top" align="center">57.77 &#x000B1; 11.18</td>
<td valign="top" align="center">0.7434 &#x000B1; 0.0236</td>
<td valign="top" align="center">57.31 &#x000B1; 3.53</td>
<td valign="top" align="center">70.53 &#x000B1; 2.62</td>
<td valign="top" align="center"><inline-formula><mml:math id="M43"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>4.320</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center">20.307</td>
</tr>
<tr>
<td valign="top" align="left"><bold>TriLightNet (Ours)</bold></td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center"><inline-formula><mml:math id="M44"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>81.25</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>0.93</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M45"><mml:mrow><mml:mstyle mathcolor="#1f3bff"><mml:mn>73.91</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>8.17</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M46"><mml:mrow><mml:mstyle mathcolor="#1f3bff"><mml:mn>65.38</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>2.91</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M47"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>0.8146</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>0.0029</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M48"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>69.39</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>3.76</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M49"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>79.06</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>2.85</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center">17.405</td>
<td valign="top" align="center"><inline-formula><mml:math id="M50"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>10.517</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The M, P, and C columns indicate the MRI, PET, and Clinical data modalities, respectively. The best and second-best results are in <inline-formula><mml:math id="M51"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mtext>red</mml:mtext></mml:mstyle></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M52"><mml:mrow><mml:mstyle mathcolor="#1f3bff"><mml:mtext>blue</mml:mtext></mml:mstyle></mml:mrow></mml:math></inline-formula>, respectively.</p>
</table-wrap-foot>
</table-wrap>
<p>TriLightNet achieves outstanding performance across several evaluation metrics, with an accuracy of 81.25%, AUROC of 0.8146, and F1-score of 69.39%, all surpassing those of the baseline models. Although ResNet&#x0002B;Concat attains the highest Sensitivity at 75.04%, its relatively low Accuracy and Precision result in suboptimal overall performance. In contrast, TriLightNet ranks second in Sensitivity while maintaining the highest Balanced Accuracy, demonstrating its strong capability to effectively integrate multimodal information and deliver more reliable predictive outcomes.</p>
<p>In terms of computational efficiency, TriLightNet has 17.405 million Params and 10.517 billion FLOPs. While models such as MDL and MultimodalADNet have fewer Params, their predictive performance is significantly inferior. Among the tri-modal approaches, TriLightNet consistently outperforms competing methods, including HFBSurv (34.123 million Params and 141.849 billion FLOPs) and IMF (67.843 million Params and 70.925 billion FLOPs). This efficiency highlights that TriLightNet not only delivers superior accuracy but also exhibits enhanced computational resource efficiency, making it particularly suitable for real-world applications that demand accurate and efficient prediction of cognitive impairment.</p>
</sec>
<sec>
<title>4.3 Ablation study</title>
<p>We conduct the following two ablation experiments: (1) Assessing the individual contributions of the HBAM and MMCA modules. (2) Evaluating the effect of KAN for the tabular encoder.</p>
<p>The results of ablation experiment (1) are shown in <xref ref-type="table" rid="T3">Table 3</xref>, demonstrating the significant contributions of both the HBAM and MMCA modules to improve the model&#x00027;s performance. Specifically, HBAM improves the model&#x00027;s accuracy by 0.5%, sensitivity by a remarkable 27.03%, AUROC by 0.1353, and F1-score by 13.91%. These improvements can be attributed to the design of channel attention within the HBAM framework, which facilitates the integration of clinical features into the image feature channels. This interaction at the channel level is followed by spatial attention enhancement, enabling the model to produce more concentrated and informative feature representations.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Ablation study on the impact of HBAM and MMCA modules.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left" colspan="2"><bold>Module</bold></th>
<th valign="top" align="center" colspan="6"><bold>Performance metrics</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>HBAM</bold></th>
<th valign="top" align="center"><bold>MMCA</bold></th>
<th valign="top" align="center"><bold>Accuracy (%)</bold> &#x02191;</th>
<th valign="top" align="center"><bold>Sensitivity (%)</bold> &#x02191;</th>
<th valign="top" align="center"><bold>Precision (%)</bold> &#x02191;</th>
<th valign="top" align="center"><bold>AUROC</bold> &#x02191;</th>
<th valign="top" align="center"><bold>F1-score (%)</bold> &#x02191;</th>
<th valign="top" align="center"><bold>Balanced accuracy (%)</bold> &#x02191;</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">73.50 &#x000B1; 10.53</td>
<td valign="top" align="center">48.37 &#x000B1; 23.95</td>
<td valign="top" align="center">64.66 &#x000B1; 16.79</td>
<td valign="top" align="center">0.6266 &#x000B1; 0.1591</td>
<td valign="top" align="center">49.13 &#x000B1; 17.86</td>
<td valign="top" align="center">66.01 &#x000B1; 9.52</td>
</tr>
<tr>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">74.00 &#x000B1; 2.67</td>
<td valign="top" align="center"><inline-formula><mml:math id="M53"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>75.40</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>7.86</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center">54.60 &#x000B1; 3.74</td>
<td valign="top" align="center">0.7619 &#x000B1; 0.0356</td>
<td valign="top" align="center">63.04 &#x000B1; 3.31</td>
<td valign="top" align="center">74.39 &#x000B1; 2.89</td>
</tr>
<tr>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">78.50 &#x000B1; 4.70</td>
<td valign="top" align="center">72.94 &#x000B1; 5.53</td>
<td valign="top" align="center">63.40 &#x000B1; 9.76</td>
<td valign="top" align="center">0.8011 &#x000B1; 0.0168</td>
<td valign="top" align="center">67.07 &#x000B1; 3.99</td>
<td valign="top" align="center">76.89 &#x000B1; 2.34</td>
</tr>
<tr>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center"><inline-formula><mml:math id="M54"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>81.25</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>0.93</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center">73.91 &#x000B1; 8.17</td>
<td valign="top" align="center"><inline-formula><mml:math id="M55"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>65.38</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>2.91</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M56"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>0.8146</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>0.0029</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M57"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>69.39</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>3.76</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M58"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>79.06</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>2.85</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The best result is in <inline-formula><mml:math id="M59"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mtext>red</mml:mtext></mml:mstyle></mml:mrow></mml:math></inline-formula>.</p>
</table-wrap-foot>
</table-wrap>
<p>For MMCA module, it increases accuracy by 5%, AUROC by 0.1745 and balance accuracy by 10.88%. Although there is a slight decrease in precision, the stability of the model improves. This performance can be primarily attributed to two key mechanisms: progressive fusion, which gradually refines dominant features from shallow to deep layers; and bidirectional symmetric enhancement, where each set of cross-attention operations is bidirectional, allowing mutual reinforcement between MRI and PET features, thus achieving comprehensive and balanced multimodal fusion.</p>
<p>Finally, when both HBAM and MMCA modules are jointly applied, the model achieves the best performance across all five evaluation metrics: an accuracy of 81.25%, precision of 65.38%, AUROC of 0.8146, F1-score of 69.39%, and balanced accuracy of 79.06%, demonstrating the complementary benefits and synergistic effect of combining both modules. This indicates that the synergy between these two modules significantly enhances the model&#x00027;s classification capabilities, highlighting the importance of integrating both the HBAM and MMCA modules into the TriLightNet model.</p>
<p><xref ref-type="table" rid="T4">Table 4</xref> presents the ablation results of experiment (2). It illustrates that KAN demonstrates a remarkable advantage over the MLP in clinical tabular feature extraction, achieving consistent improvements across all evaluation metrics. In particular, KAN increases F1-score by 1.72% and balanced accuracy by 1.98%. This improvement can be attributed to KAN&#x00027;s use of expressive B-spline basis functions for modeling clinical features, rather than simple activation functions (e.g., Sigmoid, ReLU) commonly used in MLPs. This enhanced representational capacity allows for better capture of nonlinear patterns in clinical variables, thereby improving overall performance.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Ablation study on the impact KAN in the hybrid backbone.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left" rowspan="2"><bold>Module</bold></th>
<th valign="top" align="center" colspan="6"><bold>Performance metrics</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="center"><bold>Accuracy (%)</bold> &#x02191;</th>
<th valign="top" align="center"><bold>Sensitivity (%)</bold> &#x02191;</th>
<th valign="top" align="center"><bold>Precision (%)</bold> &#x02191;</th>
<th valign="top" align="center"><bold>AUROC</bold> &#x02191;</th>
<th valign="top" align="center"><bold>F1-score (%)</bold> &#x02191;</th>
<th valign="top" align="center"><bold>Balanced accuracy (%)</bold> &#x02191;</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">MLP &#x0002B; PoolFormer</td>
<td valign="top" align="center">80.11 &#x000B1; 1.55</td>
<td valign="top" align="center">70.84 &#x000B1; 5.50</td>
<td valign="top" align="center">65.22 &#x000B1; 4.87</td>
<td valign="top" align="center">0.8101 &#x000B1; 0.0118</td>
<td valign="top" align="center">67.67 &#x000B1; 2.62</td>
<td valign="top" align="center">77.08 &#x000B1; 2.03</td>
</tr>
<tr>
<td valign="top" align="left">KAN &#x0002B; PoolFormer</td>
<td valign="top" align="center"><inline-formula><mml:math id="M60"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>81.25</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>0.93</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M61"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>73.91</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>8.17</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M62"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>65.38</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>2.91</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M63"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>0.8146</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>0.0029</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M64"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>69.39</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>3.76</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M65"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mn>79.06</mml:mn><mml:mo>&#x000B1;</mml:mo><mml:mn>2.85</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The best result is in <inline-formula><mml:math id="M66"><mml:mrow><mml:mstyle mathcolor="#ff0000"><mml:mtext>red</mml:mtext></mml:mstyle></mml:mrow></mml:math></inline-formula>.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>4.4 Model visualization and interpretability</title>
<sec>
<title>4.4.1 Metrics visualization</title>
<p>To clearly illustrate the effectiveness of our model, we present a visual comparison of key evaluation metrics using bubble charts, as depicted in <xref ref-type="fig" rid="F3">Figure 3</xref>. We focused on three crucial metrics: Balanced Accuracy, Params, and FLOPs.</p>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>Model metrics bubble chart. <bold>(A)</bold> Balanced accuracy vs. Params. <bold>(B)</bold> Balanced accuracy vs. FLOPs.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1637291-g0003.tif">
<alt-text>Scatterplots labeled (A) and (B) compare models by balanced accuracy versus parameters (in millions) and FLOPs (in gigaFLOPs), respectively. Models are color-coded and sized by the vertical axis metric, with names like ResNet&#x0002B;Concat and IMF displayed.</alt-text>
</graphic>
</fig>
<p><xref ref-type="fig" rid="F3">Figure 3A</xref> shows the relationship between Balanced Accuracy and Params. TriLightNet stands out by achieving excellent performance with a relatively low parameter count, underscoring its efficiency. <xref ref-type="fig" rid="F3">Figure 3B</xref> compares Balanced Accuracy with FLOPs, further showcasing TriLightNet&#x00027;s lightweight design and high performance.</p>
<p>Additionally, we assessed the robustness and generalization capabilities of each model by testing the trained models from each fold of the five-fold cross-validation on a test set. The Receiver Operating Characteristic (ROC) curves for each fold are presented in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig position="float" id="F4">
<label>Figure 4</label>
<caption><p>Test set performance for each of the five models trained during five-fold cross-validation. <bold>(A)</bold> Resnet &#x0002B; Concat. <bold>(B)</bold> ViT &#x0002B; Concat. <bold>(C)</bold> nnMamba &#x0002B; Concat. <bold>(D)</bold> Diamond. <bold>(E)</bold> MDL. <bold>(F)</bold> VAPL. <bold>(G)</bold> hyperfusionNet. <bold>(H)</bold> IMF. <bold>(I)</bold> HFBSurv. <bold>(J)</bold> ITCFN. <bold>(K)</bold> MultimodalADNet. <bold>(L)</bold> TriLightNet (Ours).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1637291-g0004.tif">
<alt-text>Twelve ROC curve plots labeled A to L, comparing model performance across different folds. Each plot shows true positive rate versus false positive rate, with random chance indicated by a diagonal line. Legends display results for five folds and the random chance line. Models vary, titles indicate specific model names, showing varied performance across plots.</alt-text>
</graphic>
</fig>
<p>From <xref ref-type="fig" rid="F4">Figure 4</xref>, we observe that certain triple-modal approaches, such as IMF and ITCFN, show significant fluctuations in their AUC scores across different folds, suggesting potential instability. In contrast, our proposed method, TriLightNet, displays consistently high AUC results across all five folds. This consistency provides strong empirical evidence of TriLightNet&#x00027;s superior generalization capability and robustness when handling diverse validation subsets.</p>
</sec>
<sec>
<title>4.4.2 Model interpretability</title>
<p>To improve the interpretability of our model and gain insight into how various input modalities affect the final decision, we utilize the IG method (<xref ref-type="bibr" rid="B38">Sundararajan et al., 2017</xref>), a common technique for interpreting deep neural networks. We specifically apply IG to MRI and PET image modalities to produce attribution maps that display the contribution of each voxel to the model&#x00027;s predictions. These maps are then overlaid on the original MRI and PET images for visualization. For a fair comparison, we also apply the same IG-based interpretability method to several representative multimodal fusion baselines, including IMF, ITCFN, MultimodalADNet, and HFBSurv. This allows us to compare the spatial attention patterns across models under identical conditions.</p>
<p>As illustrated in <xref ref-type="fig" rid="F5">Figures 5</xref>, <xref ref-type="fig" rid="F6">6</xref>, the attribution maps reveal distinct attention patterns across different models. For instance, TriLightNet (Ours) tends to focus on clinically relevant regions such as the hippocampus and posterior cingulate cortex, which are known to be associated with Alzheimer&#x00027;s disease progression. In contrast, some baseline models exhibit more diffused or inconsistent attention patterns. These findings suggest that our model not only achieves superior performance but also offers more focused and biologically plausible interpretability.</p>
<fig position="float" id="F5">
<label>Figure 5</label>
<caption><p>Comparison of Integrated Gradients attribution maps on MRI across different models. <bold>(A)</bold> Original. <bold>(B)</bold> IMF. <bold>(C)</bold> ITCFN. <bold>(D)</bold> MultimodalADNet. <bold>(E)</bold> HFBSurv. <bold>(F)</bold> TriLightNet (Ours).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1637291-g0005.tif">
<alt-text>MRI brain scans in axial, coronal, and sagittal views. Each row shows the same view with variations in colors, highlighting different areas of activity or features. Columns are labeled A to F.</alt-text>
</graphic>
</fig>
<fig position="float" id="F6">
<label>Figure 6</label>
<caption><p>Comparison of Integrated Gradients attribution maps on PET across different models. <bold>(A)</bold> Original. <bold>(B)</bold> IMF. <bold>(C)</bold> ITCFN. <bold>(D)</bold> MultimodalADNet. <bold>(E)</bold> HFBSurv. <bold>(F)</bold> TriLightNet (Ours).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1637291-g0006.tif">
<alt-text>PET brain images in axial, coronal, and sagittal views are displayed in a grid. The first column presents original grayscale images, while columns B to F highlight varying patterns of colored activity, primarily in blue and yellow, indicating different data overlays or signal intensities.</alt-text>
</graphic>
</fig>
</sec>
</sec>
</sec>
<sec sec-type="discussion" id="s5">
<title>5 Discussion</title>
<p>The TriLightNet, a novel lightweight triple-modal fusion network, demonstrated superior performance in predicting MCI-to-AD progression. Its strong performance can be attributed to synergistic modules: the HBAM for nuanced cross-modal feature interaction and the MMCA for efficient, hierarchical data integration, both validated by ablation studies. Compared to competing methods, TriLightNet not only surpassed existing methods in accuracy and AUROC but also achieved this with significantly lower computational costs (Params and FLOPs), highlighting its practical value in real-world clinical settings where computational resources are often limited.</p>
<p>Furthermore, the IG provided valuable insights into the decision-making process of TriLightNet. The attribution maps demonstrated that TriLightNet consistently focused on clinically relevant brain regions across both MRI and PET modalities, including the hippocampus, medial temporal lobe, and posterior cingulate cortex, which are affected early in AD. Notably, while competing models such as IMF and ITCFN showed reasonable focus on disease-related regions in MRI, they exhibited weaker and more diffuse attention patterns in PET images. In contrast, TriLightNet maintained clear and concentrated attribution in both MRI and PET, indicating its superior ability to extract complementary structural and functional information. It suggests that TriLightNet achieves not only better predictive accuracy but also more biologically plausible feature learning, potentially enhancing its interpretability and clinical trustworthiness.</p>
<p>Despite these promising results, several limitations should be acknowledged. First, although we leveraged the ADNI dataset for evaluation, external validation on independent cohorts is necessary to assess generalizability across diverse populations and imaging protocols. Second, TriLightNet currently requires complete multimodal data, future work will focus on extending the framework to handle incomplete modality scenarios, which are common in real-world clinical practice. Third, while KAN was utilized as an efficient feature extractor for tabular data, its intrinsic interpretability was not fully explored. Future work may further investigate KAN&#x00027;s potential to reveal nonlinear relationships between clinical variables and AD progression.</p>
</sec>
<sec sec-type="conclusions" id="s6">
<title>6 Conclusion</title>
<p>We introduced TriLightNet, a novel and efficient triple-modal fusion network for predicting cognitive decline in AD by integrating MRI, PET, and clinical tabular data. Extensive experiments on the ADNI dataset demonstrated TriLightNet&#x00027;s superior classification performance over state-of-the-art multimodal methods, alongside significant reductions in parameter count and computational cost. Key contributions include a hybrid KAN-PoolFormer backbone for efficient tabular feature extraction, an HBAM for enhanced imaging-clinical data interactions, and an MMCA for progressive cross-modal fusion. Beyond state-of-the-art results, TriLightNet offers interpretability via IG-based attribution maps, highlighting disease-relevant brain regions and underscoring its potential for aiding timely clinical interventions in AD progression. Future work will focus on adapting TriLightNet to handle incomplete or missing modalities and validating its generalizability across multi-center datasets. Additionally, we plan to explore the standalone application of the KAN network to clinical datasets for predictive modeling and interpretability analysis through visualization.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The data used in this study are publicly available from the Alzheimer&#x00027;s Disease Neuroimaging Initiative (ADNI) database: <ext-link ext-link-type="uri" xlink:href="http://adni.loni.usc.edu/">http://adni.loni.usc.edu/</ext-link>. Researchers can apply for access to the ADNI data by registering and submitting a data access request through the ADNI Data Sharing and Publications Committee. All data used in this work, including structural MRI, PET scans, and clinical information, were obtained following ADNI&#x00027;s data usage policies and guidelines.</p>
</sec>
<sec sec-type="ethics-statement" id="s8">
<title>Ethics statement</title>
<p>The studies involving humans were approved by Human Research Protections Program, University of California, San Diego (UCSD), USA. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study. Written informed consent was obtained from the individual(s) for the publication of any potentially identifiable images or data included in this article.</p>
</sec>
<sec sec-type="author-contributions" id="s9">
<title>Author contributions</title>
<p>XS: Writing &#x02013; original draft, Visualization, Writing &#x02013; review &#x00026; editing, Validation, Methodology. XH: Data curation, Investigation, Writing &#x02013; original draft. RZ: Formal analysis, Conceptualization, Supervision, Investigation, Writing &#x02013; review &#x00026; editing. YF: Investigation, Writing &#x02013; review &#x00026; editing, Validation. JX: Project administration, Supervision, Validation, Writing &#x02013; review &#x00026; editing. DL: Conceptualization, Writing &#x02013; review &#x00026; editing, Formal analysis, Supervision. HX: Supervision, Investigation, Writing &#x02013; review &#x00026; editing, Project administration. DS: Writing &#x02013; review &#x00026; editing, Funding acquisition, Conceptualization. CS: Writing &#x02013; review &#x00026; editing, Formal analysis, Supervision. LL: Project administration, Supervision, Writing &#x02013; review &#x00026; editing. YG: Funding acquisition, Supervision, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s10">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This research was funded by the Wenzhou Basic Scientific Research Project (Y20220454).</p>
</sec>
<ack><p>The authors would like to convey their profound appreciation to the editors and reviewers.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s11">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Atitallah</surname> <given-names>S. B.</given-names></name> <name><surname>Driss</surname> <given-names>M.</given-names></name> <name><surname>Boulila</surname> <given-names>W.</given-names></name> <name><surname>Koubaa</surname> <given-names>A.</given-names></name></person-group> (<year>2024</year>). <article-title>Enhancing early Alzheimer&#x00027;s disease detection through big data and ensemble few-shot learning</article-title>. <source>IEEE J. Biomed. Health Inf</source> . 1&#x02013;12. <pub-id pub-id-type="doi">10.1109/JBHI.2024.3473541</pub-id><pub-id pub-id-type="pmid">39356607</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bayta&#x0015F;</surname> <given-names>&#x00130;. M.</given-names></name></person-group> (<year>2024</year>). <article-title>Predicting progression from mild cognitive impairment to Alzheimer&#x00027;s dementia with adversarial attacks</article-title>. <source>IEEE J. Biomed. Health Inf</source>. <volume>28</volume>, <fpage>3750</fpage>&#x02013;<lpage>3761</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2024.3373703</pub-id><pub-id pub-id-type="pmid">38507374</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><collab>Dao D.-P. Yang H.-J. Kim J. Ho N.-H. the Alzheimer&#x00027;s Disease Neuroimaging Initiative</collab></person-group> (<year>2025</year>). <article-title>Longitudinal Alzheimer&#x00027;s disease progression prediction with modality uncertainty and optimization of information flow</article-title>. <source>IEEE J. Biomed. Health Inf</source>. <volume>29</volume>, <fpage>259</fpage>&#x02013;<lpage>272</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2024.3472462</pub-id><pub-id pub-id-type="pmid">39356605</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>T.</given-names></name> <name><surname>Xiang</surname> <given-names>D.</given-names></name> <name><surname>Schubert</surname> <given-names>K. E.</given-names></name> <name><surname>Dong</surname> <given-names>L.</given-names></name></person-group> (<year>2025</year>). <article-title>Gkan: Explainable diagnosis of Alzheimer&#x00027;s disease using graph neural network with Kolmogorov-Arnold networks</article-title>. <source>arXiv</source> [Preprint]. arXiv:2504.00946. <pub-id pub-id-type="doi">10.48550/arXiv.2504.00946</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dosovitskiy</surname> <given-names>A.</given-names></name> <name><surname>Beyer</surname> <given-names>L.</given-names></name> <name><surname>Kolesnikov</surname> <given-names>A.</given-names></name> <name><surname>Weissenborn</surname> <given-names>D.</given-names></name> <name><surname>Zhai</surname> <given-names>X.</given-names></name> <name><surname>Unterthiner</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>An image is worth 16x16 words: transformers for image recognition at scale</article-title>. <source>arXiv</source> [Preprint]. arXiv:2010.11929. <pub-id pub-id-type="doi">10.48550/arXiv.2010.11929</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Duenias</surname> <given-names>D.</given-names></name> <name><surname>Nichyporuk</surname> <given-names>B.</given-names></name> <name><surname>Arbel</surname> <given-names>T.</given-names></name> <name><surname>Raviv</surname> <given-names>T. R.</given-names></name></person-group> (<year>2025</year>). <article-title>Hyperfusion: a hypernetwork approach to multimodal integration of tabular and medical imaging data for predictive modeling</article-title>. <source>Med. Image Anal</source>. <volume>102</volume>:<fpage>103503</fpage>. <pub-id pub-id-type="doi">10.1016/j.media.2025.103503</pub-id><pub-id pub-id-type="pmid">40037055</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Elazab</surname> <given-names>A.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Abdelaziz</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Gu</surname> <given-names>J.</given-names></name> <name><surname>G&#x000F3;rriz</surname> <given-names>J. M.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Alzheimer&#x00027;s disease diagnosis from single and multimodal data using machine and deep learning models: achievements and future directions</article-title>. <source>Expert Syst. Appl</source>. <volume>255</volume>:<fpage>124780</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2024.124780</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>El-Gamal</surname> <given-names>F. E. A.</given-names></name> <name><surname>Elmogy</surname> <given-names>M.</given-names></name> <name><surname>Mahmoud</surname> <given-names>A.</given-names></name> <name><surname>Shalaby</surname> <given-names>A.</given-names></name> <name><surname>Switala</surname> <given-names>A. E.</given-names></name> <name><surname>Ghazal</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>A personalized computer-aided diagnosis system for mild cognitive impairment (MCI) using structural MRI (sMRI)</article-title>. <source>Sensors</source> <volume>21</volume>:<fpage>5416</fpage>. <pub-id pub-id-type="doi">10.3390/s21165416</pub-id><pub-id pub-id-type="pmid">34450858</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Eslamian</surname> <given-names>A.</given-names></name> <name><surname>Aghaei</surname> <given-names>A. A.</given-names></name> <name><surname>Cheng</surname> <given-names>Q.</given-names></name></person-group> (<year>2025</year>). <article-title>Tabkan: advancing tabular data analysis using Kolmogorov-Arnold network</article-title>. <source>arXiv</source> [Preprint]. arXiv:2504.06559. <pub-id pub-id-type="doi">10.48550/arXiv.2504.06559</pub-id></citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Francesconi</surname> <given-names>A.</given-names></name> <name><surname>di Biase</surname> <given-names>L.</given-names></name> <name><surname>Cappetta</surname> <given-names>D.</given-names></name> <name><surname>Rebecchi</surname> <given-names>F.</given-names></name> <name><surname>Soda</surname> <given-names>P.</given-names></name> <name><surname>Sicilia</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Class balancing diversity multimodal ensemble for Alzheimer&#x00027;s disease diagnosis and early detection</article-title>. <source>Comput. Med. Imaging Graph</source>. <volume>123</volume>:<fpage>102529</fpage>. <pub-id pub-id-type="doi">10.1016/j.compmedimag.2025.102529</pub-id><pub-id pub-id-type="pmid">40147216</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>W.</given-names></name> <name><surname>Gong</surname> <given-names>Z.</given-names></name> <name><surname>Deng</surname> <given-names>Z.</given-names></name> <name><surname>Rong</surname> <given-names>F.</given-names></name> <name><surname>Chen</surname> <given-names>C.</given-names></name> <name><surname>Ma</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Tabkanet: tabular data modeling with Kolmogorov-Arnold network and transformer</article-title>. <source>arXiv</source> [Preprint]. arXiv:2409.08806. <pub-id pub-id-type="doi">10.48550/arXiv.2409.08806</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gaser</surname> <given-names>C.</given-names></name> <name><surname>Franke</surname> <given-names>K.</given-names></name> <name><surname>Kl&#x000F6;ppel</surname> <given-names>S.</given-names></name> <name><surname>Koutsouleris</surname> <given-names>N.</given-names></name> <name><surname>Sauer</surname> <given-names>H.</given-names></name> <name><surname>Initiative</surname> <given-names>A. D. N.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Brainage in mild cognitive impaired patients: predicting the conversion to Alzheimer&#x00027;s disease</article-title>. <source>PLoS ONE</source> <volume>8</volume>:<fpage>e0067346</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0067346</pub-id><pub-id pub-id-type="pmid">23826273</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Gong</surname> <given-names>H.</given-names></name> <name><surname>Kang</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Wan</surname> <given-names>X.</given-names></name> <name><surname>Wu</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>&#x0201C;NNMAMBA: 3D biomedical image segmentation, classification and landmark detection with state space model,&#x0201D;</article-title> in <source>2025 IEEE 22nd International Symposium on Biomedical Imaging (ISBI)</source> (<publisher-loc>Houston, TX</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1109/ISBI60581.2025.10980694</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grigas</surname> <given-names>O.</given-names></name> <name><surname>Maskeliunas</surname> <given-names>R.</given-names></name> <name><surname>Dama&#x00161;evi&#x0010D;ius</surname> <given-names>R.</given-names></name></person-group> (<year>2024</year>). <article-title>Early detection of dementia using artificial intelligence and multimodal features with a focus on neuroimaging: a systematic literature review</article-title>. <source>Health Technol</source>. <volume>14</volume>, <fpage>201</fpage>&#x02013;<lpage>237</lpage>. <pub-id pub-id-type="doi">10.1007/s12553-024-00823-0</pub-id><pub-id pub-id-type="pmid">40286904</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gryshchuk</surname> <given-names>V.</given-names></name> <name><surname>Singh</surname> <given-names>D.</given-names></name> <name><surname>Teipel</surname> <given-names>S.</given-names></name> <name><surname>Dyrba</surname> <given-names>M.</given-names></name></person-group> (<year>2025</year>). <article-title>Contrastive self-supervised learning for neurodegenerative disorder classification</article-title>. <source>Front. Neuroinform</source>. <volume>19</volume>:<fpage>1527582</fpage>. <pub-id pub-id-type="doi">10.3389/fninf.2025.1527582</pub-id><pub-id pub-id-type="pmid">40034453</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Haq</surname> <given-names>E. U.</given-names></name> <name><surname>Yong</surname> <given-names>Q.</given-names></name> <name><surname>Yuan</surname> <given-names>Z.</given-names></name> <name><surname>Huarong</surname> <given-names>X.</given-names></name> <name><surname>Haq</surname> <given-names>R. U.</given-names></name></person-group> (<year>2025</year>). <article-title>Multimodal fusion diagnosis of the Alzheimer&#x00027;s disease via lightweight CNN-LSTM model using magnetic resonance imaging (MRI)</article-title>. <source>Biomed. Signal Process. Control</source> <volume>104</volume>:<fpage>107545</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2025.107545</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Deep residual learning for image recognition,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>770</fpage>&#x02013;<lpage>778</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>J.</given-names></name> <name><surname>Shen</surname> <given-names>L.</given-names></name> <name><surname>Sun</surname> <given-names>G.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Squeeze-and-excitation networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-loc>Salt Lake City, UT</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>7132</fpage>&#x02013;<lpage>7141</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2018.00745</pub-id></citation>
</ref>
<ref id="B19">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>X.</given-names></name> <name><surname>Shen</surname> <given-names>X.</given-names></name> <name><surname>Sun</surname> <given-names>Y.</given-names></name> <name><surname>Shan</surname> <given-names>X.</given-names></name> <name><surname>Min</surname> <given-names>W.</given-names></name> <name><surname>Su</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>&#x0201C;ITCFN: incomplete triple-modal co-attention fusion network for mild cognitive impairment conversion prediction,&#x0201D;</article-title> in <source>2025 IEEE 22nd International Symposium on Biomedical Imaging (ISBI)</source> (<publisher-loc>Houston, TX</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1109/ISBI60581.2025.10980706</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jabason</surname> <given-names>E.</given-names></name> <name><surname>Ahmad</surname> <given-names>M. O.</given-names></name> <name><surname>Swamy</surname> <given-names>M.</given-names></name></person-group> (<year>2025</year>). <article-title>A lightweight deep convolutional neural network extracting local and global contextual features for the classification of Alzheimer&#x00027;s disease using structural MRI</article-title>. <source>IEEE J. Biomed. Health Inf</source>. <volume>29</volume>, <fpage>2061</fpage>&#x02013;<lpage>2073</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2024.3512417</pub-id><pub-id pub-id-type="pmid">40030424</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kang</surname> <given-names>L.</given-names></name> <name><surname>Gong</surname> <given-names>H.</given-names></name> <name><surname>Wan</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Visual-attribute prompt learning for progressive mild cognitive impairment prediction,&#x0201D;</article-title> in <source>International Conference on Medical Image Computing and Computer-Assisted Intervention</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>547</fpage>&#x02013;<lpage>557</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-43904-9_53</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kohlbrenner</surname> <given-names>M.</given-names></name> <name><surname>Bauer</surname> <given-names>A.</given-names></name> <name><surname>Nakajima</surname> <given-names>S.</given-names></name> <name><surname>Binder</surname> <given-names>A.</given-names></name> <name><surname>Samek</surname> <given-names>W.</given-names></name> <name><surname>Lapuschkin</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>&#x0201C;Towards best practice in explaining neural network decisions with LRP,&#x0201D;</article-title> in <source>2020 International Joint Conference on Neural Networks (IJCNN)</source> (<publisher-loc>Glasgow</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1109/IJCNN48605.2020.9206975</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lei</surname> <given-names>Z.</given-names></name> <name><surname>Zhu</surname> <given-names>W.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Hua</surname> <given-names>C.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Shah</surname> <given-names>S. A. A.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>RLAD: a reliable hippo-guided multi-task model for Alzheimer&#x00027;s disease diagnosis</article-title>. <source>IEEE J. Biomed. Health Inf</source> . 1&#x02013;12. <pub-id pub-id-type="doi">10.1109/JBHI.2024.3412926</pub-id><pub-id pub-id-type="pmid">38861438</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>R.</given-names></name> <name><surname>Wu</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>A.</given-names></name> <name><surname>Wang</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>HFBSURV: hierarchical multimodal fusion with factorized bilinear models for cancer survival prediction</article-title>. <source>Bioinformatics</source> <volume>38</volume>, <fpage>2587</fpage>&#x02013;<lpage>2594</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btac113</pub-id><pub-id pub-id-type="pmid">35188177</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Zhao</surname> <given-names>X.</given-names></name> <name><surname>Xu</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Xing</surname> <given-names>C.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;IMF: interactive multimodal fusion model for link prediction,&#x0201D;</article-title> in <source>Proceedings of the ACM Web Conference 2023</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>ACM</publisher-name>), <fpage>2572</fpage>&#x02013;<lpage>2580</lpage>. <pub-id pub-id-type="doi">10.1145/3543507.3583554</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Ghahremani</surname> <given-names>M.</given-names></name> <name><surname>Wally</surname> <given-names>Y.</given-names></name> <name><surname>Wachinger</surname> <given-names>C.</given-names></name></person-group> (<year>2025</year>). <article-title>&#x0201C;Diamond: dementia diagnosis with multi-modal vision transformers using MRI and pet,&#x0201D;</article-title> in <source>2025 IEEE/CVF Winter Conference on Applications of Computer Vision (WACV)</source> (<publisher-loc>Tucson, AZ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>107</fpage>&#x02013;<lpage>116</lpage>. <pub-id pub-id-type="doi">10.1109/WACV61041.2025.00021</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>T.-Y.</given-names></name> <name><surname>Goyal</surname> <given-names>P.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Doll&#x000E1;r</surname> <given-names>P.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Focal loss for dense object detection,&#x0201D;</article-title> in <source>Proceedings of the IEEE International Conference on Computer Vision</source> (<publisher-loc>Venice</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>2980</fpage>&#x02013;<lpage>2988</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV.2017.324</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Peng</surname> <given-names>H.</given-names></name> <name><surname>Zheng</surname> <given-names>N.</given-names></name> <name><surname>Yang</surname> <given-names>Y.</given-names></name> <name><surname>Hu</surname> <given-names>H.</given-names></name> <name><surname>Yuan</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>&#x0201C;Efficientvit: memory efficient vision transformer with cascaded group attention,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-loc>Vancouver, BC</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>14420</fpage>&#x02013;<lpage>14430</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR52729.2023.01386</pub-id></citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Yue</surname> <given-names>L.</given-names></name> <name><surname>Xiao</surname> <given-names>S.</given-names></name> <name><surname>Yang</surname> <given-names>W.</given-names></name> <name><surname>Shen</surname> <given-names>D.</given-names></name> <name><surname>Liu</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Assessing clinical progression from subjective cognitive decline to mild cognitive impairment with incomplete multi-modal neuroimages</article-title>. <source>Med. Image Anal</source>. <volume>75</volume>:<fpage>102266</fpage>. <pub-id pub-id-type="doi">10.1016/j.media.2021.102266</pub-id><pub-id pub-id-type="pmid">34700245</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Vaidya</surname> <given-names>S.</given-names></name> <name><surname>Ruehle</surname> <given-names>F.</given-names></name> <name><surname>Halverson</surname> <given-names>J.</given-names></name> <name><surname>Solja&#x0010D;i&#x00107;</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Kan: Kolmogorov-Arnold networks</article-title>. <source>arXiv</source> [Preprint]. arXiv:2404.19756. <pub-id pub-id-type="doi">10.48550/arXiv.2404.19756</pub-id></citation>
</ref>
<ref id="B31">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Mathew</surname> <given-names>J.</given-names></name> <name><surname>Mekkayil</surname> <given-names>L.</given-names></name> <name><surname>Ramasangu</surname> <given-names>H.</given-names></name> <name><surname>Karthikeyan</surname> <given-names>B. R.</given-names></name> <name><surname>Manjunath</surname> <given-names>A. G.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Robust algorithm for early detection of Alzheimer&#x00027;s disease using multiple feature extractions,&#x0201D;</article-title> in <source>2016 IEEE Annual India Conference (INDICON)</source> (<publisher-loc>Bangalore</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1109/INDICON.2016.7839026</pub-id></citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Moradi</surname> <given-names>E.</given-names></name> <name><surname>Pepe</surname> <given-names>A.</given-names></name> <name><surname>Gaser</surname> <given-names>C.</given-names></name> <name><surname>Huttunen</surname> <given-names>H.</given-names></name> <name><surname>Tohka</surname> <given-names>J.</given-names></name></person-group> (<year>2015</year>). <article-title>Machine learning framework for early mri-based Alzheimer&#x00027;s conversion prediction in MCI subjects</article-title>. <source>Neuroimage</source> <volume>104</volume>, <fpage>398</fpage>&#x02013;<lpage>412</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2014.10.002</pub-id><pub-id pub-id-type="pmid">25312773</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mueller</surname> <given-names>S. G.</given-names></name> <name><surname>Weiner</surname> <given-names>M. W.</given-names></name> <name><surname>Thal</surname> <given-names>L. J.</given-names></name> <name><surname>Petersen</surname> <given-names>R. C.</given-names></name> <name><surname>Jack</surname> <given-names>C. R.</given-names></name> <name><surname>Jagust</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2005</year>). <article-title>Ways toward an early diagnosis in Alzheimer&#x00027;s disease: the Alzheimer&#x00027;s disease neuroimaging initiative (ADNI)</article-title>. <source>Alzheimers Dement</source>. <volume>1</volume>, <fpage>55</fpage>&#x02013;<lpage>66</lpage>. <pub-id pub-id-type="doi">10.1016/j.jalz.2005.06.003</pub-id><pub-id pub-id-type="pmid">17476317</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peng</surname> <given-names>L.</given-names></name> <name><surname>Cai</surname> <given-names>S.</given-names></name> <name><surname>Wu</surname> <given-names>Z.</given-names></name> <name><surname>Shang</surname> <given-names>H.</given-names></name> <name><surname>Zhu</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>MMGPL: multimodal medical data analysis with graph prompt learning</article-title>. <source>Med. Image Anal</source>. <volume>97</volume>:<fpage>103225</fpage>. <pub-id pub-id-type="doi">10.1016/j.media.2024.103225</pub-id><pub-id pub-id-type="pmid">38908306</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Poeta</surname> <given-names>E.</given-names></name> <name><surname>Giobergia</surname> <given-names>F.</given-names></name> <name><surname>Pastor</surname> <given-names>E.</given-names></name> <name><surname>Cerquitelli</surname> <given-names>T.</given-names></name> <name><surname>Baralis</surname> <given-names>E.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;A benchmarking study of Kolmogorov-Arnold networks on tabular data,&#x0201D;</article-title> in <source>In 2024 IEEE 18th International Conference on Application of Information and Communication Technologies (AICT)</source> (<publisher-loc>Turin</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1109/AICT61888.2024.10740444</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qiu</surname> <given-names>Z.</given-names></name> <name><surname>Yang</surname> <given-names>P.</given-names></name> <name><surname>Xiao</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Xiao</surname> <given-names>X.</given-names></name> <name><surname>Qin</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>3D multimodal fusion network with disease-induced joint learning for early Alzheimer&#x00027;s disease diagnosis</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>43</volume>, <fpage>3161</fpage>&#x02013;<lpage>3175</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2024.3386937</pub-id><pub-id pub-id-type="pmid">38607706</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Selvaraju</surname> <given-names>R. R.</given-names></name> <name><surname>Cogswell</surname> <given-names>M.</given-names></name> <name><surname>Das</surname> <given-names>A.</given-names></name> <name><surname>Vedantam</surname> <given-names>R.</given-names></name> <name><surname>Parikh</surname> <given-names>D.</given-names></name> <name><surname>Batra</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>&#x0201C;Grad-cam: visual explanations from deep networks via gradient-based localization,&#x0201D;</article-title> in <source>Proceedings of the IEEE International Conference on Computer Vision</source> (<publisher-loc>Venice</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>618</fpage>&#x02013;<lpage>626</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV.2017.74</pub-id></citation>
</ref>
<ref id="B38">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sundararajan</surname> <given-names>M.</given-names></name> <name><surname>Taly</surname> <given-names>A.</given-names></name> <name><surname>Yan</surname> <given-names>Q.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Axiomatic attribution for deep networks,&#x0201D;</article-title> in <source>International Conference on Machine Learning, Volume 70</source> (<publisher-loc>PMLR</publisher-loc>), <fpage>3319</fpage>&#x02013;<lpage>3328</lpage>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tan</surname> <given-names>M.</given-names></name> <name><surname>Le</surname> <given-names>Q. V.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Efficientnet: rethinking model scaling for convolutional neural networks,&#x0201D;</article-title> in <source>Proceedings of the 36th International Conference on Machine Learning, ICML 2019, 9-15 June 2019, Long Beach, California, USA, Volume 97 of Proceedings of Machine Learning Research</source>, eds. K. Chaudhuri, and R. Salakhutdinov (Long Beach, CA: PMLR), <fpage>6105</fpage>&#x02013;<lpage>6114</lpage>.<pub-id pub-id-type="pmid">35077359</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Verma</surname> <given-names>A.</given-names></name> <name><surname>Verma</surname> <given-names>A.</given-names></name> <name><surname>Sharma</surname> <given-names>R. P.</given-names></name></person-group> (<year>2025</year>). <source>VGG-KAN: A Hybrid Approach for Alzheimer&#x00027;s Disease Diagnosis using Kolmogorov Arnold Network</source>. <pub-id pub-id-type="doi">10.36227/techrxiv.174803834.41610666/v1</pub-id></citation>
</ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Lei</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>T.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Shan</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Hope: hybrid-granularity ordinal prototype learning for progression prediction of mild cognitive impairment</article-title>. <source>IEEE J. Biomed. Health Inf</source>. <volume>28</volume>, <fpage>6429</fpage>&#x02013;<lpage>6440</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2024.3357453</pub-id><pub-id pub-id-type="pmid">38261490</pub-id></citation></ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>D.</given-names></name> <name><surname>Honnorat</surname> <given-names>N.</given-names></name> <name><surname>Fox</surname> <given-names>P. T.</given-names></name> <name><surname>Ritter</surname> <given-names>K.</given-names></name> <name><surname>Eickhoff</surname> <given-names>S. B.</given-names></name> <name><surname>Seshadri</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Deep neural network heatmaps capture Alzheimer&#x00027;s disease patterns reported in a large meta-analysis of neuroimaging studies</article-title>. <source>Neuroimage</source> <volume>269</volume>:<fpage>119929</fpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2023.119929</pub-id><pub-id pub-id-type="pmid">36740029</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y.</given-names></name></person-group> (<year>2025</year>). <article-title>&#x0201C;Application of Kolmogorov-Arnold networks combined with graph convolutional networks in Alzheimer&#x00027;s disease diagnosis,&#x0201D;</article-title> in <source>2025 5th International Conference on Consumer Electronics and Computer Engineering (ICCECE)</source> (<publisher-loc>Dongguan</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>204</fpage>&#x02013;<lpage>207</lpage>. <pub-id pub-id-type="doi">10.1109/ICCECE65250.2025.10985338</pub-id></citation>
</ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wen</surname> <given-names>J.</given-names></name> <name><surname>Thibeau-Sutre</surname> <given-names>E.</given-names></name> <name><surname>Diaz-Melo</surname> <given-names>M.</given-names></name> <name><surname>Samper-Gonz&#x000E1;lez</surname> <given-names>J.</given-names></name> <name><surname>Routier</surname> <given-names>A.</given-names></name> <name><surname>Bottani</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Convolutional neural networks for classification of Alzheimer&#x00027;s disease: overview and reproducible evaluation</article-title>. <source>Med. Image Anal</source>. <volume>63</volume>:<fpage>101694</fpage>. <pub-id pub-id-type="doi">10.1016/j.media.2020.101694</pub-id><pub-id pub-id-type="pmid">32417716</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Woo</surname> <given-names>S.</given-names></name> <name><surname>Park</surname> <given-names>J.</given-names></name> <name><surname>Lee</surname> <given-names>J.-Y.</given-names></name> <name><surname>Kweon</surname> <given-names>I. S.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;CBAM: convolutional block attention module,&#x0201D;</article-title> in <source>Proceedings of the European Conference on Computer Vision (ECCV), Volume 11211</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>3</fpage>&#x02013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-01234-2_1</pub-id></citation>
</ref>
<ref id="B46">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>W.</given-names></name> <name><surname>Luo</surname> <given-names>M.</given-names></name> <name><surname>Zhou</surname> <given-names>P.</given-names></name> <name><surname>Si</surname> <given-names>C.</given-names></name> <name><surname>Zhou</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>&#x0201C;Metaformer is actually what you need for vision,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-loc>New Orleans, LA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>10819</fpage>&#x02013;<lpage>10829</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.01055</pub-id></citation>
</ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yun</surname> <given-names>S.</given-names></name> <name><surname>Choi</surname> <given-names>I.</given-names></name> <name><surname>Peng</surname> <given-names>J.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Bao</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Q.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Flex-MOE: modeling arbitrary modality combination via the flexible mixture-of-experts</article-title>. <source>arXiv</source> [Preprint]. arXiv:2410.08245. <pub-id pub-id-type="doi">10.48550/arXiv.2410.08245</pub-id></citation>
</ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Wu</surname> <given-names>X.</given-names></name> <name><surname>Tang</surname> <given-names>X.</given-names></name> <name><surname>Zhou</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name> <name><surname>Wu</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2025a</year>). <article-title>Asynchronous functional brain network construction with spatiotemporal transformer for MCI classification</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>44</volume>, <fpage>1168</fpage>&#x02013;<lpage>1180</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2024.3486086</pub-id><pub-id pub-id-type="pmid">39446548</pub-id></citation></ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Sun</surname> <given-names>K.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Xie</surname> <given-names>F.</given-names></name> <name><surname>Guo</surname> <given-names>Q.</given-names></name> <name><surname>Shen</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2025b</year>). <article-title>A modality-flexible framework for Alzheimer&#x00027;s disease diagnosis following clinical routine</article-title>. <source>IEEE J. Biomed. Health Inf</source>. <volume>29</volume>, <fpage>535</fpage>&#x02013;<lpage>546</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2024.3472011</pub-id><pub-id pub-id-type="pmid">39352829</pub-id></citation></ref>
</ref-list>
</back>
</article>