<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Cell Dev. Biol.</journal-id>
<journal-title>Frontiers in Cell and Developmental Biology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Cell Dev. Biol.</abbrev-journal-title>
<issn pub-type="epub">2296-634X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1608325</article-id>
<article-id pub-id-type="doi">10.3389/fcell.2025.1608325</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Cell and Developmental Biology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>MSLI-Net: retinal disease detection network based on multi-segment localization and multi-scale interaction</article-title>
<alt-title alt-title-type="left-running-head">Qi et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fcell.2025.1608325">10.3389/fcell.2025.1608325</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Qi</surname>
<given-names>Zhenjia</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3030152/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Hong</surname>
<given-names>Jin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2707232/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cheng</surname>
<given-names>Jilan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Long</surname>
<given-names>Guoli</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Hanyu</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Siyue</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3030216/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Cao</surname>
<given-names>Shuangliang</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3034243/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Information Engineering</institution>, <institution>Nanchang University</institution>, <addr-line>Nanchang</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Advanced Energy</institution>, <institution>Sun Yat-sen University</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Radiological Sciences</institution>, <institution>University of California Los Angeles</institution>, <addr-line>Los Angeles</addr-line>, <addr-line>CA</addr-line>, <country>United States</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Radiation Oncology, The Affiliated Cancer Hospital of Zhengzhou University &#x26; Henan Cancer Hospital</institution>, <addr-line>Zhengzhou</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1491541/overview">Yanwu Xu</ext-link>, Baidu, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1886483/overview">Shumao Pang</ext-link>, Guangzhou Medical University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3039196/overview">Jiaojiao Yu</ext-link>, Hubei University of Economics, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Jin Hong, <email>hongjin@ncu.edu.cn</email>; Shuangliang Cao, <email>csliangup@gmail.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>06</day>
<month>06</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>13</volume>
<elocation-id>1608325</elocation-id>
<history>
<date date-type="received">
<day>08</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>12</day>
<month>05</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Qi, Hong, Cheng, Long, Wang, Li and Cao.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Qi, Hong, Cheng, Long, Wang, Li and Cao</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>The retina plays a critical role in visual perception, yet lesions affecting it can lead to severe and irreversible visual impairment. Consequently, early diagnosis and precise identification of these retinal lesions are essential for slowing disease progression. Optical coherence tomography (OCT) stands out as a pivotal imaging modality in ophthalmology due to its exceptional performance, while the inherent complexity of retinal structures and significant noise interference present substantial challenges for both manual interpretation and AI-assisted diagnosis.</p>
</sec>
<sec>
<title>Methods</title>
<p>We propose MSLI-Net, a novel framework built upon the ResNet50 backbone, which enhances the global receptive field via a multi-scale dilation fusion module (MDF) to better capture long-range dependencies. Additionally, a multi-segmented lesion localization module (LLM) is integrated within each branch of a modified feature pyramid network (FPN) to effectively extract critical features while suppressing background noise through parallel branch refinement, and a wavelet subband spatial attention module (WSSA) is designed to significantly improve the model&#x2019;s overall performance in noise suppression by collaboratively processing and exchanging information between the low- and high-frequency subbands extracted through wavelet decomposition.</p>
</sec>
<sec>
<title>Results</title>
<p>Experimental evaluation on the OCT-C8 dataset demonstrates that MSLI-Net achieves 96.72% accuracy in retinopathy classification, underscoring its strong discriminative performance and promising potential for clinical application.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>This model provides new research ideas for the early diagnosis of retinal diseases and helps drive the development of future high-precision medical imaging-assisted diagnostic systems.</p>
</sec>
</abstract>
<kwd-group>
<kwd>retinal disease detection</kwd>
<kwd>multi-scale feature fusion</kwd>
<kwd>lesion localization</kwd>
<kwd>wavelet transform</kwd>
<kwd>noise suppression</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Molecular and Cellular Pathology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>The eye plays an indispensable role in how we perceive the world. Its retina, which primarily receives, adjusts, and relays visual stimuli from the environment, supplies the brain with essential visual information, serving as the core structure for our visual perception (<xref ref-type="bibr" rid="B12">Grossniklaus et al., 2015</xref>; <xref ref-type="bibr" rid="B28">Kermany et al., 2018</xref>). Consequently, retinal diseases are predisposed to causing severe visual impairment and even permanent blindness. At the same time, retinal diseases typically lack pronounced early clinical symptoms, and patients often fail to notice changes in their condition in time, missing the optimal window for intervention. Therefore, early diagnosis combined with high-precision detection is vital in slowing disease progression and minimizing visual impairment (<xref ref-type="bibr" rid="B39">Pennington and DeAngelis, 2016</xref>; <xref ref-type="bibr" rid="B41">Robinson, 2003</xref>). <xref ref-type="fig" rid="F1">Figure 1</xref> shows the OCT images of normal retina and seven common retinal diseases.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Eight categories of OCT images. <bold>(A)</bold> AMD, <bold>(B)</bold> CNV, <bold>(C)</bold> CSR, <bold>(D)</bold> DME, <bold>(E)</bold> DR, <bold>(F)</bold> DRUSEN, <bold>(G)</bold> MH, <bold>(H)</bold> NORMAL.</p>
</caption>
<graphic xlink:href="fcell-13-1608325-g001.tif"/>
</fig>
<p>As a non-contact, non-invasive imaging technique, optical coherence tomography (OCT) utilizes low-coherence interferometry to obtain high-resolution cross-sectional images of biological tissues. With its excellent imaging performance, OCT has become an indispensable diagnostic tool in ophthalmology, playing an increasingly important role in early screening, clinical diagnosis, and efficacy assessment of retinal diseases (<xref ref-type="bibr" rid="B22">Huang et al., 1991</xref>). Although this technology has significantly improved the diagnostic efficiency and accuracy of doctors in detecting related conditions, manual interpretation of retinal OCT images still faces considerable challenges in clinical settings. On one hand, as the incidence of retinal diseases continues to rise, the relative scarcity of specialized healthcare resources makes it challenging to meet the ever-growing demand for diagnosis and treatment. On the other hand, the identification of lesion features in OCT images relies heavily on the doctor&#x2019;s professional knowledge and clinical experience, rendering the diagnostic process highly subjective and potentially compromising diagnostic accuracy (<xref ref-type="bibr" rid="B52">Tsuji et al., 2020</xref>; <xref ref-type="bibr" rid="B13">He et al., 2023</xref>). In this context, the precise classification of OCT images to distinguish various types of retinal lesions has emerged as an indispensable component in the diagnosis of retinal diseases (<xref ref-type="bibr" rid="B29">Khalil et al., 2024</xref>). Therefore, the development of automated retinal image diagnosis systems is essential to assist clinicians in accurately detecting retinal pathologies.</p>
<p>In recent years, deep learning technology has made significant progress in both natural language processing and computer vision, which has promoted the development of various AI-driven diagnostic techniques. Among these, convolutional neural networks (CNN) have increasingly been applied in medical image analysis owing to their excellent feature extraction and pattern recognition capabilities (<xref ref-type="bibr" rid="B66">Zhang et al., 2024</xref>; <xref ref-type="bibr" rid="B65">Zhang et al., 2025</xref>; <xref ref-type="bibr" rid="B11">Gong et al., 2024</xref>; <xref ref-type="bibr" rid="B24">Ji et al., 2022</xref>; <xref ref-type="bibr" rid="B43">Simonyan and Zisserman, 2014</xref>). Numerous CNN-based models have been developed to address complex tasks such as disease detection (<xref ref-type="bibr" rid="B15">Hong et al., 2019b</xref>; <xref ref-type="bibr" rid="B43">Simonyan and Zisserman, 2014</xref>), image segmentation (<xref ref-type="bibr" rid="B19">Hong et al., 2022a</xref>; <xref ref-type="bibr" rid="B33">Li et al., 2024</xref>; <xref ref-type="bibr" rid="B20">Hong et al., 2022b</xref>), and classification (<xref ref-type="bibr" rid="B17">Hong et al., 2020a</xref>; <xref ref-type="bibr" rid="B18">Hong et al., 2020b</xref>; <xref ref-type="bibr" rid="B69">Zhu et al., 2022</xref>; <xref ref-type="bibr" rid="B53">Wan et al., 2024</xref>; <xref ref-type="bibr" rid="B63">Zhang et al., 2022</xref>), substantially enhancing both the automation and diagnostic efficiency in medical image processing. In the realm of ophthalmic image analysis, CNN have been extensively employed for OCT image classification and lesion detection, yielding noteworthy results (<xref ref-type="bibr" rid="B40">Qian et al., 2025</xref>; <xref ref-type="bibr" rid="B46">Subramanian et al., 2022</xref>; <xref ref-type="bibr" rid="B31">Laouarem et al., 2024</xref>; <xref ref-type="bibr" rid="B44">Song et al., 2025</xref>). For instance, Qian et al. enhanced the model&#x2019;s feature representation by fusing the outputs of multiple DenseBlocks based on DenseNet121 and replaced the positive and negative sample pairs in the conventional triplet loss with the class proxy concept. This modification enabled more efficient and accurate classification of retinal OCT images (<xref ref-type="bibr" rid="B40">Qian et al., 2025</xref>).</p>
<p>However, the inherent characteristics of OCT retinal images pose a significant challenge to the discriminative performance of existing models. On the one hand, because OCT is a grayscale imaging technique, subtle lesion features are not clear enough to be accurately identified. Moreover, there is a certain diversity in the shape, size and spatial distribution of lesion regions, which also increases the difficulty of lesion localization (<xref ref-type="bibr" rid="B5">Cheng et al., 2025</xref>; <xref ref-type="bibr" rid="B43">Simonyan and Zisserman, 2014</xref>; <xref ref-type="bibr" rid="B34">Lo et al., 2019</xref>). On the other hand, limited by the performance of imaging equipment and hardware conditions, OCT images are often accompanied by unavoidable noise interference during the acquisition process, resulting in many CNN-based models capturing a large amount of speckle noise information while learning lesion features, which makes the model unable to accurately distinguish between important and unimportant features, thus affecting the model&#x2019;s discriminative ability. In addition, the process of performing downsampling operations to extract high-level semantic features in CNN-based models often inevitably leads to the reduction of the spatial resolution of the image, resulting in the loss of some of the critical lesion information, which together with the residual noise interference weakens the model&#x2019;s discriminative ability for the lesion region (<xref ref-type="bibr" rid="B64">Zhang et al., 2019</xref>; <xref ref-type="bibr" rid="B67">Zhao et al., 2022</xref>; <xref ref-type="bibr" rid="B1">Alaba and Ball, 2024</xref>).</p>
<p>To address these challenges, this study proposes a retinal disease detection network (MSLI-Net) built upon multi-stage localization and multi-scale interaction. Initially, the network utilizes the residual blocks of ResNet50 for preliminary feature extraction from retinal images. It then employs a Multiscale Dilation Fusion Module (MDF) to enhance the feature representation across scales and expand the model&#x2019;s receptive field. Subsequently, a Multi-segmented Lesion Localization Fusion Module (LLM) is adopted to emphasize the lesion regions and suppress background noise. Finally, we introduce an MSA module (<xref ref-type="bibr" rid="B56">Xiao et al., 2023</xref>) and design a Wavelet Subband Spatial Attention Module (WSSA) to further refine the feature representations in the lesion regions, thereby achieving more precise disease detection. The contributions of this paper are as follows:<list list-type="simple">
<list-item>
<p>1) Our MSLI-Net is based on the ResNet50 network framework, which effectively integrates the MDF, LLM and WSSA, realizing the complementary advantages between shallow high-resolution features and deep semantic information. This enables the model to significantly enhance the representation of lesion regions in retinal OCT images, and effectively improves classification performance.</p>
</list-item>
<list-item>
<p>2) We design a multiscale dilation fusion module (MDF), which effectively extracts multiscale feature information by introducing convolutional branches with different dilation factors and deeply fuses it with original image features. It effectively enhances the global receptive field and improves the model&#x2019;s ability to model long-range dependencies.</p>
</list-item>
<list-item>
<p>3) We propose a multi-segmented lesion localization fusion module (LLM). By constructing multiple parallel branches, the LLM realizes the hierarchical extraction of local features as well as the enhancement of key channel features of a lesion. This design effectively mitigates the limitations of the traditional channel attention mechanism that is susceptible to interference in the context of complex noise while enhancing the accurate localization of the lesion region.</p>
</list-item>
<list-item>
<p>4) We develop the wavelet subband spatial attention module (WSSA) based on the introduction of the MSA module. This module decomposes the input features into four subbands of different frequencies by discrete wavelet transform, and realizes feature interaction and information fusion across subbands. The module is capable of extracting lesion-related features in greater detail while effectively suppressing noise interference.</p>
</list-item>
<list-item>
<p>5) We evaluate our model on the publicly available OCT-C8 dataset, achieving 96.72% accuracy in retinal OCT classification, demonstrating that this model has a strong discriminative capability in this domain.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s2">
<title>2 Related work</title>
<p>In recent years, the advancement of deep learning technology and its extensive application in medical image analysis have propelled research and produced significant results in fundus image analysis (<xref ref-type="bibr" rid="B68">Zheng et al., 2024</xref>; <xref ref-type="bibr" rid="B57">Xu et al., 2022a</xref>; <xref ref-type="bibr" rid="B58">Xu et al., 2022b</xref>). Early studies mainly focused on transfer learning and architecture optimization for classical convolutional neural networks (CNN). Wang et al. employed a transfer learning strategy by fine-tuning various classical CNN models (including VGG16, ResNet18, ResNet50, and InceptionV3) that were pre-trained on the ImageNet dataset, thereby achieving higher precision in retinal OCT image classification (<xref ref-type="bibr" rid="B54">Wang et al., 2019</xref>). Meanwhile, steady progress has been made in refining the model architecture itself, such as Karthik et al., who proposed Edgen blocks to replace the residual connection method in the traditional ResNet50 and designed a novel activation function to further enhance the network&#x2019;s ability to capture image boundary features and effectively highlight key lesion information (<xref ref-type="bibr" rid="B27">Karthik and Mahadevappa, 2023</xref>). Sunija et al. also designed OCTnet based on the ResNet50 architecture, achieving excellent classification performance while significantly reducing the number of model parameters (<xref ref-type="bibr" rid="B47">Sunija et al., 2021</xref>).</p>
<p>In addition to optimizing traditional CNN architectures, recent research has also focused on fusing CNN and Transformer architectures to further enhance the model&#x2019;s feature representation and global modeling capabilities. Laouarem et al. proposed a hybrid model, HTC-Retina, that combines the advantages of CNN in local feature extraction with the capability of a visual Transformer for global dependency modeling, effectively overcoming the limitations of a single architecture in image analysis (<xref ref-type="bibr" rid="B31">Laouarem et al., 2024</xref>). Similarly, the CRAT network mitigates the common attention collapse problem in deep Transformers by introducing the Re-Attention module to dynamically adjust the multi-head self-attention mechanism (<xref ref-type="bibr" rid="B60">Yang et al., 2025</xref>). Moreover, the introduction of the Swin Poly transformer network further broadens the research boundaries of fusion modeling, and its mechanism of establishing flexible connectivity between image regions significantly improves the model&#x2019;s ability to facilitate information exchange among multi-scale features (<xref ref-type="bibr" rid="B13">He et al., 2023</xref>).</p>
<p>In addition to integrating different model architectures, task-level co-design has emerged as a prominent research topic. Diao et al. proposed an innovative method that tightly integrates segmentation and classification tasks (<xref ref-type="bibr" rid="B6">Diao et al., 2023</xref>). This approach employs an auxiliary segmentation branch within the classification network (CM-CNN) to generate a complementary mask for the input image, which is subsequently used to enhance the original features and effectively guide the classification network to focus on the features of the lesion region, thereby improving classification performance. Moreover, the application of the Grad-CAM algorithm enables CM-CNN to generate a class activation map (CAM) that further assists the segmentation network (CAM-UNet) in refining its segmentation accuracy, ultimately achieving more precise feature extraction and segmentation of the lesion regions. This model exhibits excellent performance on both classification and segmentation tasks, demonstrating the potential of explicit information interaction between tasks in enhancing diagnostic performance.</p>
<sec id="s2-1">
<title>2.1 Image cropping and local feature extraction</title>
<p>It has been shown that the classification performance of deep learning models can be effectively improved by an appropriate cropping strategy for retinal OCT images (<xref ref-type="bibr" rid="B2">Awais et al., 2017</xref>). Some researchers have manually cropped out rectangular boxes containing lesion regions at the image preprocessing stage to help neural networks capture key lesion features more effectively (<xref ref-type="bibr" rid="B26">Kaothanthong et al., 2023</xref>). However, this cropping method is not only cumbersome and time-consuming, but also poses the risk of degrading the model performance by mistakenly deleting important lesion features. To address these issues, some studies have proposed dividing the OCT images into fixed-size patches and extracting features from each patch individually, thereby improving the model&#x2019;s ability to extract features from local lesion regions while preserving the overall spatial structure of the image (<xref ref-type="bibr" rid="B7">Dutta et al., 2023</xref>).</p>
<p>In other related studies, Sharma et al. proposed a network structure called AELGNet, which successfully achieved efficient capture of both subtle and global features of plant leaf images by partitioning the image feature map into four fixed patches and extracting local features using independent RSA and RCA mechanisms, respectively (<xref ref-type="bibr" rid="B42">Sharma and Vardhan, 2025</xref>). However, since most retinal lesions tend to be concentrated in a few localized regions of the image, dividing the patches in a fixed manner and indiscriminately extracting features not only wastes computational resources but also amplifies irrelevant background noise, thereby reducing the model&#x2019;s classification accuracy.</p>
<p>Unlike existing methods, our proposed LLM employs parallel cropping branches based on the characteristics of retinal OCT images, allowing the model to automatically locate the lesion region and extract key features, effectively mitigating interference from background noise and thereby improving the model&#x2019;s discriminative capacity and robustness.</p>
</sec>
<sec id="s2-2">
<title>2.2 Discrete wavelet transform</title>
<p>As a common method for image denoising, the discrete wavelet transform can decompose the signal into low-frequency subbands and high-frequency subbands (<xref ref-type="bibr" rid="B10">Gao and Yan, 2010</xref>). Specifically, the low-frequency subbands mainly retain the color and structural information of the image, while the high-frequency subbands preserve detailed features such as edges, textures, and high-frequency noise. This property renders the wavelet transform particularly advantageous in image processing (<xref ref-type="bibr" rid="B62">Yu et al., 2025</xref>; <xref ref-type="bibr" rid="B4">Burrus et al., 1998</xref>; <xref ref-type="bibr" rid="B59">Xu et al., 2020</xref>). Some researchers have attempted to eliminate the HH subband, where the noise is most concentrated, and have introduced an attention mechanism solely for the remaining three subbands, employing the wavelet transform as a downsampling operation to mitigate noise interference (<xref ref-type="bibr" rid="B67">Zhao et al., 2022</xref>). Alaba et al. strengthened the information of the LL subband by efficiently fusing the important features in the LH and HL subbands and passing them to the LL subband (<xref ref-type="bibr" rid="B1">Alaba and Ball, 2024</xref>). Finder et al. enhanced the model&#x2019;s receptive field by implementing multi-level wavelet decomposition and independently processing the LL subband (<xref ref-type="bibr" rid="B9">Finder et al., 2024</xref>). Although these methods have improved the performance of the wavelet transform in image analysis to some degree, most studies have focused only on low-frequency information or have achieved feature extraction and denoising at the expense of discarding high-frequency information, thereby limiting the comprehensive utilization of the potential information contained in all subbands.</p>
<p>To this end, we propose the WSSA, which synergistically processes all subbands while fully preserving all subband features, adaptively suppressing background noise and highlighting key edge and structural information. Unlike previous approaches that focus solely on information from a single subband, WSSA enables synergistic processing and information interaction between the low-frequency and high-frequency subbands, thereby more comprehensively enhancing the model&#x2019;s performance in noise suppression and lesion perception.</p>
</sec>
<sec id="s2-3">
<title>2.3 Dilated convolution</title>
<p>The convolutional kernel is a core component of convolutional neural networks (CNN), but when expanding the receptive field, traditional methods often require stacking multiple convolutional layers into a deep network, which significantly increases the number of parameters. To address this issue, Yu et al. proposed achieving an exponential expansion of the receptive field by introducing different dilation factors, so that the receptive field expands exponentially while parameters grow only linearly (<xref ref-type="bibr" rid="B61">Yu and Koltun, 2015</xref>). Based on this design concept, some researchers proposed parallel dilated convolution modules for more efficient image processing tasks (<xref ref-type="bibr" rid="B8">Feng et al., 2020</xref>; <xref ref-type="bibr" rid="B32">Li et al., 2019</xref>; <xref ref-type="bibr" rid="B3">Bui et al., 2024</xref>). For example, Kamran et al. replaced the traditional 3 &#xd7; 3 convolution with two parallel dilated convolutions with a dilation factor of 2 in the residual block to enhance the network&#x2019;s ability to model contextual information (<xref ref-type="bibr" rid="B25">Kamran et al., 2019</xref>), and Li et al. extracted spatial features from the feature map using a parallel dilated convolution module (<xref ref-type="bibr" rid="B32">Li et al., 2019</xref>). Although these designs expanded the receptive field, they did not fully consider the fusion mechanism with the original feature map, which resulted in the loss of local details. In contrast, the MDF designed in our work further strengthens fusion with the original feature map while employing dilated convolution to capture multi-scale features and enlarge the receptive field, thereby preserving local details and enhancing the model&#x2019;s ability to capture long-range dependencies. Experimental results show that the network using the MDF outperforms the traditional design employing a single dilated convolution module in terms of accuracy.</p>
</sec>
</sec>
<sec sec-type="methods" id="s3">
<title>3 Methods</title>
<sec id="s3-1">
<title>3.1 Overall framework</title>
<p>We propose a new network structure MSLI-Net as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, the overall architecture of MSLI-Net is composed of three core modules, namely, the multi-scale dilation fusion module (MDF), the multi-segmented lesion localization fusion module (LLM), and the wavelet subband spatial attention module (WSSA). MSLI-Net comprehensively extracts lesion features, effectively enhances focus on key pathological regions, and improves classification accuracy and discriminative performance for retinal OCT images.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Overall architecture of MSLI-Net.</p>
</caption>
<graphic xlink:href="fcell-13-1608325-g002.tif"/>
</fig>
<p>Specifically, we feed the image into the multi-scale dilation fusion module after initial feature extraction at various stages of the ResNet50 network. This module extracts multi-scale feature representations with enlarged receptive fields through dilated convolutions with different dilation factors and effectively fuses them with the original feature maps, thereby better incorporating both global semantic context and local detailed features present in the image.</p>
<p>On this basis, we introduced a feature pyramid network (FPN) structure removing the upsampling branches, used the MDF outputs as inputs for each FPN branch, and designed a Multi-segmented Lesion Localization Fusion Module to realize the refinement of the feature maps output from the MDF. In view of the inherent characteristics of retinal OCT images, we innovatively introduced the strategy of parallel cropping in this module to retain and extract feature information segment by segment, which effectively enhanced the localization ability of the lesion region.</p>
<p>Meanwhile, we introduce the MSA module and design the wavelet subband spatial attention module. Considering that when the feature map size is odd, the structural distortion may be caused by the wavelet transform and its inverse transform, the WSSA only process the feature maps produced by the LLM in the first three branches of the FPN. This module effectively enhances key edge information through inter-subband feature interaction and fusion, while suppressing irrelevant noise interference.</p>
<p>Finally, we perform global average pooling on the feature maps output from each of the four branches of the FPN to reduce the spatial dimensionality, and subsequently perform stacked fusion on them in terms of channel dimensions to achieve deep interaction and complementary information between features at different scales. We then feed the fused feature maps into a classifier for retinal image classification. Through this strategy, our network fully fuses the local detail information carried by the high-resolution shallow feature maps with the global semantic information expressed by the deep feature maps, thereby strengthening multi-scale contextual relevance and improving classification accuracy and model robustness.</p>
</sec>
<sec id="s3-2">
<title>3.2 Multi-scale dilation fusion module (MDF)</title>
<p>To obtain a larger receptive field without reducing the spatial resolution of the feature maps, Yu et al. proposed achieving this by introducing different dilation factors&#x2014;that is, by effectively increasing the spacing between values in the convolution kernel (<xref ref-type="bibr" rid="B61">Yu and Koltun, 2015</xref>; <xref ref-type="bibr" rid="B45">Song et al., 2024</xref>). However, due to its sparse sampling pattern resembling a checkerboard, it is prone to triggering the grid effect, which leads to the loss of local information and affects the completeness of feature expression (<xref ref-type="bibr" rid="B36">Mehta et al., 2018</xref>). In order to fuse global and local features more effectively, this paper proposes a multi-scale dilation fusion module (MDF). This module effectively improves the overall performance of the model by fully fusing the features extracted by the convolution with different dilation factors with the original features.</p>
<p>The structure of the MDF is shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. Let the input feature map be <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi mathvariant="italic">x</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="italic">R</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi mathvariant="italic">W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi mathvariant="italic">H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi mathvariant="italic">C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denote the width, height and number of channels of the feature map, respectively. First, MDF obtains the new intermediate feature map <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="italic">R</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="italic">C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> by channel compression of the input feature map <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mi mathvariant="italic">x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The computational process is shown in <xref ref-type="disp-formula" rid="e1">Equation 1</xref>.<disp-formula id="e1">
<mml:math id="m7">
<mml:mrow>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="italic">Relu</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="italic">BN</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">Conv</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="italic">x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Overall architecture of MDF.</p>
</caption>
<graphic xlink:href="fcell-13-1608325-g003.tif"/>
</fig>
<p>Subsequently, different dilation factors (6, 12, and 18) are used to perform convolution operations on <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:mi mathvariant="italic">F</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to extract multi-scale context features, which are denoted as <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="italic">R</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="italic">C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. To enhance the complementarity between features at different scales, each branch of the extracted feature <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is passed through a 1 <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 1 convolutional layer and a batch normalization (BN) layer, to unify the feature scales and adjust the weights to obtain <inline-formula id="inf11">
<mml:math id="m12">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, and at the same time, the same operation is performed on the feature map <inline-formula id="inf12">
<mml:math id="m13">
<mml:mrow>
<mml:mi mathvariant="italic">F</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to obtain <inline-formula id="inf13">
<mml:math id="m14">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> The computational process is shown in <xref ref-type="disp-formula" rid="e2">Equations 2</xref>, <xref ref-type="disp-formula" rid="e3">3</xref>.<disp-formula id="e2">
<mml:math id="m15">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>B</mml:mi>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
<disp-formula id="e3">
<mml:math id="m16">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="italic">BN</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">Conv</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="italic">F</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf14">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">Dc</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2013;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the inflated convolution with convolution kernel size 3 <inline-formula id="inf15">
<mml:math id="m18">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 3 and dilation factor <inline-formula id="inf16">
<mml:math id="m19">
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> used in branch <inline-formula id="inf17">
<mml:math id="m20">
<mml:mrow>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Subsequently, the branch features are fused and the ReLU activation function is introduced to enhance the nonlinear representation, and the fused feature map <inline-formula id="inf18">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mi mathvariant="italic">r</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is obtained. The computational procedure is shown in <xref ref-type="disp-formula" rid="e4">Equation 4</xref>.<disp-formula id="e4">
<mml:math id="m22">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mi mathvariant="italic">r</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="italic">Relu</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="italic">i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mn>3</mml:mn>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>Next, the fused features are linearly combined between channels by pointwise convolution, so as to further mine the feature relationships between channels and enrich the feature representation. Finally, the number of channels is reduced to the original dimension <inline-formula id="inf19">
<mml:math id="m23">
<mml:mrow>
<mml:mi mathvariant="italic">C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and summed elementwise with the input feature map <inline-formula id="inf20">
<mml:math id="m24">
<mml:mrow>
<mml:mi mathvariant="italic">x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to finalize the full fusion of features at different scales. The process is shown in <xref ref-type="disp-formula" rid="e5">Equation 5</xref>.<disp-formula id="e5">
<mml:math id="m25">
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Re</mml:mtext>
<mml:mi>l</mml:mi>
<mml:mi>u</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-3">
<title>3.3 Multi-segmented lesion localization fusion module (LLM)</title>
<p>Considering that, in OCT images, the retina typically appears as a horizontally elongated structure while lesions usually occupy only localized regions, we divided the feature map uniformly along the vertical axis into seven subregions and observed that the retinal region is primarily contained within four contiguous segments. To achieve accurate lesion localization, effective extraction of key features, and suppression of irrelevant background noise, this paper proposes a multi-segmented lesion localization fusion module (LLM), as illustrated in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Overall architecture of LLM.</p>
</caption>
<graphic xlink:href="fcell-13-1608325-g004.tif"/>
</fig>
<p>We assume that the original feature map is <inline-formula id="inf21">
<mml:math id="m26">
<mml:mrow>
<mml:mi mathvariant="italic">q</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="italic">R</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The LLM divides the feature map into seven subregions along (<inline-formula id="inf22">
<mml:math id="m27">
<mml:mrow>
<mml:mi mathvariant="italic">H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>). The process is shown in <xref ref-type="disp-formula" rid="e6">Equation 6</xref>.<disp-formula id="e6">
<mml:math id="m28">
<mml:mrow>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mn>4</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mn>5</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mn>6</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mn>7</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <inline-formula id="inf23">
<mml:math id="m29">
<mml:mrow>
<mml:mi mathvariant="italic">H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the height on a single channel of the original feature map and <inline-formula id="inf24">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">h</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the subregion divided along the height. Then the four consecutive subregions in the feature map are extracted sequentially from top to bottom in each of the four branches to form a subfeature map, i.e., <inline-formula id="inf25">
<mml:math id="m31">
<mml:mrow>
<mml:mi mathvariant="italic">q</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="italic">R</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The specific <inline-formula id="inf26">
<mml:math id="m32">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is shown in <xref ref-type="disp-formula" rid="e7">Equation 7</xref>.<disp-formula id="e7">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
<p>Subsequently, the SE channel attention mechanism is introduced to process the sub-feature maps of the above four parallel branches with channel-level features, which further enhances the key channel features in each branch, and yields <inline-formula id="inf27">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">Q</mml:mi>
<mml:mi mathvariant="italic">I</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="italic">R</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mi mathvariant="italic">I</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. In order to enable the model to more accurately identify which consecutive subregions the lesions are specifically located in, we restore the SE-processed sub-feature maps to their original sizes by re-stitching the sub-feature maps with the discarded portions, i.e., obtaining <inline-formula id="inf28">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">Q</mml:mi>
<mml:mi mathvariant="italic">OI</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="italic">R</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mi mathvariant="italic">OI</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, and unify the feature scales and adjust the weights through a 1 <inline-formula id="inf29">
<mml:math id="m36">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 1 convolutional layer and a batch normalization (BN) layer. In addition, in order to more fully realize the complementary advantages of global and local features, and avoid the situation that a small number of images may have incomplete feature extraction due to the local attention mechanism of the model, we additionally add a fifth branch, which directly performs channel-level feature extraction on the original feature map <inline-formula id="inf30">
<mml:math id="m37">
<mml:mrow>
<mml:mi mathvariant="normal">q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and undergoes unified feature scale adjustment and weight fusion operation with the four parallel cropping branches. The process is shown by <xref ref-type="disp-formula" rid="e8">Equations 8</xref>&#x2013;<xref ref-type="disp-formula" rid="e10">10</xref>.<disp-formula id="e8">
<mml:math id="m38">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="italic">Q</mml:mi>
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>B</mml:mi>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
<disp-formula id="e9">
<mml:math id="m39">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="italic">q</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>B</mml:mi>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
<disp-formula id="e10">
<mml:math id="m40">
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="italic">i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mn>4</mml:mn>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mi>q</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
<p>With this fusion approach, lesion localization is further enhanced, effectively reducing the susceptibility of the traditional channel-attention mechanism to complex background noise interference.</p>
</sec>
<sec id="s3-4">
<title>3.4 Wavelet subband spatial attention module (WSSA)</title>
<p>As an effective mathematical approach for addressing nonstationary signal decomposition, the wavelet transform can capture information at various frequencies and time positions by adjusting its scale and translation parameters, thereby reflecting the local variation characteristics of a signal. In the context of the commonly used two-dimensional discrete wavelet transform, the Haar wavelet decomposes the input feature map into four subbands via low-pass and high-pass filters, which correspond to the low-frequency subband (LL), horizontal high-frequency subband (LH), vertical high-frequency subband (HL), and diagonal high-frequency subband (HH). Among these, the low-frequency subband encapsulates the image&#x2019;s color and structural information, while the high-frequency subbands contain abundant detail and texture information. Subsequently, the signal is then reconstructed through the inverse wavelet transform. During reconstruction, wavelet-based edge detection is first applied to enhance edge features in each subband, then thresholding is performed to eliminate noise.</p>
<p>However, for retinal OCT images, due to their inherent speckle noise characteristics, it is difficult to effectively denoise them by simply using traditional wavelet transform methods. Therefore, to further suppress noise interference and highlight edge features, we propose the wavelet subband spatial attention module (WSSA) on the basis of the multiscale attention (MSA) module, which is structured as shown in <xref ref-type="fig" rid="F5">Figure 5A</xref>.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>
<bold>(A)</bold> Overall architecture of WSSA; <bold>(B)</bold> Architecture of MSA.</p>
</caption>
<graphic xlink:href="fcell-13-1608325-g005.tif"/>
</fig>
<p>In this module, we first use the wavelet transform to decompose the original feature map at multiple scales, and obtain four subbands containing low-frequency and high-frequency information, <inline-formula id="inf31">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">I</mml:mi>
<mml:mi mathvariant="italic">LL</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">I</mml:mi>
<mml:mi mathvariant="italic">LH</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">I</mml:mi>
<mml:mi mathvariant="italic">HL</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">I</mml:mi>
<mml:mi mathvariant="italic">HH</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="italic">R</mml:mi>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="italic">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="italic">H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. In order to realize more efficient collaborative modeling and information interaction between different frequency bands, we stack the four subbands along the channel dimensions, and construct the feature map, <inline-formula id="inf32">
<mml:math id="m42">
<mml:mrow>
<mml:mi mathvariant="italic">I</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="italic">R</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>4</mml:mn>
<mml:mi mathvariant="italic">C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The specific process is in <xref ref-type="disp-formula" rid="e11">Equation 11</xref>:<disp-formula id="e11">
<mml:math id="m43">
<mml:mrow>
<mml:mi mathvariant="italic">I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="italic">Concat</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">I</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">I</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">I</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">I</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>
</p>
<p>Next, we apply an average pooling operation (AvgPool) to the fused feature map <inline-formula id="inf33">
<mml:math id="m44">
<mml:mrow>
<mml:mi mathvariant="italic">I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to obtain the smoothed feature map <inline-formula id="inf34">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">I</mml:mi>
<mml:mi mathvariant="italic">a</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. To further enhance the global dependency modeling capability of the features, we introduce the Multihead Self-Attention Mechanism (MSA) to mine the long-distance dependencies and enhance the feature representation capability. <xref ref-type="fig" rid="F5">Figure 5B</xref> shows the MSA, and <xref ref-type="disp-formula" rid="e12">Equation 12</xref> gives its mathematical expression.<disp-formula id="e12">
<mml:math id="m46">
<mml:mrow>
<mml:mi mathvariant="italic">Att</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">Conv</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mn>3</mml:mn>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>h</mml:mi>
<mml:msub>
<mml:mi>i</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mn>5</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">I</mml:mi>
<mml:mi mathvariant="italic">a</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
</p>
<p>Where <inline-formula id="inf35">
<mml:math id="m47">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>h</mml:mi>
<mml:msub>
<mml:mi>i</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the four feed-forward paths illustrated in <xref ref-type="fig" rid="F5">Figure 5B</xref>, and <inline-formula id="inf36">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">DConv</mml:mi>
<mml:mrow>
<mml:mn>5</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes a depthwise convolution with a 5 <inline-formula id="inf37">
<mml:math id="m49">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 5 kernel (<xref ref-type="bibr" rid="B56">Xiao et al., 2023</xref>). The feature map <inline-formula id="inf38">
<mml:math id="m50">
<mml:mi mathvariant="italic">Att</mml:mi>
</mml:math>
</inline-formula> obtained after processing by this module is then used to generate the attention weight map <inline-formula id="inf39">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mi mathvariant="italic">q</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="italic">R</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>4</mml:mn>
<mml:mi mathvariant="italic">C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> by the sigmoid activation function. Subsequently, we re-divide <inline-formula id="inf40">
<mml:math id="m52">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mi mathvariant="italic">q</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> along the channel dimension into four sub-modules <inline-formula id="inf41">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mi mathvariant="italic">qi</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="italic">R</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. We then multiply each <inline-formula id="inf42">
<mml:math id="m54">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mtext>qi</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with its corresponding initial wavelet subband and perform inverse wavelet transform to obtain the output feature map. The specific process is in <xref ref-type="disp-formula" rid="e13">Equation 13</xref>:<disp-formula id="e13">
<mml:math id="m55">
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">q</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
</p>
<p>This module enables the global structural information embedded in the low-frequency subbands to effectively guide the recognition of edge details in the high-frequency subbands, effectively suppressing noise interference. At the same time, the fine-grained edge features captured by the high-frequency subbands feed back to the low-frequency subbands, enhancing their ability to perceive the edge region.</p>
</sec>
</sec>
<sec sec-type="results|discussion" id="s4">
<title>4 Results and discussion</title>
<sec id="s4-1">
<title>4.1 Datasets</title>
<p>We use the publicly available OCT-C8 dataset (<xref ref-type="bibr" rid="B37">Obuli, 2021</xref>) to evaluate the performance of the model proposed in this paper. The dataset contains a total of 24,000 optical coherence tomography (OCT) images of seven types of retinal diseases as well as normal retina: Age-related Macular Degeneration (AMD), Choroidal Neovascularization (CNV), Central Serous Retinopathy (CSR), Diabetic Macular Edema (DME), Diabetic Retinopathy (DR), Yellow deposits under the retina (Drusen), Macular Hole (MH), and Healthy eyes with no abnormalities (NORMAL). Each category contains 3,000 images. The official data split is 2,300 images for training, 350 for validation, and 350 for testing. Considering that increasing the training sample size can improve the model generalization ability, in this paper, the original training set and the validation set are combined as the training set (2,650 images) for model training, and the test set (350 images) remains unchanged.</p>
</sec>
<sec id="s4-2">
<title>4.2 Evaluation metrics</title>
<p>To evaluate the effectiveness of the model, we use Accuracy (ACC), Precision, Sensitivity, and F1-score as the classification metrics. The formulas for these metrics are as follows.<disp-formula id="e14">
<mml:math id="m56">
<mml:mrow>
<mml:mi mathvariant="italic">Accuracy</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>
<disp-formula id="e15">
<mml:math id="m57">
<mml:mrow>
<mml:mi mathvariant="italic">Precision</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mi mathvariant="italic">TP</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>
<disp-formula id="e16">
<mml:math id="m58">
<mml:mrow>
<mml:mi mathvariant="italic">Sensitivity</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
<disp-formula id="e17">
<mml:math id="m59">
<mml:mrow>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">Precision</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">Sensitivity</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">Precision</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="italic">Sensitivity</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>where <inline-formula id="inf43">
<mml:math id="m60">
<mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the number of samples in the test set whose classifier prediction results match the true labels, and <inline-formula id="inf44">
<mml:math id="m61">
<mml:mrow>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the total number of samples in the test set. In addition, the symbols <inline-formula id="inf45">
<mml:math id="m62">
<mml:mtext>TP</mml:mtext>
</mml:math>
</inline-formula>, <inline-formula id="inf46">
<mml:math id="m63">
<mml:mtext>FP</mml:mtext>
</mml:math>
</inline-formula>, and <inline-formula id="inf47">
<mml:math id="m64">
<mml:mtext>FN</mml:mtext>
</mml:math>
</inline-formula> used in <xref ref-type="disp-formula" rid="e14">Equations 14</xref>&#x2013;<xref ref-type="disp-formula" rid="e17">17</xref> denote the number of samples that are true positive (both the actual label and the classification result are in the positive class), false positives (the true label is in the negative class while the classifier predicts the positive class), and false negatives (the true label is in the positive class but the classifier predicts the negative class), respectively.</p>
</sec>
<sec id="s4-3">
<title>4.3 Implementation details</title>
<p>The training and testing of this experiment were done on a single NVIDIA RTX 4090 GPU. In the data preprocessing stage, we uniformly resize the input feature map to 224 &#xd7; 224 pixels and normalize the image using pre-calculated mean and standard deviation. During the training process, the loss function was chosen to be cross-entropy loss and the model was optimized using the Adam optimizer, where the optimizer parameters were set to <inline-formula id="inf48">
<mml:math id="m65">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.9</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf49">
<mml:math id="m66">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.999</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. The weight decay parameter was set to <inline-formula id="inf50">
<mml:math id="m67">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> to reduce the risk of overfitting. In addition, in this study, the learning rate was fixed to 0.001, the batch size was set to 64, and trained for 60 epochs. In order to improve the training efficiency and reduce the memory consumption, the mixed-precision training technique provided by PyTorch, i.e., autocast and GradScaler, is used in the experimental process. In the performance evaluation of the model, the model weights at the 60th epoch were used for testing, and the experiments are repeated independently under the same experimental conditions for six times, and the average of the results of the six experiments is taken as the performance metrics of the model. The average of the six experimental results was finally taken as the model performance index.</p>
</sec>
<sec id="s4-4">
<title>4.4 Performance of our proposed method</title>
<p>In this section, we evaluate the classification performance of the proposed MSLI-Net model on the OCT-C8 retinal image dataset and analyze it in comparison with several representative convolutional neural network architectures, including ResNet50 (<xref ref-type="bibr" rid="B14">He et al., 2016</xref>), VGG16 (<xref ref-type="bibr" rid="B43">Simonyan and Zisserman, 2014</xref>), GoogLeNet (<xref ref-type="bibr" rid="B48">Szegedy et al., 2015</xref>), InceptionV3 (<xref ref-type="bibr" rid="B49">Szegedy et al., 2016</xref>), DenseNet121 (<xref ref-type="bibr" rid="B30">KQ, 2018</xref>) and EfficientNetB3 (<xref ref-type="bibr" rid="B51">Tan and Le, 2019</xref>), among others. In addition to the classical architectures, we also compare them with some of the models that have performed well in the retinal OCT image analysis task in recent years, including CTransCNN (<xref ref-type="bibr" rid="B55">Wu et al., 2023</xref>), MedViT (<xref ref-type="bibr" rid="B35">Manzari et al., 2023</xref>) and MRVM (<xref ref-type="bibr" rid="B70">Zuo et al., 2024</xref>). According to the experimental results shown in <xref ref-type="table" rid="T1">Table 1</xref>, DenseNet121 has the highest accuracy of 96.41% on the OCT-C8 dataset among the compared baseline models, followed by the MRVM model with an accuracy of 96.20%. Our MSLI-Net achieves a classification accuracy of 96.72% and outperformed the other compared models in all metrics. This demonstrates clear superiority in performance.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Performance comparison of the OCT-C8 dataset (%).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Method</th>
<th align="center">Accuracy</th>
<th align="center">Precision</th>
<th align="center">Sensitivity</th>
<th align="center">F1</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">ResNet50 (<xref ref-type="bibr" rid="B14">He et al., 2016</xref>)</td>
<td align="center">95.08</td>
<td align="center">95.34</td>
<td align="center">95.08</td>
<td align="center">95.09</td>
</tr>
<tr>
<td align="center">VGG16 (<xref ref-type="bibr" rid="B43">Simonyan and Zisserman, 2014</xref>)</td>
<td align="center">95.69</td>
<td align="center">95.77</td>
<td align="center">95.69</td>
<td align="center">95.69</td>
</tr>
<tr>
<td align="center">GoogLeNet (<xref ref-type="bibr" rid="B48">Szegedy et al., 2015</xref>)</td>
<td align="center">95.86</td>
<td align="center">96.00</td>
<td align="center">95.86</td>
<td align="center">95.84</td>
</tr>
<tr>
<td align="center">InceptionV3 (<xref ref-type="bibr" rid="B49">Szegedy et al., 2016</xref>)</td>
<td align="center">89.72</td>
<td align="center">90.96</td>
<td align="center">89.72</td>
<td align="center">89.76</td>
</tr>
<tr>
<td align="center">DenseNet121 (<xref ref-type="bibr" rid="B30">KQ, 2018</xref>)</td>
<td align="center">96.41</td>
<td align="center">96.46</td>
<td align="center">96.41</td>
<td align="center">96.40</td>
</tr>
<tr>
<td align="center">EfficientNetb3 (<xref ref-type="bibr" rid="B51">Tan and Le, 2019</xref>)</td>
<td align="center">92.57</td>
<td align="center">93.13</td>
<td align="center">92.57</td>
<td align="center">92.52</td>
</tr>
<tr>
<td align="center">CTransCNN (<xref ref-type="bibr" rid="B55">Wu et al., 2023</xref>)</td>
<td align="center">94.69</td>
<td align="center">94.69</td>
<td align="center">94.69</td>
<td align="center">94.94</td>
</tr>
<tr>
<td align="center">MedViT (<xref ref-type="bibr" rid="B35">Manzari et al., 2023</xref>)</td>
<td align="center">95.96</td>
<td align="center">95.96</td>
<td align="center">95.96</td>
<td align="center">95.95</td>
</tr>
<tr>
<td align="center">MRVM (<xref ref-type="bibr" rid="B70">Zuo et al., 2024</xref>)</td>
<td align="center">96.20</td>
<td align="center">96.21</td>
<td align="center">96.21</td>
<td align="center">96.19</td>
</tr>
<tr>
<td align="center">MSLI-Net (Ours)</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>96.75</bold>
</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>96.72</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best result under each evaluation metric.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>
<xref ref-type="fig" rid="F6">Figure 6</xref> shows the training metrics for MSLI-Net, where the left graph shows the training-accuracy curve and the right graph shows the training-loss curve. It can be observed that both curves eventually stabilize without significant overfitting. <xref ref-type="fig" rid="F7">Figure 7</xref> further shows the confusion matrix obtained from one representative experiment. As can be seen, our model achieves 100% classification accuracy on AMD and DR categories, and relatively lower classification accuracy on CNV, DME and DRUSEN, but still maintains a high overall level. These results fully demonstrate the strong generalization capability of MSLI-Net in the task of automatic classification of multi-category retinal OCT images, further validating the effectiveness of the proposed method.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Accuracy and loss during the training process.</p>
</caption>
<graphic xlink:href="fcell-13-1608325-g006.tif"/>
</fig>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>The confusion matrix of the results, with an accuracy of 96.93%.</p>
</caption>
<graphic xlink:href="fcell-13-1608325-g007.tif"/>
</fig>
</sec>
<sec id="s4-5">
<title>4.5 Ablation study</title>
<p>To evaluate the contribution of each module in the proposed model to the overall performance, we conducted systematic ablation experiments on the OCT-C8 dataset, the results of which were shown in <xref ref-type="table" rid="T2">Table 2</xref>. The accuracy was 95.08% when using ResNet50 alone, which we adopted as our baseline. We first built the ResNet50&#x2b;MDF architecture by adding the multi-scale dilation fusion module (MDF) to each stage of ResNet50, at which point the model accuracy was improved to 96.04%, an improvement of about 1% from the baseline. Next, we introduced the feature pyramid network (FPN) that removes the up-sampling branches, and added the multi-segmented lesion localization fusion module (LLM) on top of it to form the FPN-ResNet50&#x2b;MDF &#x2b; LLM architecture. The results showed that this combination further improves the model accuracy to 96.46%. Finally, we incorporated the Wavelet Subband Spatial Attention module (WSSA) into the first three FPN branches to form the full MSLI-Net; this achieved 96.72% accuracy. These results confirm that each module synergistically enhances overall performance.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Ablation experiment results on OCT-C8 dataset (%).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Method</th>
<th align="center">Accuracy</th>
<th align="center">Precision</th>
<th align="center">Sensitivity</th>
<th align="center">F1</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">ResNet50</td>
<td align="center">95.08</td>
<td align="center">95.34</td>
<td align="center">95.08</td>
<td align="center">95.09</td>
</tr>
<tr>
<td align="center">ResNet50&#x2b;MDF</td>
<td align="center">96.04</td>
<td align="center">96.09</td>
<td align="center">96.04</td>
<td align="center">96.04</td>
</tr>
<tr>
<td align="center">FPN-ResNet50&#x2b;MDF &#x2b; LLM</td>
<td align="center">96.46</td>
<td align="center">96.51</td>
<td align="center">96.46</td>
<td align="center">96.46</td>
</tr>
<tr>
<td align="center">FPN-ResNet50&#x2b;LLM &#x2b; WSSA</td>
<td align="center">95.45</td>
<td align="center">95.61</td>
<td align="center">95.45</td>
<td align="center">95.44</td>
</tr>
<tr>
<td align="center">FPN-ResNet50&#x2b;MDF &#x2b; WSSA</td>
<td align="center">95.79</td>
<td align="center">95.89</td>
<td align="center">95.79</td>
<td align="center">95.78</td>
</tr>
<tr>
<td align="center">MSLI-Net (Ours)</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>96.75</bold>
</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>96.72</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best result under each evaluation metric.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>To further verify the impact of each module on the overall performance, we removed MDF (FPN-ResNet50&#x2b;LLM &#x2b; WSSA) and LLM (FPN-ResNet50&#x2b;MDF &#x2b; WSSA) from the full model, respectively, and analyzed their performance in comparison with the complete MSLI-Net model. The experimental results show that after removing MDF and LLM, the classification accuracy of the model is 95.45% and 95.79%, respectively, both of which show a decrease compared with the complete structure. This verifies the key role of each module in the performance improvement. The result further demonstrate that there is a close synergistic dependency between the modules, and the absence of any sub-module will weaken the discriminative ability of the model, thus affecting the overall performance.</p>
<p>In order to verify the effectiveness of the multi-scale dilation fusion module (MDF) proposed in this paper, we reproduced three representative dilated convolution modules and individually replaced the MDF with each of them for comparative experiments. Specifically, we reproduced the proposed dilated feature enhancement module (DFE) designed by <xref ref-type="bibr" rid="B3">Bui et al. (2024)</xref>; the Multi-scale Context Block (MSCB) proposed by <xref ref-type="bibr" rid="B38">Peng et al. (2023)</xref>; and the ASPP module (<xref ref-type="bibr" rid="B34">Lo et al., 2019</xref>) used by Lo et al. in their work. The experimental results are shown in <xref ref-type="table" rid="T3">Table 3</xref>, where the model accuracy reached 96.42% when the ASPP was used instead of MDF, 96.20% when MSCB was used, and only 96.08% when MDF was replaced by the DFE module. In contrast, using our proposed MDF within the same network architecture, the model achieved an accuracy of 96.72%. Moreover, our model contained 89.69 million parameters and 33.46 GFLOPs&#x2014;an increase relative to the MSCB model (46.13 million, 20.58 GFLOPs) but still far smaller than both the DFE (206.67 million, 67.98 GFLOPs) and the ASPP (228.94 million, 72.94 GFLOPs) counterparts. These results demonstrate that our MDF effectively enhances feature extraction and semantic understanding performance while maintaining a lightweight architecture.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Comparison of different inflated convolutional modules on OCT-C8 dataset (%).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Method</th>
<th align="center">Accuracy</th>
<th align="center">Precision</th>
<th align="center">Sensitivity</th>
<th align="center">F1</th>
<th align="center">Params(M)</th>
<th align="center">FLOPs (GFLOPs)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">DFE (<xref ref-type="bibr" rid="B3">Bui et al., 2024</xref>)</td>
<td align="center">96.08</td>
<td align="center">96.23</td>
<td align="center">96.08</td>
<td align="center">96.08</td>
<td align="center">206.67</td>
<td align="center">67.98</td>
</tr>
<tr>
<td align="center">MSCB (<xref ref-type="bibr" rid="B38">Peng et al., 2023</xref>)</td>
<td align="center">96.20</td>
<td align="center">96.26</td>
<td align="center">96.20</td>
<td align="center">96.20</td>
<td align="center">46.13</td>
<td align="center">20.58</td>
</tr>
<tr>
<td align="center">ASPP (<xref ref-type="bibr" rid="B34">Lo et al., 2019</xref>)</td>
<td align="center">96.42</td>
<td align="center">96.48</td>
<td align="center">96.42</td>
<td align="center">96.41</td>
<td align="center">228.94</td>
<td align="center">72.94</td>
</tr>
<tr>
<td align="center">MDF(Ours)</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>96.75</bold>
</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>89.69</bold>
</td>
<td align="center">
<bold>33.46</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best result under each evaluation metric.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>In addition, to evaluate the multi-segmented lesion localization fusion module (LLM), we devised two comparison schemes: one did not introduce a cropping strategy at all and only used the SE channel attention mechanism (<xref ref-type="bibr" rid="B21">Hu et al., 2018</xref>) for feature processing; the other used the cropping strategy proposed by <xref ref-type="bibr" rid="B42">Sharma and Vardhan (2025)</xref> in the AELGNet model, i.e., to divided the feature map into four patches using a four-quadrant partitioning strategy, and then applied the SE channel attention mechanism to each patch. The comparison results were shown in <xref ref-type="table" rid="T4">Table 4</xref>, when only the SE channel attention mechanism was used, the model achieved an accuracy of 96.17%. The accuracy dropped to 95.81% when the AELGNet cropping strategy was used, which may have been due to the fact that retinal structures in OCT images are usually distributed in long horizontal strips, and some patches may contain only background information when dividing the feature map with this cropping strategy. Meanwhile, the four patches were processed indiscriminately during the feature extraction process, which led to the amplification of the interference of irrelevant noise in the background region, and ultimately reduced the effectiveness of the model feature extraction. In contrast, our proposed LLM enabled the model to pay more attention to the features in the retinal region, and as shown in the experimental results in <xref ref-type="table" rid="T4">Table 4</xref>, the classification accuracy and various indexes of the model when using the LLM were significantly better than those of the comparative methods using the SE module and adopting the cropping strategy in AELGNet, thus verifying the effectiveness of the LLM in the retinal OCT image classification task.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Comparison of different cropping strategies on OCT-C8 dataset (%).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Method</th>
<th align="center">Accuracy</th>
<th align="center">Precision</th>
<th align="center">Sensitivity</th>
<th align="center">F1</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Cropping Strategy in AELGNet (<xref ref-type="bibr" rid="B42">Sharma and Vardhan, 2025</xref>)</td>
<td align="center">95.81</td>
<td align="center">95.93</td>
<td align="center">95.81</td>
<td align="center">95.80</td>
</tr>
<tr>
<td align="center">SE (<xref ref-type="bibr" rid="B21">Hu et al., 2018</xref>)</td>
<td align="center">96.17</td>
<td align="center">96.28</td>
<td align="center">96.17</td>
<td align="center">96.17</td>
</tr>
<tr>
<td align="center">LLM(Ours)</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>96.75</bold>
</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>96.72</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best result under each evaluation metric.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>To verify the effectiveness of the proposed wavelet subband spatial attention module (WSSA), we designed two sets of comparison experiments. In the first set of experiments, we replaced the WSSA module, in turn, with the following four methods: (1) using the wavelet transform and its inverse transform (OWT) (<xref ref-type="bibr" rid="B50">Talukder and Harada, 2010</xref>) without any processing; (2) adopting the WTConv (<xref ref-type="bibr" rid="B9">Finder et al., 2024</xref>) proposed by Finder et al. which involves a convolutional operation for each wavelet subband individually; (3) replicating the WCAM proposed by <xref ref-type="bibr" rid="B1">Alaba and Ball (2024)</xref>, which is processed by fusing the features of the LH and HL subbands to the LL subband; (4) the MSA (<xref ref-type="bibr" rid="B56">Xiao et al., 2023</xref>) applying independently to each wavelet subband, constituting the OWT &#x2b; MSA.</p>
<p>In addition, our WSSA module stacks all wavelet subbands, extracts subband weights via global average pooling, and then refines these weights using the MSA module. These weights are then multiplied with the original subband features to facilitate inter-subband information interaction. Finally, we perform the inverse wavelet transform to restore the image size. Therefore, we further designed a second set of comparative experiments to comprehensively evaluate the advantages of the WSSA. Specifically, without altering the remaining process, we independently excluded each of the LL, LH, HL, and HH subbands&#x2014;resulting in four modules referred to as w/o LL, w/o LH, w/o HL, and w/o HH&#x2014;in which only the remaining three subbands are stacked and processed. This setup allows us to analyze the role of each subband in the process of information fusion.</p>
<p>
<xref ref-type="table" rid="T5">Table 5</xref> shows the performance comparison results of models using different modules in the first set of experiments for the retinal OCT image classification task. The results show that the classification accuracy of the model using OWT is only 95.92%, indicating that although the wavelet transform possesses some image processing capability, the lack of subsequent feature extraction may lead to the disruption of intrinsic structure, which affects the classification performance. Further, the models using WTconv and OWT &#x2b; MSA obtain classification accuracies of 96.04% and 96.23%, respectively, indicating that there is a close intrinsic correlation between the wavelet subbands, and it is difficult to effectively tap the potential complementary information of each subband by only performing independent feature extraction, thus restricting the enhancement of the model&#x2019;s discriminative ability. In contrast, the classification accuracy of the model using WCAM that fuses the LH and HL with the LL subband features is 96.44%, which verifies that the information interaction between the subbands helps to fully mine the feature information.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Comparison of the first set of wavelet strategies on the OCT-C8 dataset (%).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Method</th>
<th align="center">Accuracy</th>
<th align="center">Precision</th>
<th align="center">Sensitivity</th>
<th align="center">F1</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">OWT (<xref ref-type="bibr" rid="B50">Talukder and Harada, 2010</xref>)</td>
<td align="center">95.92</td>
<td align="center">96.01</td>
<td align="center">95.92</td>
<td align="center">95.92</td>
</tr>
<tr>
<td align="center">WTConv (<xref ref-type="bibr" rid="B9">Finder et al., 2024</xref>)</td>
<td align="center">96.04</td>
<td align="center">96.17</td>
<td align="center">96.04</td>
<td align="center">96.03</td>
</tr>
<tr>
<td align="center">WCAM (<xref ref-type="bibr" rid="B1">Alaba and Ball, 2024</xref>)</td>
<td align="center">96.44</td>
<td align="center">96.48</td>
<td align="center">96.44</td>
<td align="center">96.43</td>
</tr>
<tr>
<td align="center">OWT &#x2b; MSA</td>
<td align="center">96.23</td>
<td align="center">96.32</td>
<td align="center">96.23</td>
<td align="center">96.22</td>
</tr>
<tr>
<td align="center">WSSA (Ours)</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>96.75</bold>
</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>96.72</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best result under each evaluation metric.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>
<xref ref-type="table" rid="T6">Table 6</xref> displays the results of the second set of comparative experiments. It can be seen that the model accuracy using the second set of comparison methods (w/o LL, w/o LH, w/o HL, and w/o HH) is distributed between 96.12% and 96.31%, highlighting that the synergistic effect of each wavelet subband in feature extraction is indispensable. The model accuracy reaches 96.72% when using our proposed WSSA. This suggests that omitting any sub-band may degrade feature representation, thereby impairing the model&#x2019;s overall discriminative performance. The effectiveness of WSSA in the retinal OCT image classification task is also demonstrated.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Comparison of the second set of wavelet strategies on the OCT-C8 dataset (%).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Method</th>
<th align="center">Accuracy</th>
<th align="center">Precision</th>
<th align="center">Sensitivity</th>
<th align="center">F1</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">w/o LL</td>
<td align="center">96.12</td>
<td align="center">96.21</td>
<td align="center">96.12</td>
<td align="center">96.12</td>
</tr>
<tr>
<td align="center">w/o LH</td>
<td align="center">96.31</td>
<td align="center">96.38</td>
<td align="center">96.32</td>
<td align="center">96.31</td>
</tr>
<tr>
<td align="center">w/o HL</td>
<td align="center">96.30</td>
<td align="center">96.36</td>
<td align="center">96.30</td>
<td align="center">96.29</td>
</tr>
<tr>
<td align="center">w/o HH</td>
<td align="center">96.23</td>
<td align="center">96.34</td>
<td align="center">96.23</td>
<td align="center">96.22</td>
</tr>
<tr>
<td align="center">WSSA (Ours)</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>96.75</bold>
</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>96.72</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best result under each evaluation metric.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s4-6">
<title>4.6 Robustness of the noise processing module</title>
<p>OCT image quality varies significantly because acquisition is affected by external factors such as imaging-equipment performance and ambient-light interference. Some images even contain severe speckle noise. These issues pose major challenges for subsequent image processing and analysis. In order to verify the effectiveness of our WSSA model for the denoising of retinal OCT images, we added a multiplicative scattering noise model (<xref ref-type="bibr" rid="B23">Huang et al., 2019</xref>) to the test dataset and used the peak signal-to-noise ratio (PSNR) to measure the noise level. It can be expressed by the following Equation:<disp-formula id="e18">
<mml:math id="m68">
<mml:mrow>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(18)</label>
</disp-formula>
<disp-formula id="e19">
<mml:math id="m69">
<mml:mrow>
<mml:mi mathvariant="italic">PSNR</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>20</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>lg</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="italic">Max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mi mathvariant="italic">MSE</mml:mi>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(19)</label>
</disp-formula>where, in <xref ref-type="disp-formula" rid="e18">Equation 18</xref>, <inline-formula id="inf51">
<mml:math id="m70">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the original image undisturbed by noise; <inline-formula id="inf52">
<mml:math id="m71">
<mml:mrow>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is a set of Gaussian noise obeying a mean of 0 and a variance of s (<inline-formula id="inf53">
<mml:math id="m72">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>), whose variance increases with the increase of the gray value of the image; and <inline-formula id="inf54">
<mml:math id="m73">
<mml:mrow>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the image obtained after adding the noise. Also in <xref ref-type="disp-formula" rid="e19">Equation 19</xref>, <inline-formula id="inf55">
<mml:math id="m74">
<mml:mrow>
<mml:mi>Max</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the maximum pixel value of the image and <inline-formula id="inf56">
<mml:math id="m75">
<mml:mtext>MSE</mml:mtext>
</mml:math>
</inline-formula> denotes the mean square error between the image with noise and the original image.</p>
<p>We trained the model using the original training dataset, and added multiplicative scattering noise of five variance levels (<inline-formula id="inf57">
<mml:math id="m76">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; <inline-formula id="inf58">
<mml:math id="m77">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>.</mml:mo>
<mml:msup>
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf59">
<mml:math id="m78">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>.</mml:mo>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf60">
<mml:math id="m79">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>.</mml:mo>
<mml:msup>
<mml:mn>3</mml:mn>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf61">
<mml:math id="m80">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>.</mml:mo>
<mml:msup>
<mml:mn>4</mml:mn>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf62">
<mml:math id="m81">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>.</mml:mo>
<mml:msup>
<mml:mn>5</mml:mn>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>) to the test set during the testing phase. These noise intensities correspond to PSNR values of 30.85 dB, 25.22 dB, 21.95 dB, 19.66 dB, and 18.01 dB, respectively. <xref ref-type="fig" rid="F8">Figure 8</xref> demonstrates the retinal OCT images under different degrees of noise. It can be observed that as noise intensity increases, the PSNR value gradually decreases, the image quality decreases significantly, and the noise interference becomes increasingly pronounced. To verify the robustness of the proposed module in different noise environments, we used the models in <xref ref-type="table" rid="T5">Tables 5</xref>, <xref ref-type="table" rid="T6">6</xref> for comparative analysis. <xref ref-type="table" rid="T7">Tables 7</xref>, <xref ref-type="table" rid="T8">8</xref> show the test results of each model under different noise.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Test dataset after adding different levels of noise.</p>
</caption>
<graphic xlink:href="fcell-13-1608325-g008.tif"/>
</fig>
<table-wrap id="T7" position="float">
<label>TABLE 7</label>
<caption>
<p>Comparison of the first set of wavelet strategies under different noise intensities (dB).</p>
</caption>
<table>
<thead>
<tr>
<th align="left">MethodNoise intensity</th>
<th align="center">Noiseless</th>
<th align="center">30.85</th>
<th align="center">25.22</th>
<th align="center">21.95</th>
<th align="center">19.66</th>
<th align="center">18.01</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">OWT (<xref ref-type="bibr" rid="B50">Talukder and Harada, 2010</xref>)</td>
<td align="center">95.92</td>
<td align="center">95.92</td>
<td align="center">95.87</td>
<td align="center">95.66</td>
<td align="center">95.22</td>
<td align="center">94.35</td>
</tr>
<tr>
<td align="center">WTconv (<xref ref-type="bibr" rid="B9">Finder et al., 2024</xref>)</td>
<td align="center">96.04</td>
<td align="center">95.67</td>
<td align="center">95.94</td>
<td align="center">95.61</td>
<td align="center">95.14</td>
<td align="center">94.5</td>
</tr>
<tr>
<td align="center">WCAM (<xref ref-type="bibr" rid="B1">Alaba and Ball, 2024</xref>)</td>
<td align="center">96.44</td>
<td align="center">96.44</td>
<td align="center">96.19</td>
<td align="center">96.30</td>
<td align="center">96.00</td>
<td align="center">95.33</td>
</tr>
<tr>
<td align="center">OWT &#x2b; MSA</td>
<td align="center">96.23</td>
<td align="center">95.58</td>
<td align="center">95.64</td>
<td align="center">95.68</td>
<td align="center">95.17</td>
<td align="center">95.07</td>
</tr>
<tr>
<td align="center">WSSA (Ours)</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>96.77</bold>
</td>
<td align="center">
<bold>96.68</bold>
</td>
<td align="center">
<bold>96.62</bold>
</td>
<td align="center">
<bold>96.32</bold>
</td>
<td align="center">
<bold>95.82</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best result under each evaluation metric.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T8" position="float">
<label>TABLE 8</label>
<caption>
<p>Comparison of the second set of wavelet strategies under different noise intensities (dB).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">MethodNoise intensity</th>
<th align="center">Noiseless</th>
<th align="center">30.85</th>
<th align="center">25.22</th>
<th align="center">21.95</th>
<th align="center">19.66</th>
<th align="center">18.01</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">w/o LL</td>
<td align="center">96.12</td>
<td align="center">95.82</td>
<td align="center">95.69</td>
<td align="center">95.99</td>
<td align="center">95.19</td>
<td align="center">94.44</td>
</tr>
<tr>
<td align="center">w/o LH</td>
<td align="center">96.31</td>
<td align="center">96.31</td>
<td align="center">96.17</td>
<td align="center">96.08</td>
<td align="center">95.56</td>
<td align="center">94.73</td>
</tr>
<tr>
<td align="center">w/o HL</td>
<td align="center">96.30</td>
<td align="center">96.28</td>
<td align="center">96.31</td>
<td align="center">96.06</td>
<td align="center">95.88</td>
<td align="center">95.49</td>
</tr>
<tr>
<td align="center">w/o HH</td>
<td align="center">96.23</td>
<td align="center">96.08</td>
<td align="center">96.04</td>
<td align="center">95.79</td>
<td align="center">95.57</td>
<td align="center">94.87</td>
</tr>
<tr>
<td align="center">WSSA (Ours)</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>96.77</bold>
</td>
<td align="center">
<bold>96.68</bold>
</td>
<td align="center">
<bold>96.62</bold>
</td>
<td align="center">
<bold>96.32</bold>
</td>
<td align="center">
<bold>95.82</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best result under each evaluation metric.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>As shown in <xref ref-type="table" rid="T7">Tables 7</xref>, <xref ref-type="table" rid="T8">8</xref>, the overall performance of all modules decreases with the increase of noise intensity. Among them, except for the MSLI-Net model proposed in this study and its variant without the HL subband&#x2014;which both exhibit a performance degradation within 1%&#x2014;all other models experience a degradation exceeding 1%, specifically ranging from 1.11% to 1.68%. In addition, our model shows significant performance degradation only when the noise intensity decreases to 19.66 dB, and its fluctuation of no more than 0.1% between no noise and a noise intensity of 21.95 dB, demonstrating good stability. Moreover, across all noise levels, the overall performance of our model is always better than that of other comparative methods, indicating that the method in this paper has good robustness under high-intensity noise interference.</p>
<p>In order to verify the robustness of the proposed LLM in noisy environments, we conducted systematic tests on each module listed in <xref ref-type="table" rid="T4">Table 4</xref> under different noise intensities, and the test results are shown in <xref ref-type="table" rid="T9">Table 9</xref>. From the results, it can be seen that the classification accuracy of the model with the AELGNet cropping strategy decreases by 0.55% when noise is first added, which is the largest decrease among the three modules, indicating that the strategy is more sensitive to noise, and further proving that this cropping approach may amplify the interference of extraneous noise in the background region. As noise intensity increases, the performance of the model using only the SE module is more significantly impaired under high-intensity noise. It is worth noting that the model (MSLI-Net) using LLM shows better classification accuracy than the other two strategies across all noise levels, which fully verifies that the LLM method has stronger noise robustness in complex noise environments.</p>
<table-wrap id="T9" position="float">
<label>TABLE 9</label>
<caption>
<p>Comparison of the LLM and each module under different noise intensities (dB).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">MethodNoise intensity</th>
<th align="center">Noiseless</th>
<th align="center">30.85</th>
<th align="center">25.22</th>
<th align="center">21.95</th>
<th align="center">19.66</th>
<th align="center">18.01</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Cropping Strategy in AELGNet (<xref ref-type="bibr" rid="B42">Sharma and Vardhan, 2025</xref>)</td>
<td align="center">96.30</td>
<td align="center">95.75</td>
<td align="center">95.69</td>
<td align="center">95.39</td>
<td align="center">95.88</td>
<td align="center">95.49</td>
</tr>
<tr>
<td align="center">SE (<xref ref-type="bibr" rid="B21">Hu et al., 2018</xref>)</td>
<td align="center">96.23</td>
<td align="center">95.96</td>
<td align="center">95.81</td>
<td align="center">95.58</td>
<td align="center">95.57</td>
<td align="center">94.87</td>
</tr>
<tr>
<td align="center">WSSA (Ours)</td>
<td align="center">
<bold>96.72</bold>
</td>
<td align="center">
<bold>96.77</bold>
</td>
<td align="center">
<bold>96.68</bold>
</td>
<td align="center">
<bold>96.62</bold>
</td>
<td align="center">
<bold>96.32</bold>
</td>
<td align="center">
<bold>95.82</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best result under each evaluation metric.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s4-7">
<title>4.7 Visualization and analysis</title>
<p>In order to visually assess the effectiveness of the MDF proposed in this paper in multi-scale feature extraction and its ability to deeply integrate with the original image features, this paper introduces the Grad-CAM method to visualize and analyze the image regions that each of the comparative modules pays attention to in the classification decision-making process. The method effectively reveals the ability of each module to pay attention to the key regions of the input image through the generation of heat maps.</p>
<p>In this paper, based on each module in <xref ref-type="table" rid="T3">Table 3</xref>, its corresponding heat map is generated at different network stages of ResNet50 and visualized for comparison. As shown in <xref ref-type="fig" rid="F9">Figure 9</xref>, in the layer1 stage of ResNet50, each module shows high consistency in focusing on the lesion region. As the network deepens, the MDF consistently maintains high consistency with the backbone network at all stages and further enhances its ability to focus on lesion regions. In contrast, during the first two stages, the MSCB consistently aligns its focus on the lesion regions with that of the backbone network and remains relatively stable; however, the region of focus deviates significantly in the third stage. Conversely, DFE and ASPP show a tendency of divergence of the total attention region after the first stage, and although they briefly enhance the ability of layer2 of ResNet50 to focus on the lesion, by the third and fourth stages, their ability to focus on the lesion decreases significantly, and neither of them is able to focus on the lesion portion well.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Visualization and analysis of MDF, DFE, MSCB and ASPP.</p>
</caption>
<graphic xlink:href="fcell-13-1608325-g009.tif"/>
</fig>
<p>The comparative results heat map visualization further validates the advantages of the MDF module in feature fusion and semantic modeling. Especially In deeper layers, MDF is still able to maintain a stable and precise attention region, reflecting stronger discriminative and semantic retention abilities.</p>
<p>Meanwhile, we also visualize and compare the mentioned models in <xref ref-type="table" rid="T4">Table 4</xref>. As shown in <xref ref-type="fig" rid="F10">Figure 10</xref>, in the shallow stage (c2 and c3), the models using SE, the cropping strategy in AELGNet, and the LLM proposed in this paper are able to locate the lesion region more accurately, which reflects a good initial discriminative ability. However, in the deeper network stages (c4 and p5), SE and cropping strategy in AELGNet gradually demonstrate increased background attention, compared to LLM which still maintains a stable focus on the lesion region in the deeper stages. This result fully demonstrates the effectiveness of our LLM&#x2019;s performance for lesion localization.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Visualization and analysis of SE, cropping strategy in AELGNet, and LLM.</p>
</caption>
<graphic xlink:href="fcell-13-1608325-g010.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>In this study, we propose a novel network architecture called MSLI-Net for the classification task of retinal optical coherence tomography (OCT) images. The model effectively enhances the feature extraction capability of the model for the lesion region and significantly improves the overall classification performance through three core modules, namely, the multi-scale dilation fusion module (MDF), the multi-segmented lesion localization fusion module (LLM), and the wavelet subband spatial attention module (WSSA). On the publicly available OCT-C8 dataset, this method achieves a classification accuracy of 96.72%. We further confirm the critical role of each of the three modules, MDF, LLM and WSSA, in network performance improvement through comprehensive ablation experiments. Meanwhile, MSLI-Net still exhibits robust performance under stronger noise interference environment. MSLI-Net architecture not only has practical value for OCT image classification, but also provides new research ideas and effective technical references for future network design for similar tasks.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.34740/KAGGLE/DSV/2736749">https://doi.org/10.34740/KAGGLE/DSV/2736749</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>ZQ: Conceptualization, Formal Analysis, Investigation, Methodology, Software, Validation, Visualization, Writing &#x2013; original draft. JH: Conceptualization, Data curation, Funding acquisition, Methodology, Project administration, Resources, Supervision, Validation, Writing &#x2013; review and editing. JC: Conceptualization, Investigation, Software, Visualization, Writing &#x2013; review and editing. GL: Conceptualization, Software, Validation, Visualization, Writing &#x2013; review and editing. HW: Software, Validation, Visualization, Writing &#x2013; review and editing. SL: Validation, Visualization, Writing &#x2013; review and editing. SC: Formal Analysis, Resources, Supervision, Visualization, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported in part by the National Natural Science Foundation of China (62466033), in part by the Jiangxi Provincial Natural Science Foundation (20242BAB20070).</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The authors declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Alaba</surname>
<given-names>S. Y.</given-names>
</name>
<name>
<surname>Ball</surname>
<given-names>J. E.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>WCAM: wavelet convolutional attention module SoutheastCon 2024</article-title>. <conf-name>SoutheastCon 2024</conf-name>, <conf-loc>Atlanta, GA, United States</conf-loc>, <conf-date>15-24 March 2024</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>854</fpage>&#x2013;<lpage>859</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Awais</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Muller</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Meriaudeau</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Classification of sd-OCT images using a deep learning approach</article-title>. <source>IEEE International Conference on Signal and Image Processing Applications</source>, <fpage>8120661</fpage>&#x2013;<lpage>492</lpage>. <pub-id pub-id-type="doi">10.1109/ICSIPA.2017.8120661</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bui</surname>
<given-names>P.-N.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>D.-T.</given-names>
</name>
<name>
<surname>Bum</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Choo</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Multi-scale feature enhancement in multi-task learning for medical image analysis</article-title>. <comment>arXiv preprint arXiv:2412.00351</comment>. <pub-id pub-id-type="doi">10.48550/arXiv.2412.00351</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Burrus</surname>
<given-names>C. S.</given-names>
</name>
<name>
<surname>Gopinath</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>1998</year>). <source>Wavelets and wavelet transforms</source>. <edition>houston edition</edition>. <publisher-loc>Houston, TX</publisher-loc>: <publisher-name>rice university</publisher-name>, <fpage>98</fpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Long</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>WaveNet-SF: a hybrid network for retinal disease detection based on wavelet transform in the spatial-frequency domain</article-title>. <comment>arXiv preprint arXiv:2501.11854</comment>. <pub-id pub-id-type="doi">10.48550/arXiv.2501.11854</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Diao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xiang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Classification and segmentation of OCT images for age-related macular degeneration based on dual guidance networks</article-title>. <source>Biomed. Signal Process. Control</source> <volume>84</volume>, <fpage>104810</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2023.104810</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dutta</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Sathi</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Hossain</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Dewan</surname>
<given-names>M. A. A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Conv-ViT: a convolution and vision transformer-based hybrid feature extraction method for retinal disease detection</article-title>. <source>J. Imaging</source> <volume>9</volume> (<issue>7</issue>), <fpage>140</fpage>. <pub-id pub-id-type="doi">10.3390/jimaging9070140</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>CPFNet: context pyramid fusion network for medical image segmentation</article-title>. <source>IEEE Trans. Med. imaging</source> <volume>39</volume> (<issue>10</issue>), <fpage>3008</fpage>&#x2013;<lpage>3018</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2020.2983721</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Finder</surname>
<given-names>S. E.</given-names>
</name>
<name>
<surname>Amoyal</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Treister</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Freifeld</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2024</year>). <source>Wavelet convolutions for large receptive fields European Conference on Computer Vision</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer Nature Switzerland</publisher-name>), <fpage>363</fpage>&#x2013;<lpage>380</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-72949-2_21</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>R. X.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2010</year>). In <source>Wavelets: theory and applications for manufacturing</source>, <source>From fourier transform to wavelet transform: a historical perspective</source>. <fpage>17</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-4419-1545-0_2</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gong</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W. T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X. M.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y. J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S. J.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Development and research status of intelligent ophthalmology in China</article-title>. <source>Int. J. Ophthalmol.</source> <volume>17</volume> (<issue>12</issue>), <fpage>2308</fpage>&#x2013;<lpage>2315</lpage>. <pub-id pub-id-type="doi">10.18240/ijo.2024.12.20</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grossniklaus</surname>
<given-names>H. E.</given-names>
</name>
<name>
<surname>Geisert</surname>
<given-names>E. E.</given-names>
</name>
<name>
<surname>Nickerson</surname>
<given-names>J. M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Introduction to the retina</article-title>. <source>Prog. Mol. Biol. Transl. Sci.</source> <volume>134</volume>, <fpage>383</fpage>&#x2013;<lpage>396</lpage>. <pub-id pub-id-type="doi">10.1016/bs.pmbts.2015.06.001</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>An interpretable transformer network for the retinal disease classification using optical coherence tomography</article-title>. <source>Sci. Rep.</source> <volume>13</volume> (<issue>1</issue>), <fpage>3637</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-023-30853-z</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Deep residual learning for image recognition</article-title>. <conf-name>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</conf-name>, <fpage>770</fpage>&#x2013;<lpage>778</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S. H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019b</year>). <article-title>Improvement of cerebral microbleeds detection based on discriminative feature learning</article-title>. <source>Fundam. Inf.</source> <volume>168</volume> (<issue>2-4</issue>), <fpage>231</fpage>&#x2013;<lpage>248</lpage>. <pub-id pub-id-type="doi">10.3233/fi-2019-1830</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y. D.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019a</year>). <article-title>Detecting cerebral microbleeds with transfer learning</article-title>. <source>Mach. Vis. Appl.</source> <volume>30</volume> (<issue>7</issue>), <fpage>1123</fpage>&#x2013;<lpage>1133</lpage>. <pub-id pub-id-type="doi">10.1007/s00138-019-01029-5</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S. H.</given-names>
</name>
<name>
<surname>Peet</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y. D.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2020a</year>). <article-title>Brain age prediction of children using routine brain MR images via deep learning</article-title>. <source>Front. Neurology</source> <volume>11</volume>, <fpage>584682</fpage>. <pub-id pub-id-type="doi">10.3389/fneur.2020.584682</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S. H.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020b</year>). <article-title>Classification of cerebral microbleeds based on fully-optimized convolutional neural network</article-title>. <source>Multimedia Tools Appl.</source> <volume>79</volume> (<issue>21</issue>), <fpage>15151</fpage>&#x2013;<lpage>15169</lpage>. <pub-id pub-id-type="doi">10.1007/s11042-018-6862-z</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>S. C. H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2022a</year>). <article-title>Unsupervised domain adaptation for cross-modality liver segmentation via joint adversarial learning and self-learning</article-title>. <source>Appl. Soft Comput.</source> <volume>121</volume>, <fpage>108729</fpage>. <pub-id pub-id-type="doi">10.1016/j.asoc.2022.108729</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y. D.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>Source-free unsupervised domain adaptation for cross-modality abdominal multi-organ segmentation</article-title>. <source>Knowledge-Based Syst.</source> <volume>250</volume>, <fpage>109155</fpage>. <pub-id pub-id-type="doi">10.1016/j.knosys.2022.109155</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Squeeze-and-excitation networks</article-title>. <source>Proc. IEEE Conf. Comput. Vis. pattern Recognit.</source>, <fpage>7132</fpage>&#x2013;<lpage>7141</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Swanson</surname>
<given-names>E. A.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>C. P.</given-names>
</name>
<name>
<surname>Schuman</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>Stinson</surname>
<given-names>W. G.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>1991</year>). <article-title>Optical coherence tomography</article-title>. <source>science</source> <volume>254</volume> (<issue>5035</issue>), <fpage>1178</fpage>&#x2013;<lpage>1181</lpage>. <pub-id pub-id-type="doi">10.1126/science.1957169</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Rabbani</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Automatic classification of retinal optical coherence tomography images with layer guided convolutional neural network</article-title>. <source>IEEE Signal Process. Lett.</source> <volume>26</volume> (<issue>7</issue>), <fpage>1026</fpage>&#x2013;<lpage>1030</lpage>. <pub-id pub-id-type="doi">10.1109/lsp.2019.2917779</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ji</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Qian</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Research progress of artificial intelligence image analysis in systemic disease&#x2010;related ophthalmopathy</article-title>. <source>Dis. Markers</source> <volume>2022</volume> (<issue>1</issue>), <fpage>3406890</fpage>. <pub-id pub-id-type="doi">10.1155/2022/3406890</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kamran</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Saha</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sabbir</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Tavakkoli</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Optic-net: a novel convolutional neural network for diagnosis of retinal diseases from optical tomography images</article-title>, <conf-name>2019 18th IEEE international conference on machine learning and applications (ICMLA)</conf-name>, (<publisher-name>IEEE</publisher-name>), <fpage>964</fpage>&#x2013;<lpage>971</lpage>. <pub-id pub-id-type="doi">10.1109/ICMLA.2019.00165</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaothanthong</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Limwattanayingyong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Silpa-Archa</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tadarati</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Amphornphruet</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Singhanetr</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>The classification of common macular diseases using deep learning on optical coherence tomography images with and without prior automated segmentation</article-title>. <source>Diagnostics</source> <volume>13</volume> (<issue>2</issue>), <fpage>189</fpage>. <pub-id pub-id-type="doi">10.3390/diagnostics13020189</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karthik</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Mahadevappa</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Convolution neural networks for optical coherence tomography (OCT) image classification</article-title>. <source>Biomed. Signal Process. Control</source> <volume>79</volume>, <fpage>104176</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2022.104176</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kermany</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Goldbaum</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Valentim</surname>
<given-names>C. C. S.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Baxter</surname>
<given-names>S. L.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Identifying medical diagnoses and treatable diseases by image-based deep learning</article-title>. <source>Cell</source> <volume>172</volume> (<issue>5</issue>), <fpage>1122</fpage>&#x2013;<lpage>1131</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2018.02.010</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khalil</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Mehmood</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>OCTNet: a modified multi-scale attention feature fusion network with InceptionV3 for retinal OCT image classification</article-title>. <source>Mathematics</source> <volume>12</volume> (<issue>19</issue>), <fpage>3003</fpage>. <pub-id pub-id-type="doi">10.3390/math12193003</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kq</surname>
<given-names>H. G. L. Z. W.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Densely connected convolutional networks</article-title>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laouarem</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kara-Mohamed</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bourennane</surname>
<given-names>E. B.</given-names>
</name>
<name>
<surname>Hamdi-Cherif</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Htc-retina: a hybrid retinal diseases classification model using transformer-convolutional neural network from optical coherence tomography images</article-title>. <source>Comput. Biol. Med.</source> <volume>178</volume>, <fpage>108726</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2024.108726</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X. D.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Z. Z.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Deep learning-based automated detection of retinal diseases using optical coherence tomography images</article-title>. <source>Biomed. Opt. express</source> <volume>10</volume> (<issue>12</issue>), <fpage>6204</fpage>&#x2013;<lpage>6226</lpage>. <pub-id pub-id-type="doi">10.1364/BOE.10.006204</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Source-free unsupervised adaptive segmentation for knee joint MRI</article-title>. <source>Biomed. Signal Process. Control</source> <volume>92</volume>, <fpage>106028</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2024.106028</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lo</surname>
<given-names>S.-Y.</given-names>
</name>
<name>
<surname>Hang</surname>
<given-names>H.-M.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>S.-W.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>J.-J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Efficient dense modules of asymmetric convolution for real-time semantic segmentation</article-title>, <source>Proceedings of the 1st ACM International Conference on Multimedia in Asia</source>, <fpage>1</fpage>&#x2013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1145/3338533.3366558</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Manzari</surname>
<given-names>O. N.</given-names>
</name>
<name>
<surname>Ahmadabadi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kashiani</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shokouhi</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Ayatollahi</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>MedViT: a robust vision transformer for generalized medical image classification</article-title>. <source>Comput. Biol. Med.</source> <volume>157</volume>, <fpage>106791</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.106791</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mehta</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rastegari</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Caspi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shapiro</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hajishirzi</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Espnet: efficient spatial pyramid of dilated convolutions for semantic segmentation</source>, <source>Proceedings of the european conference on computer vision (ECCV)</source>. <fpage>552</fpage>&#x2013;<lpage>568</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-01249-6_34</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Obuli</surname>
<given-names>S. N.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Retinal OCT image classification - C8</article-title>,&#x201d;. <publisher-loc>San Francisco, CA</publisher-loc>: <publisher-name>Kaggle</publisher-name>. <pub-id pub-id-type="doi">10.34740/KAGGLE/DSV/2736749</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhuo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Multi-scale-denoising residual convolutional network for retinal disease classification using OCT</article-title>. <source>Sensors</source> <volume>24</volume> (<issue>1</issue>), <fpage>150</fpage>. <pub-id pub-id-type="doi">10.3390/s24010150</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pennington</surname>
<given-names>K. L.</given-names>
</name>
<name>
<surname>DeAngelis</surname>
<given-names>M. M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Epidemiology of age-related macular degeneration (AMD): associations with cardiovascular disease phenotypes and lipid factors</article-title>. <source>Eye Vis.</source> <volume>3</volume>, <fpage>34</fpage>&#x2013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1186/s40662-016-0063-5</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qian</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xi</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Enhanced diagnosis of thyroid-associated eye diseases based on deep learning: a novel triplet loss design strategy</article-title>. <source>Biomed. Signal Process. Control</source> <volume>100</volume>, <fpage>107161</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2024.107161</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Robinson</surname>
<given-names>B. E.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Prevalence of Asymptomatic Eye Disease Pr&#xe9;valence des maladies oculaires asymptomatiques</article-title>. <source>Rev. Can. D&#x27;Optom&#xe9;trie</source> <volume>65</volume> (<issue>5</issue>), <fpage>175</fpage>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sharma</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Vardhan</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>AELGNet: attention-based enhanced local and global features network for medicinal leaf and plant classification</article-title>. <source>Comput. Biol. Med.</source> <volume>184</volume>, <fpage>109447</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2024.109447</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Simonyan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zisserman</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Very deep convolutional networks for large-scale image recognition</article-title>. <comment>arXiv preprint arXiv:1409.1556</comment>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Construction of a predictive model for the efficacy of anti-VEGF therapy in macular edema patients based on OCT imaging: a retrospective study</article-title>. <source>Front. Med.</source> <volume>12</volume>, <fpage>1505530</fpage>. <pub-id pub-id-type="doi">10.3389/fmed.2025.1505530</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhuo</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Real-time semantic segmentation network with an enhanced backbone based on Atrous spatial pyramid pooling module</article-title>. <source>Eng. Appl. Artif. Intell.</source> <volume>133</volume>, <fpage>107988</fpage>. <pub-id pub-id-type="doi">10.1016/j.engappai.2024.107988</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Subramanian</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shanmugavadivel</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Naren</surname>
<given-names>O. S.</given-names>
</name>
<name>
<surname>Premkumar</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Rankish</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Classification of retinal oct images using deep learning</article-title>, <conf-name>2022 international conference on computer communication and informatics (ICCCI)</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1109/ICCCI54379.2022.9740985</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sunija</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Kar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gayathri</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gopi</surname>
<given-names>V. P.</given-names>
</name>
<name>
<surname>Palanisamy</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>OCTnet: a lightweight cnn for retinal disease classification from optical coherence tomography images</article-title>. <source>Comput. methods programs Biomed.</source> <volume>200</volume>, <fpage>105877</fpage>. <pub-id pub-id-type="doi">10.1016/j.cmpb.2020.105877</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Szegedy</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sermanet</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Reed</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Anguelov</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Going deeper with convolutions</article-title>, <conf-name>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</conf-name>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2015.7298594</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Szegedy</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Vanhoucke</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Ioffe</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shlens</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wojna</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Rethinking the inception architecture for computer vision</article-title>, <conf-name>Proceedings of the IEEE Conference on Computer Vision and Pattern rRecognition</conf-name>, <fpage>2818</fpage>&#x2013;<lpage>2826</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2016.308</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Talukder</surname>
<given-names>K. H.</given-names>
</name>
<name>
<surname>Harada</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Haar wavelet based approach for image compression and quality assessment of compressed image</article-title>. <comment>arXiv preprint arXiv:1010.4084</comment>.</citation>
</ref>
<ref id="B51">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Efficientnet: rethinking model scaling for convolutional neural networks</article-title>,&#x201d; in <source>International conference on machine learning</source>. <publisher-loc>Long Beach, CA</publisher-loc>: <publisher-name>PMLR</publisher-name>, <fpage>6105</fpage>&#x2013;<lpage>6114</lpage>.</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tsuji</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hirose</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fujimori</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hirose</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Oyama</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Saikawa</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Classification of optical coherence tomography images using a capsule network</article-title>. <source>BMC Ophthalmol.</source> <volume>20</volume>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1186/s12886-020-01382-4</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>DBPF-net: dual-branch structural feature extraction reinforcement network for ocular surface disease image classification</article-title>. <source>Front. Med.</source> <volume>10</volume>, <fpage>1309097</fpage>. <pub-id pub-id-type="doi">10.3389/fmed.2023.1309097</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Deep learning for quality assessment of retinal OCT images</article-title>. <source>Biomed. Opt. express</source> <volume>10</volume> (<issue>12</issue>), <fpage>6057</fpage>&#x2013;<lpage>6072</lpage>. <pub-id pub-id-type="doi">10.1364/BOE.10.006057</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>CTransCNN: combining transformer and CNN in multilabel medical image classification</article-title>. <source>Knowledge-Based Syst.</source> <volume>281</volume>, <fpage>111030</fpage>. <pub-id pub-id-type="doi">10.1016/j.knosys.2023.111030</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>ADNet: lane shape prediction via anchor decomposition</article-title>. <source>Proceedings of the IEEE/CVF International Conference on Computer Vision</source>, <fpage>6404</fpage>&#x2013;<lpage>6413</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV51070.2023.00589</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2022a</year>). <article-title>A few-shot learning-based retinal vessel segmentation method for assisting in the central serous chorioretinopathy laser surgery</article-title>. <source>Front. Med.</source> <volume>9</volume>, <fpage>821565</fpage>. <pub-id pub-id-type="doi">10.3389/fmed.2022.821565</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>An intelligent location method of key boundary points for assisting the diameter measurement of central serous chorioretinopathy lesion area</article-title>. <source>Comput. Biol. Med.</source> <volume>147</volume>, <fpage>105730</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.105730</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Weakly supervised detection of central serous chorioretinopathy based on local binary patterns and discrete wavelet transform</article-title>. <source>Comput. Biol. Med.</source> <volume>127</volume>, <fpage>104056</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2020.104056</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>CRAT: advanced transformer-based deep learning algorithms in OCT image classification</article-title>. <source>Biomed. Signal Process. Control</source> <volume>104</volume>, <fpage>107544</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2025.107544</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Koltun</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Multi-scale context aggregation by dilated convolutions</article-title>. <comment>arXiv Prepr. arXiv:1511.07122</comment>.</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ning</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>ANC-Net: a novel multi-scale active noise cancellation network for rotating machinery fault diagnosis based on discrete wavelet transform</article-title>. <source>Expert Syst. Appl.</source> <volume>265</volume>, <fpage>125937</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2024.125937</pub-id>
</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhuo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Rong</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Hypermixed convolutional neural network for retinal vein occlusion classification</article-title>. <source>Dis. Markers</source> <volume>2022</volume> (<issue>1</issue>), <fpage>1730501</fpage>. <pub-id pub-id-type="doi">10.1155/2022/1730501</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Koniusz</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Deep stacked hierarchical multi-patch network for image deblurring</article-title>, <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>, <fpage>5978</fpage>&#x2013;<lpage>5986</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2019.00613</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Si</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>A scoping review of advancements in machine learning for glaucoma: current trends and future direction</article-title>. <source>Front. Med.</source> <volume>12</volume>, <fpage>1573329</fpage>. <pub-id pub-id-type="doi">10.3389/fmed.2025.1573329</pub-id>
</citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Mijit</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Effective automatic classification methods via deep learning for myopic maculopathy</article-title>. <source>Front. Med.</source> <volume>11</volume>, <fpage>1492808</fpage>. <pub-id pub-id-type="doi">10.3389/fmed.2024.1492808</pub-id>
</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Shu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Wavelet-Attention CNN for image classification</article-title>. <source>Multimed. Syst.</source> <volume>28</volume> (<issue>3</issue>), <fpage>915</fpage>&#x2013;<lpage>924</lpage>. <pub-id pub-id-type="doi">10.1007/s00530-022-00889-8</pub-id>
</citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Research on an artificial intelligence-based myopic maculopathy grading method using EfficientNet</article-title>. <source>Indian J. Ophthalmol.</source> <volume>72</volume> (<issue>Suppl. 1</issue>), <fpage>S53</fpage>&#x2013;<lpage>S59</lpage>. <pub-id pub-id-type="doi">10.4103/IJO.IJO_48_23</pub-id>
</citation>
</ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Screening of common retinal diseases using six-category models based on EfficientNet</article-title>. <source>Front. Med.</source> <volume>9</volume>, <fpage>808402</fpage>. <pub-id pub-id-type="doi">10.3389/fmed.2022.808402</pub-id>
</citation>
</ref>
<ref id="B70">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zuo</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ping</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Multi-resolution visual Mamba with multi-directional selective mechanism for retinal disease detection</article-title>. <source>Front. Cell Dev. Biol.</source> <volume>12</volume>, <fpage>1484880</fpage>. <pub-id pub-id-type="doi">10.3389/fcell.2024.1484880</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>