<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Med.</journal-id>
<journal-title>Frontiers in Medicine</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Med.</abbrev-journal-title>
<issn pub-type="epub">2296-858X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmed.2025.1596726</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Medicine</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>OculusNet: Detection of retinal diseases using a tailored web-deployed neural network and saliency maps for explainable AI</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Umair</surname> <given-names>Muhammad</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2964496/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ahmad</surname> <given-names>Jawad</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/961429/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Saidani</surname> <given-names>Oumaima</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Alshehri</surname> <given-names>Mohammed S.</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3002411/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Al Mazroa</surname> <given-names>Alanoud</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Hanif</surname> <given-names>Muhammad</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2976325/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ullah</surname> <given-names>Rahmat</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2907504/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Khan</surname> <given-names>Muhammad Shahbaz</given-names></name>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2964279/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Faculty of Engineering, Multimedia University</institution>, <addr-line>Cyberjaya</addr-line>, <country>Malaysia</country></aff>
<aff id="aff2"><sup>2</sup><institution>Cybersecurity Center, Prince Mohammad Bin Fahd University</institution>, <addr-line>Al Khobar</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Information Systems, College of Computer and Information Sciences, Princess Nourah bint Abdulrahman University</institution>, <addr-line>Riyadh</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Computer Science, College of Computer and Information Sciences, Najran University</institution>, <addr-line>Najran</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff5"><sup>5</sup><institution>Department of Informatics, School of Business, &#x000D6;rebro Universitet</institution>, <addr-line>&#x000D6;rebro</addr-line>, <country>Sweden</country></aff>
<aff id="aff6"><sup>6</sup><institution>School of Computer Science and Electronic Engineering (CSEE), University of Essex</institution>, <addr-line>Colchester</addr-line>, <country>United Kingdom</country></aff>
<aff id="aff7"><sup>7</sup><institution>School of Computing, Engineering and the Built Environment, Edinburgh Napier University</institution>, <addr-line>Edinburgh</addr-line>, <country>United Kingdom</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Yanda Meng, University of Exeter, United Kingdom</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Mohan Bhandari, Samridhhi College, Nepal</p>
<p>M. Abdul Jawad, National Institute of Technology, Srinagar, India</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Muhammad Hanif <email>muhammad.hanif&#x00040;oru.se</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>02</day>
<month>07</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1596726</elocation-id>
<history>
<date date-type="received">
<day>20</day>
<month>03</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>06</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Umair, Ahmad, Saidani, Alshehri, Al Mazroa, Hanif, Ullah and Khan.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Umair, Ahmad, Saidani, Alshehri, Al Mazroa, Hanif, Ullah and Khan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Retinal diseases are among the leading causes of blindness worldwide, requiring early detection for effective treatment. Manual interpretation of ophthalmic imaging, such as optical coherence tomography (OCT), is traditionally time-consuming, prone to inconsistencies, and requires specialized expertise in ophthalmology. This study introduces OculusNet, an efficient and explainable deep learning (DL) approach for detecting retinal diseases using OCT images. The proposed method is specifically tailored for complex medical image patterns in OCTs to identify retinal disorders, such as choroidal neovascularization (CNV), diabetic macular edema (DME), and age-related macular degeneration characterized by drusen. The model benefits from Saliency Map visualization, an Explainable AI (XAI) technique, to interpret and explain how it reaches conclusions when identifying retinal disorders. Furthermore, the proposed model is deployed on a web page, allowing users to upload retinal OCT images and receive instant detection results. This deployment demonstrates significant potential for integration into ophthalmic departments, enhancing diagnostic accuracy and efficiency. In addition, to ensure an equitable comparison, a transfer learning approach has been applied to four pre-trained models: VGG19, MobileNetV2, VGG16, and DenseNet-121. Extensive evaluation reveals that the proposed OculusNet model achieves a test accuracy of 95.48% and a validation accuracy of 98.59%, outperforming all other models in comparison. Moreover, to assess the proposed model&#x00027;s reliability and generalizability, the Matthews Correlation Coefficient and Cohen&#x00027;s Kappa Coefficient have been computed, validating that the model can be applied in practical clinical settings to unseen data.</p></abstract>
<kwd-group>
<kwd>retina</kwd>
<kwd>retinal disorder</kwd>
<kwd>explainable AI</kwd>
<kwd>artificial intelligence</kwd>
<kwd>ophthalmic imaging</kwd>
<kwd>neural networks</kwd>
</kwd-group>
<counts>
<fig-count count="15"/>
<table-count count="13"/>
<equation-count count="19"/>
<ref-count count="41"/>
<page-count count="22"/>
<word-count count="9049"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Ophthalmology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Retinal imaging technologies, such as optical coherence tomography (OCT), have become essential tools in ophthalmology due to their high resolution, non-invasive nature, and ability to reveal critical microstructural details of retinal layers. In humans and most vertebrates, the retina is a thin, light-sensitive membrane located at the back of the eye (<xref ref-type="bibr" rid="B1">1</xref>). It consists of several layers, including one made up of light-sensitive cells known as photoreceptors. The retina converts incoming light into neural signals (<xref ref-type="bibr" rid="B2">2</xref>&#x02013;<xref ref-type="bibr" rid="B4">4</xref>). The human eye contains two types of photoreceptors: rods and cones. Rod photoreceptors are responsible for black-and-white vision and motion detection, performing particularly well in low-light conditions. Cone photoreceptors, on the other hand, are responsible for color and central vision. These receptors function well in medium to bright light. Rods occupy the entire retina; however, cones are located and clustered in a small central area of the retina known as the macula. Furthermore, there is a slight depression at the center of the macula, called the fovea. The fovea is the point in the retina primarily responsible for color vision and visual acuity (the sharpness of eyesight). The captured information is processed by the retina and transmitted to the brain through the optic nerve for further visual recognition (<xref ref-type="bibr" rid="B2">2</xref>, <xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B5">5</xref>). All of these parts of the retina are critical for eyesight, and most eyesight-related diseases primarily occur due to damage or disease in the retina (<xref ref-type="bibr" rid="B3">3</xref>, <xref ref-type="bibr" rid="B6">6</xref>).</p>
<p>Several diseases can damage the retina, including choroidal neovascularization (CNV), diabetic macular edema (DME), and age-related macular degeneration characterized by drusen. These disorders lead to visual impairment and even blindness. Conditions affecting the retina have a critical impact on patients, as eyesight (i.e., vision) plays a vital role in human life. Therefore, scientists have been exploring new and effective tools for diagnosing and detecting retinal conditions early (<xref ref-type="bibr" rid="B2">2</xref>, <xref ref-type="bibr" rid="B7">7</xref>, <xref ref-type="bibr" rid="B8">8</xref>). Recently, OCT has proven to be a promising non-invasive technique for micro-scale imaging of biological tissues (<xref ref-type="bibr" rid="B9">9</xref>). OCT technology captures images of the retina in a cross-sectional format using light waves (<xref ref-type="bibr" rid="B10">10</xref>, <xref ref-type="bibr" rid="B11">11</xref>). OCT is significant in various medical applications, with ophthalmology being its largest commercial application. OCT is preferred for the monitoring and detection of several retinal diseases (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B13">13</xref>). This technology has evolved through various configurations since its inception, i.e., time-domain OCT (<xref ref-type="bibr" rid="B14">14</xref>), spectral-domain OCT (<xref ref-type="bibr" rid="B15">15</xref>), and swept-source OCT (<xref ref-type="bibr" rid="B16">16</xref>). Owing to the aforementioned technological advancements, OCT is the most preferred and reliable method for diagnosing eye diseases.</p>
<p>Recently, the field of biomedicine has significantly evolved in the detection and analysis of diseases. Traditional methods of disease detection were often unreliable and time-consuming. However, with the advent of artificial intelligence (AI), accuracy in disease detection has increased exponentially, and turnaround times have greatly improved. Deep learning (DL) techniques have become integral to the biomedical field, providing fast and reliable results for disease detection. Therefore, this paper presents an efficient and explainable approach to classify retinal disorders, such as CNV, DME, and Drusen, from normal conditions.</p>
<p>The main contributions presented by this study include the following:</p>
<list list-type="order">
<list-item><p>An efficient and explainable DL model specifically designed for complex medical image patterns in OCTs classifies retinal disorders, such as CNV, DME, and Drusen, differentiating them from normal conditions. The results of reliability parameters validate that OculusNet can be effectively used in practical settings on unseen data.</p></list-item>
<list-item><p>The proposed approach utilizes saliency map visualization, an explainable AI (XAI) technique, to visualize the most influential pixels and interpret how the proposed model makes its decisions while identifying retinal disorders. Results from Saliency Maps have also been compared to other XAI techniques, such as GradCam&#x0002B;&#x0002B;, SHAP, and Lime.</p></list-item>
<list-item><p>The trained weights of the OculusNet model are deployed on a webpage using the Streamlit server, which is accessible to all devices connected to the same network. This deployment demonstrates significant potential for integration into ophthalmic departments, enhancing diagnostic accuracy and efficiency.</p></list-item>
<list-item><p>The transfer learning technique has been applied to ensure a fair comparison. It has been applied to four state-of-the-art models that have been proven to be the best for classification problems. The utilized pre-trained models include VGG16, VGG19, MobileNetV2, and DenseNet-121.</p></list-item>
</list>
<p>The structure of the rest of the article is organized as follows: Section 2 presents the literature related to this study, Section 3 discusses the dataset and data pre-processing methodologies, Section 4 presents the utilized methodology along with the design and architecture of the proposed OculusNet model, and Section 5 discusses the performance parameters that are utilized for this study and reports the obtained results. Finally, a conclusion to this study is provided in Section 6.</p></sec>
<sec id="s2">
<title>2 Related research</title>
<p>OCTs provide high-resolution cross-sectional images of the retina, which play a potential role in diagnosing retinal disorders (<xref ref-type="bibr" rid="B17">17</xref>). Recently, DL has been extensively utilized to detect retinal disorders efficiently (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B18">18</xref>&#x02013;<xref ref-type="bibr" rid="B20">20</xref>). The development of novel image processing models has enhanced noise reduction and retinal layer segmentation in OCT images, thereby facilitating the accurate diagnosis of retinal disorders (<xref ref-type="bibr" rid="B21">21</xref>). The application of OCT in pediatric ophthalmology has been revolutionary, enabling the visualization of retinal structures in infants and neonates (<xref ref-type="bibr" rid="B6">6</xref>), which is crucial for the diagnosis of retinal-related diseases. The integration of deep learning (DL) methods has addressed reliability issues in OCT image analysis, leading to improved diagnostic accuracy (<xref ref-type="bibr" rid="B18">18</xref>, <xref ref-type="bibr" rid="B22">22</xref>). Additionally, OCT&#x00027;s capability to differentiate between retinoschisis and retinal detachment has been confirmed, showcasing its diagnostic versatility. For the diagnosis of age-related macular degeneration diseases, several methods have been proposed that detect retinal pigment epithelium (RPE) through OCT images (<xref ref-type="bibr" rid="B23">23</xref>). Recently, various algorithms have also been developed for the classification of eye diseases. For example, the authors in Muni Nagamani and Rayachoti (<xref ref-type="bibr" rid="B24">24</xref>) present a DL approach that utilizes OCT images for the classification of retinal diseases using a modified ResNet50 model. Their study shows that they used a single-view retinal image set along with applied segmentation. Similarly, in Wang et al. (<xref ref-type="bibr" rid="B25">25</xref>), a semi-supervision-based approach named Caps-cGAN has been proposed to reduce noise in OCT images, particularly speckle noise. Furthermore, to extend the automated diagnosis of eye diseases, a semi-automated approach is presented in Shin et al. (<xref ref-type="bibr" rid="B26">26</xref>). The authors utilized pig eye images in this approach and achieved an accuracy of 83.89%. The classification of retinal diseases, namely DME, Drusen, and CNV, has been performed in Adel et al. (<xref ref-type="bibr" rid="B27">27</xref>), using the Inception and Xception models with 6,000 OCT images.</p>
<p>Recent literature also demonstrates the use of pre-trained models for detecting retinal disorders. For instance, in Islam et al. (<xref ref-type="bibr" rid="B28">28</xref>), a total of 109,309 images were utilized for four classes: CNV, Drusen, DME, and Normal, using 11 pre-trained models. Similarly, several pre-trained models have been presented to demonstrate the effectiveness of CNNs in detecting various retinal diseases (<xref ref-type="bibr" rid="B29">29</xref>&#x02013;<xref ref-type="bibr" rid="B31">31</xref>).</p>
<p>Moreover, the authors in Jawad et al. (<xref ref-type="bibr" rid="B32">32</xref>) apply four Swin Transformer variants to multi-classify fundus images, utilizing local window self-attention to address the limited global modeling of traditional CNNs. Swin-L outperforms earlier research, with final scores reaching up to 0.97 and an AUC exceeding 0.95 on three ODIR test splits and an external retina dataset, demonstrating strong generalization. Metric standard deviations remain below 0.05, and one-way ANOVA indicates non-significant differences among models (<italic>p</italic> = 0.32&#x02013;0.94), confirming the statistical stability of their results.</p>
<p>Furthermore, in a study by Abdul Jawad and Khursheed (<xref ref-type="bibr" rid="B33">33</xref>), the authors build a DenseNet-based deep-and-dense CNN that classifies BreakHis breast-cancer slides into benign/malignant subtypes across all magnifications, achieving up to 96.6% patient-level and 91.8% image-level accuracy, with <italic>t</italic>-tests showing that the gains over earlier CNNs are significant. Additionally, in another study by Abdul Jawad and Khursheed (<xref ref-type="bibr" rid="B34">34</xref>), the authors introduce an automatic, color-balanced reference-image selector that, when paired with Reinhard, Macenko, and Vahadane normalization, consistently boosts SSIM, QSSIM, and PCC on BreakHis and BACH datasets; Wilcoxon tests confirm that the improvements compared to random selection are also significant.</p>
<p>Furthermore, the authors in Bhandari et al. (<xref ref-type="bibr" rid="B35">35</xref>) proposed a lightweight convolutional neural network with only 983,716 trainable parameters. They employed this architecture to classify OCT images of three retinal pathologies: CNV, DME, and Drusen, achieving a test accuracy of 94.29% and a validation accuracy of 94.12%. To interpret the model&#x00027;s decisions, the authors applied two explainable AI techniques: Local Interpretable Model-Agnostic Explanations (LIME) and Shapley Additive exPlanations (SHAP), which highlighted clinically relevant retinal regions. The same network was subsequently tested on two additional medical imaging tasks: COVID-19 detection from chest X-rays and kidney stone classification. Similarly, in another study by Bhandari et al. (<xref ref-type="bibr" rid="B41">41</xref>), the authors utilized the proposed model with explainable AI techniques, including LIME, Gradient-weighted Class Activation Mapping (Grad-CAM), and SHAP.</p>
<p>Thus, the majority of existing approaches lack interpretability and explanations of the decision-making process, which is extremely important in clinical settings. Therefore, in addition to designing a tailored deep learning model for the efficient and accurate detection of retinal disorders, this paper focuses on integrating explainable AI to visualize and interpret the features on which the identification of retinal disorders is based.</p></sec>
<sec id="s3">
<title>3 Data preparation</title>
<sec>
<title>3.1 Dataset</title>
<p>In this study, retinal OCT images were used to train DL models for the classification of retinal disorders. From the accessed dataset [&#x0201C;Retinal OCT Images (optical coherence tomography)&#x0201D; data (<xref ref-type="bibr" rid="B36">36</xref>)], a balanced, high-quality subset of 6,200 images (1,550 per class) was constructed using a two-step procedure: quality screening and class balancing with computational constraints. During quality screening, many raw files were found to contain white borders or artifacts unrelated to retinal tissue. These were excluded to prevent the models from learning irrelevant or misleading patterns. For class balancing with computational limits, all experiments were conducted on Google Colab, where limited GPU memory could not accommodate the entire dataset without frequent crashes. A balanced subset of 1,550 images per class provided a practical compromise between dataset representativeness and hardware constraints. The dataset comprises four classes: normal, CNV disorder, DME disorder, and Drusen disorder. It was divided into three sets: training, testing, and validation. A two-stage split strategy was used. In the first stage, the dataset was split into 80% for training and 20% for testing. In the second stage, the training set was further divided, with 80% used for model training and 20% for validation. Details of the dataset are provided in <xref ref-type="table" rid="T1">Table 1</xref>, and representative sample images of retinal OCT scans are shown in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Summary of dataset classes and distributions.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#727779;color:#ffffff">
<th valign="top" align="left"><bold>Class name</bold></th>
<th valign="top" align="center" colspan="4"><bold>Splitting details</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#727779;color:#ffffff">
<td/>
<td valign="top" align="center"><bold>Training set</bold></td>
<td valign="top" align="center"><bold>Validation set</bold></td>
<td valign="top" align="center"><bold>Testing set</bold></td>
<td valign="top" align="center"><bold>Total</bold></td>
</tr> <tr>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">992</td>
<td valign="top" align="center">248</td>
<td valign="top" align="center">310</td>
<td valign="top" align="center">1,550</td>
</tr> <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">992</td>
<td valign="top" align="center">248</td>
<td valign="top" align="center">310</td>
<td valign="top" align="center">1,550</td>
</tr> <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">992</td>
<td valign="top" align="center">248</td>
<td valign="top" align="center">310</td>
<td valign="top" align="center">1,550</td>
</tr> <tr>
<td valign="top" align="left">NORMAL</td>
<td valign="top" align="center">992</td>
<td valign="top" align="center">248</td>
<td valign="top" align="center">310</td>
<td valign="top" align="center">1,550</td>
</tr></tbody>
</table>
</table-wrap>



<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Samples of the dataset images of each class. <bold>(a)</bold> CNV-disorder class. <bold>(b)</bold> DME-disorder class. <bold>(c)</bold> Drusen-disorder class. <bold>(d)</bold> Normal-no disorder class.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1596726-g0001.tif"/>
</fig>

</sec>

<sec>
<title>3.2 Data preprocessing and augmentation</title>
<p>To train the DL on the OCT images, the dataset was preprocessed. The OCT images in the dataset primarily contain noisy pixels. Noisy pixels lead to incorrect feature extraction during model training, resulting in underfitting and overfitting issues. To eliminate noisy pixels, a preprocessing function <italic>Prep</italic><sub><italic>func</italic></sub> given in <xref ref-type="disp-formula" rid="E1">Equation 1</xref> was utilized, which normalizes image pixels from the range [0, 255] to [&#x02212;1, 1]. Samples of the preprocessed and rescaled images for each model are displayed in <xref ref-type="fig" rid="F2">Figure 2</xref>. It is evident that there are no noisy pixels (compared to the sample images shown in <xref ref-type="fig" rid="F1">Figure 1</xref>).</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi><mml:mi>u</mml:mi><mml:mi>n</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>x</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mn>255</mml:mn></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>5</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mn>2</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>After scaling the images, data augmentation was applied using a 30&#x000B0; rotation and horizontal flipping. Samples of the augmented images are illustrated in <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Images after preprocessing: <bold>(a)</bold> CNV-disorder class; <bold>(b)</bold> DME-disorder class; <bold>(c)</bold> Drusen-disorder class; and <bold>(d)</bold> Normal-no disorder class.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1596726-g0002.tif"/>
</fig>

<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Data augmentation process.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1596726-g0003.tif"/>
</fig>


</sec></sec>
<sec id="s4">
<title>4 Methodology and experiments</title>
<p>The methodology employed in this study consists of four key stages: data preparation, data augmentation, model training with explainable AI, and model evaluation. An overview of the employed methodology is illustrated in <xref ref-type="fig" rid="F4">Figure 4</xref>. The data preparation and data augmentation processes have already been discussed in the previous section. This section details the model architecture and evaluation parameters.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Utilized methodology for the classification of retinal diseases.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1596726-g0004.tif"/>
</fig>


<sec>
<title>4.1 The proposed OculusNet model</title>
<sec>
<title>4.1.1 Architecture of the proposed model</title>
<p>A DL model named OculusNet has been proposed in this study. The OculusNet model comprises nine depthwise separable (DWS) convolutional layers, four max pooling layers, and four batch normalization layers. A max pool size of 2 &#x000D7; 2 has been kept for this model. For each DWS layer, a kernel size of 3 &#x000D7; 3 has been utilized, and in each 2D separable convolutional layer, the &#x0201C;ReLU&#x0201D; activation function is utilized. The model summary and parameter information for each layer are shown in <xref ref-type="table" rid="T2">Table 2</xref>. The model architecture of the proposed CNN (i.e., OculusNet) is depicted in <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Summary of layer outputs and parameters.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#727779;color:#ffffff">
<th valign="top" align="left"><bold>Layers</bold></th>
<th valign="top" align="center"><bold>Output shape</bold></th>
<th valign="top" align="center"><bold>Parameters</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">separable_conv2d</td>
<td valign="top" align="center">222, 222, 32</td>
<td valign="top" align="center">155</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_1</td>
<td valign="top" align="center">220, 220, 32</td>
<td valign="top" align="center">1,344</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_2</td>
<td valign="top" align="center">218, 218, 32</td>
<td valign="top" align="center">1,344</td>
</tr> <tr>
<td valign="top" align="left">Max_Pooling2d</td>
<td valign="top" align="center">109, 109, 32</td>
<td valign="top" align="center">0</td>
</tr> <tr>
<td valign="top" align="left">Batch_Normalization</td>
<td valign="top" align="center">109, 109, 32</td>
<td valign="top" align="center">128</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_3</td>
<td valign="top" align="center">107, 107, 64</td>
<td valign="top" align="center">2,400</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_4</td>
<td valign="top" align="center">105, 105, 64</td>
<td valign="top" align="center">4,763</td>
</tr> <tr>
<td valign="top" align="left">Max_Pooling2d_1</td>
<td valign="top" align="center">52, 52, 64</td>
<td valign="top" align="center">0</td>
</tr> <tr>
<td valign="top" align="left">Batch_Normalization_1</td>
<td valign="top" align="center">52, 52, 64</td>
<td valign="top" align="center">256</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_5</td>
<td valign="top" align="center">50, 50, 128</td>
<td valign="top" align="center">8,896</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_6</td>
<td valign="top" align="center">48, 48, 128</td>
<td valign="top" align="center">17,664</td>
</tr> <tr>
<td valign="top" align="left">Max_Pooling2d_2</td>
<td valign="top" align="center">24, 24, 128</td>
<td valign="top" align="center">0</td>
</tr> <tr>
<td valign="top" align="left">Batch_Normalization_2</td>
<td valign="top" align="center">24, 24, 128</td>
<td valign="top" align="center">512</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_7</td>
<td valign="top" align="center">22, 22, 256</td>
<td valign="top" align="center">34,176</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_8</td>
<td valign="top" align="center">20, 20, 256</td>
<td valign="top" align="center">68,096</td>
</tr> <tr>
<td valign="top" align="left">Max_Pooling2d_3</td>
<td valign="top" align="center">6, 6, 256</td>
<td valign="top" align="center">0</td>
</tr> <tr>
<td valign="top" align="left">Batch_Normalization_3</td>
<td valign="top" align="center">6, 6, 256</td>
<td valign="top" align="center">1,024</td>
</tr> <tr>
<td valign="top" align="left">Flatten</td>
<td valign="top" align="center">9216</td>
<td valign="top" align="center">0</td>
</tr> <tr>
<td valign="top" align="left">Dense</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">1,179,776</td>
</tr> <tr>
<td valign="top" align="left">Dense_1</td>
<td valign="top" align="center">64</td>
<td valign="top" align="center">8,256</td>
</tr> <tr>
<td valign="top" align="left">Dense_2</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">260</td>
</tr></tbody>
</table>
</table-wrap>

<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Architectural overview of the proposed OculusNet model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1596726-g0005.tif"/>
</fig>


</sec>
<sec>
<title>4.1.2 Depthwise separable convolutional layers</title>
<p>Depthwise Separable Convolution (DWS) is a computationally efficient alternative to standard convolution, designed to reduce the number of trainable parameters and floating-point operations in convolutional neural networks. It decomposes a standard convolution into two distinct operations: <italic>depthwise</italic> convolution and <italic>pointwise</italic> convolution (<xref ref-type="bibr" rid="B37">37</xref>, <xref ref-type="bibr" rid="B38">38</xref>). The visual comparison between standard and DWS convolutional layers is illustrated in <xref ref-type="fig" rid="F6">Figures 6</xref>, <xref ref-type="fig" rid="F7">7</xref>, respectively.</p>




<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Standard convolutional neural layers.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1596726-g0006.tif"/>
</fig>

<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Depthwise convolutional neural layers.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1596726-g0007.tif"/>
</fig>



<p>In a standard convolutional layer, an input image tensor with the shape <italic>A</italic><sub><italic>h</italic></sub>&#x000D7;<italic>A</italic><sub><italic>w</italic></sub>&#x000D7;<italic>N</italic> (where <italic>A</italic><sub><italic>h</italic></sub> and <italic>A</italic><sub><italic>w</italic></sub> represent the height and width, and <italic>N</italic> signifies the number of input channels) is convolved with <italic>n</italic> filters, each sized <italic>K</italic><sub><italic>d</italic></sub>&#x000D7;<italic>K</italic><sub><italic>d</italic></sub>&#x000D7;<italic>N</italic>. This produces an output tensor of shape <inline-formula><mml:math id="M2"><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x000D7;</mml:mo><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x000D7;</mml:mo><mml:mi>n</mml:mi></mml:math></inline-formula>, where <inline-formula><mml:math id="M3"><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math id="M4"><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> depend on the stride and padding. The computational cost for this operation is described by <xref ref-type="disp-formula" rid="E2">Equation 2</xref>:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mi>S</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x000D7;</mml:mo><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>n</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In contrast, depthwise separable convolution divides the operation described above into two stages:</p>
<list list-type="simple">
<list-item><p>1. <bold>Depthwise convolution:</bold> Applies a single <italic>K</italic><sub><italic>d</italic></sub>&#x000D7;<italic>K</italic><sub><italic>d</italic></sub> filter to each input channel (no cross-channel mixing), generating <italic>N</italic> feature maps. The computational cost is as follows:</p></list-item></list>
<disp-formula id="E3"><label>(3)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mi>D</mml:mi><mml:mi>W</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x000D7;</mml:mo><mml:mi>N</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<list list-type="simple">
<list-item><p>2. <bold>Pointwise convolution:</bold> Applies 1 &#x000D7; 1 &#x000D7; <italic>N</italic> filters to combine the output of the depthwise stage across channels, generating <italic>n</italic> output channels. The computational cost is as follows:</p></list-item></list>
<disp-formula id="E4"><label>(4)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mi>W</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>n</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The total cost of a depthwise separable convolution is the sum of <xref ref-type="disp-formula" rid="E3">Equations 3</xref>, <xref ref-type="disp-formula" rid="E4">4</xref>, as indicated in <xref ref-type="disp-formula" rid="E5">Equation 5</xref>:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mi>C</mml:mi><mml:mi>M</mml:mi><mml:mo>=</mml:mo><mml:mi>D</mml:mi><mml:mi>W</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>P</mml:mi><mml:mi>W</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x000D7;</mml:mo><mml:mi>N</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>To understand the efficiency of DWS, the ratio of its computational cost to that of standard convolution (<xref ref-type="disp-formula" rid="E2">Equation 2</xref>) is derived in <xref ref-type="disp-formula" rid="E6">Equation 6</xref>:</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>C</mml:mi><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x000D7;</mml:mo><mml:mi>N</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x000D7;</mml:mo><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>For instance, with <italic>n</italic> &#x0003D; 256 filters and a kernel size of <italic>K</italic><sub><italic>d</italic></sub> &#x0003D; 3, <xref ref-type="disp-formula" rid="E6">Equation 6</xref> yields:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>C</mml:mi><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>256</mml:mn></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>9</mml:mn></mml:mrow></mml:mfrac><mml:mo>&#x02248;</mml:mo><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>115</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>This indicates that depthwise separable convolution requires only about 11.5% of the computational cost of standard convolution while still producing feature representations with comparable effectiveness. This significant reduction in operations and parameters makes DWS particularly suitable for lightweight architectures such as OculusNet, which are intended for deployment in resource-constrained environments such as mobile or web applications.</p>
<p><xref ref-type="fig" rid="F6">Figure 6</xref> illustrates the standard convolution operation, where each filter operates on all input channels simultaneously. In contrast, <xref ref-type="fig" rid="F7">Figure 7</xref> shows the DWS operation, which performs filtering channel-wise followed by channel mixing, effectively decoupling spatial and cross-channel computations.</p></sec>
<sec>
<title>4.1.3 Workflow of the utilized architecture</title>
<p>The coding flow utilized for the experiments was organized into four main stages: data preprocessing, building model architecture, model training, and model evaluation. Initially, the necessary libraries were imported, followed by the dataset, where preprocessing functions were applied to standardize images and split the dataset into training, validation, and testing subsets. Hyperparameters were selected, and data augmentation techniques were employed using an image data generator library to enhance the model&#x00027;s generalization capability. For building the models, the OculusNet model was defined from scratch by adding convolutional, max-pooling, flatten, and dense layers, after which a model summary was printed. Similarly, pre-trained models were imported from Keras libraries with ImageNet weights, followed by adding flatten and dense layers and printing their summaries. During model training, the models were trained on the training dataset and validated on the validation dataset, while loss and accuracy graphs were generated. The model evaluation step involved obtaining the confusion matrices and classification reports for both validation and testing datasets to comprehensively assess the models&#x00027; performance. This structured approach ensures a coherent and efficient workflow, leading to reliable and reproducible results.</p></sec></sec>
<sec>
<title>4.2 Explainable AI using saliency maps</title>
<p>DL models are typically referred to as black box models due to the complicated interpretation of outputs or predicted results from the trained DL models. However, several visualization techniques are available, such as Grad-CAM (<xref ref-type="bibr" rid="B39">39</xref>&#x02013;<xref ref-type="bibr" rid="B41">41</xref>), and saliency maps, which are often known as class activation maps. By using saliency maps, one can compute the effect of each input pixel on the final prediction, highlighting the influential pixels in the image that the model uses to classify the given image. However, Grad-CAM does not calculate pixel by pixel; instead, it generates a heatmap of the input pixels (<xref ref-type="bibr" rid="B39">39</xref>, <xref ref-type="bibr" rid="B40">40</xref>).</p>
<p>In this study, saliency maps have been utilized for class-specific results on images. Mathematically, the saliency map can be explained by <xref ref-type="disp-formula" rid="E8">Equation 8</xref>, where <italic>w</italic> is the weight of each pixel. <italic>S</italic><sub><italic>n</italic></sub> represents the score of the specific class <italic>n</italic>, which is acquired by the trained model. <italic>I</italic> represents the pixel&#x00027;s values of the given image. Additionally, other techniques, such as Grad-CAM, have also been employed (Grad-CAM&#x0002B;&#x0002B; is selected for this case), along with LIME and SHAP.</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M11"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mi>w</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>6</mml:mn><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mn>6</mml:mn><mml:mi>I</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
<sec>
<title>4.3 Transfer learning on pre-trained models for comparison</title>
<p>To ensure a fair comparison, transfer learning has been employed to optimize pre-trained models that are well-known for classification tasks. The models used include VGG-16, VGG-19, MobileNetV2, and DenseNet-121. The VGG16 model consists of 13 convolutional layers and three dense layers, while the VGG19 model comprises 16 convolutional layers with three dense layers. In contrast, MobileNetV2 is a lightweight neural network with fewer parameters, comprising 28 convolutional layers that utilize depthwise separable convolution. Lastly, the DenseNet121 model includes 121 layers organized into four dense blocks. The architectural details of the aforementioned models are shown in <xref ref-type="fig" rid="F8">Figure 8</xref>.</p>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>Architectural overview of the pre-trained model used in this study.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1596726-g0008.tif"/>
</fig>


<p>The aforementioned models are pre-trained on ImageNet, a large dataset comprising 1,000 classes and almost 1,281,167 images. In this study, the convolutional layers of the aforementioned models were frozen to utilize the pre-trained weights, and the fully connected layers were replaced to retrain the model for classifying retinal OCT images. Additionally, to prevent underfitting during the training phase, batch normalization layers and a dropout layer with a size of 0.25 were added to the fully connected layers. The details regarding the layers and parameters of all models are provided in <xref ref-type="table" rid="T3">Table 3</xref>, which includes the input layer size of 224 &#x000D7; 224 &#x000D7; 3 (<italic>height</italic>&#x000D7;<italic>width</italic>&#x000D7;<italic>dimension</italic>) and an output layer size of 4, 1, where 4 represents the four classes and 1 is the final output layer.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Model architectures and layer details.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#727779;color:#ffffff">
<th valign="top" align="left"><bold>Model name</bold></th>
<th valign="top" align="center"><bold>No. of layers</bold></th>
<th valign="top" align="center"><bold>Size of input layer</bold></th>
<th valign="top" align="center"><bold>Size of output layer</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">VGG19</td>
<td valign="top" align="center">19</td>
<td valign="top" align="center">(224, 224, 3)</td>
<td valign="top" align="center">(4, 1)</td>
</tr> <tr>
<td valign="top" align="left">VGG16</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">(224, 224, 3)</td>
<td valign="top" align="center">(4, 1)</td>
</tr> <tr>
<td valign="top" align="left">MobileNet</td>
<td valign="top" align="center">28</td>
<td valign="top" align="center">(224, 224, 3)</td>
<td valign="top" align="center">(4, 1)</td>
</tr> <tr>
<td valign="top" align="left">DenseNet-121</td>
<td valign="top" align="center">121</td>
<td valign="top" align="center">(224, 224, 3)</td>
<td valign="top" align="center">(4, 1)</td>
</tr> <tr>
<td valign="top" align="left">OculusNet</td>
<td valign="top" align="center">12</td>
<td valign="top" align="center">(224, 224, 3)</td>
<td valign="top" align="center">(4, 1)</td>
</tr></tbody>
</table>
</table-wrap>


<p>Moreover, the trainable and non-trainable parameters of the utilized models are detailed in <xref ref-type="table" rid="T4">Table 4</xref>. These parameters are often referred to as model summaries and are derived from the parameters used in the layers of the model. For the pre-trained models, non-trainable parameters were the sum of the frozen layers (frozen with ImageNet weights), while the trainable parameters were the sum of the FC layer parameters (for which the model weights were not frozen). The weights of the model were updated only for the trainable parameters during the training phase. Additionally, the FC layers remain the same for all the utilized models. The first dense layer uses the ReLU activation function with 128 units, and the second dense layer consists of 64 units. The third dense layer, which is the output layer, has four units and uses softmax as the activation function. Softmax, in this context, is responsible for providing the output.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Model parameters and size details.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#727779;color:#ffffff">
<th valign="top" align="left"><bold>Model name</bold></th>
<th valign="top" align="center"><bold>Total parameters</bold></th>
<th valign="top" align="center"><bold>Trainable parameters</bold></th>
<th valign="top" align="center"><bold>Non-trainable parameters</bold></th>
<th valign="top" align="center"><bold>Model size (MB)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">VGG19</td>
<td valign="top" align="center">23,246,340</td>
<td valign="top" align="center">3,220,932</td>
<td valign="top" align="center">20,025,408</td>
<td valign="top" align="center">16.54</td>
</tr> <tr>
<td valign="top" align="left">VGG16</td>
<td valign="top" align="center">17,936,644</td>
<td valign="top" align="center">3,220,932</td>
<td valign="top" align="center">14,715,712</td>
<td valign="top" align="center">26.21</td>
</tr> <tr>
<td valign="top" align="left">MobileNet</td>
<td valign="top" align="center">9,664,132</td>
<td valign="top" align="center">6,433,220</td>
<td valign="top" align="center">3,230,912</td>
<td valign="top" align="center">6.22</td>
</tr> <tr>
<td valign="top" align="left">DenseNet-121</td>
<td valign="top" align="center">13,472,772</td>
<td valign="top" align="center">6,433,220</td>
<td valign="top" align="center">7,039,552</td>
<td valign="top" align="center">26.56</td>
</tr> <tr>
<td valign="top" align="left">OculusNet</td>
<td valign="top" align="center">1,329,023</td>
<td valign="top" align="center">1,328,063</td>
<td valign="top" align="center">960</td>
<td valign="top" align="center">5.07</td>
</tr></tbody>
</table>
</table-wrap>

</sec></sec>
<sec id="s5">
<title>5 Experiment and results</title>
<sec>
<title>5.1 Evaluation parameters</title>
<sec>
<title>5.1.1 Performance parameters</title>
<p>For the experimental parameters, a grid search strategy was used, with the number of epochs set to 100 and the batch size set to 32. A learning rate of 0.00001 with RMS optimizer was applied. For performance evaluation, key performance parameters, including precision, recall, specificity, F1 score, and accuracy, were employed. These parameters were derived from the confusion matrices. The 4 &#x000D7; 4 confusion matrix with character representation is given in <xref ref-type="table" rid="T5">Table 5</xref>. The analysis of the confusion matrix has been presented for both the validation and testing datasets. Considering the characters used in <xref ref-type="table" rid="T5">Table 5</xref>, the TP and TN samples for each respective class are listed in <xref ref-type="table" rid="T6">Table 6</xref>. Similarly, the FP and FN samples for each respective class are provided in <xref ref-type="table" rid="T7">Table 7</xref>.</p>


<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Representation of the confusion matrix.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#727779;color:#ffffff">
<th valign="top" align="left" colspan="5"><bold>Confusion matrix</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#727779;color:#ffffff">
<td/>
<td valign="top" align="center" colspan="4"><bold>Predicted classes</bold></td>
</tr>
 <tr style="background-color:#727779;color:#ffffff">
<td valign="top" align="left"><bold>Actual classes</bold></td>
<td valign="top" align="center"><bold>CNV</bold></td>
<td valign="top" align="center"><bold>DME</bold></td>
<td valign="top" align="center"><bold>Drusen</bold></td>
<td valign="top" align="center"><bold>NORMAL</bold></td>
</tr> <tr>
<td valign="top" align="left"><bold>CNV</bold></td>
<td valign="top" align="center">TP<sub><italic>CNV</italic></sub></td>
<td valign="top" align="center">F<sub><italic>AB</italic></sub></td>
<td valign="top" align="center">F<sub><italic>AC</italic></sub></td>
<td valign="top" align="center">F<sub><italic>AD</italic></sub></td>
</tr> <tr>
<td valign="top" align="left"><bold>DME</bold></td>
<td valign="top" align="center">F<sub><italic>BA</italic></sub></td>
<td valign="top" align="center">TP<sub><italic>DME</italic></sub></td>
<td valign="top" align="center">F<sub><italic>BC</italic></sub></td>
<td valign="top" align="center">F<sub><italic>BD</italic></sub></td>
</tr> <tr>
<td valign="top" align="left"><bold>Drusen</bold></td>
<td valign="top" align="center">F<sub><italic>CA</italic></sub></td>
<td valign="top" align="center">F<sub><italic>CB</italic></sub></td>
<td valign="top" align="center">TP<sub><italic>Drusen</italic></sub></td>
<td valign="top" align="center">F<sub><italic>CD</italic></sub></td>
</tr> <tr>
<td valign="top" align="left"><bold>NORMAL</bold></td>
<td valign="top" align="center">F<sub><italic>DA</italic></sub></td>
<td valign="top" align="center">F<sub><italic>DB</italic></sub></td>
<td valign="top" align="center">F<sub><italic>DC</italic></sub></td>
<td valign="top" align="center">TP<sub><italic>NORMAL</italic></sub></td>
</tr></tbody>
</table>
</table-wrap>

<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Class-wise true positive and true negative.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#727779;color:#ffffff">
<th valign="top" align="left"><bold>Class</bold></th>
<th valign="top" align="center"><bold>True positive (TP)</bold></th>
<th valign="top" align="center"><bold>True negative (TN)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><bold>CNV</bold></td>
<td valign="top" align="center">TP<sub><italic>CNV</italic></sub></td>
<td valign="top" align="center">TP<sub><italic>DME</italic></sub> &#x0002B; F<sub><italic>BC</italic></sub> &#x0002B; F<sub><italic>BD</italic></sub> &#x0002B; F<sub><italic>CB</italic></sub> &#x0002B; TP<sub><italic>Drusen</italic></sub> &#x0002B; F<sub><italic>CD</italic></sub> &#x0002B; F<sub><italic>DB</italic></sub> &#x0002B; F<sub><italic>DC</italic></sub> &#x0002B; TP<sub><italic>NORMAL</italic></sub></td>
</tr> <tr>
<td valign="top" align="left"><bold>DME</bold></td>
<td valign="top" align="center">TP<sub><italic>DME</italic></sub></td>
<td valign="top" align="center">TP<sub><italic>CNV</italic></sub> &#x0002B; F<sub><italic>AC</italic></sub> &#x0002B; F<sub><italic>AD</italic></sub> &#x0002B; F<sub><italic>CA</italic></sub> &#x0002B; TP<sub><italic>Drusen</italic></sub> &#x0002B; F<sub><italic>CD</italic></sub> &#x0002B; F<sub><italic>DA</italic></sub> &#x0002B; F<sub><italic>DC</italic></sub> &#x0002B; TP<sub><italic>NORMAL</italic></sub></td>
</tr> <tr>
<td valign="top" align="left"><bold>Drusen</bold></td>
<td valign="top" align="center">TP<sub><italic>Drusen</italic></sub></td>
<td valign="top" align="center">TP<sub><italic>CNV</italic></sub> &#x0002B; F<sub><italic>AB</italic></sub> &#x0002B; F<sub><italic>AD</italic></sub> &#x0002B; F<sub><italic>BA</italic></sub> &#x0002B; TP<sub><italic>DME</italic></sub> &#x0002B; F<sub><italic>BD</italic></sub> &#x0002B; F<sub><italic>DA</italic></sub> &#x0002B; F<sub><italic>DB</italic></sub> &#x0002B; TP<sub><italic>NORMAL</italic></sub></td>
</tr> <tr>
<td valign="top" align="left"><bold>NORMAL</bold></td>
<td valign="top" align="center">TP<sub><italic>NORMAL</italic></sub></td>
<td valign="top" align="center">TP<sub><italic>CNV</italic></sub> &#x0002B; F<sub><italic>AB</italic></sub> &#x0002B; F<sub><italic>AC</italic></sub> &#x0002B; F<sub><italic>BA</italic></sub> &#x0002B; TP<sub><italic>DME</italic></sub> &#x0002B; F<sub><italic>BC</italic></sub> &#x0002B; F<sub><italic>CA</italic></sub> &#x0002B; F<sub><italic>CB</italic></sub> &#x0002B; TP<sub><italic>Drusen</italic></sub></td>
</tr></tbody>
</table>
</table-wrap>

<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>Class-wise false positives and false negatives.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#727779;color:#ffffff">
<th valign="top" align="left"><bold>Class</bold></th>
<th valign="top" align="center"><bold>False positive (FP)</bold></th>
<th valign="top" align="center"><bold>False negative (FN)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><bold>CNV</bold></td>
<td valign="top" align="center">F<sub><italic>BA</italic></sub> &#x0002B; F<sub><italic>CA</italic></sub> &#x0002B; F<sub><italic>DA</italic></sub></td>
<td valign="top" align="center">F<sub><italic>AB</italic></sub> &#x0002B; F<sub><italic>AC</italic></sub> &#x0002B; F<sub><italic>AD</italic></sub></td>
</tr> <tr>
<td valign="top" align="left"><bold>DME</bold></td>
<td valign="top" align="center">F<sub><italic>AB</italic></sub> &#x0002B; F<sub><italic>CB</italic></sub> &#x0002B; F<sub><italic>DB</italic></sub></td>
<td valign="top" align="center">F<sub><italic>BA</italic></sub> &#x0002B; F<sub><italic>BC</italic></sub> &#x0002B; F<sub><italic>BD</italic></sub></td>
</tr> <tr>
<td valign="top" align="left"><bold>Drusen</bold></td>
<td valign="top" align="center">F<sub><italic>AC</italic></sub> &#x0002B; F<sub><italic>BC</italic></sub> &#x0002B; F<sub><italic>DC</italic></sub></td>
<td valign="top" align="center">F<sub><italic>CA</italic></sub> &#x0002B; F<sub><italic>CB</italic></sub> &#x0002B; F<sub><italic>CD</italic></sub></td>
</tr> <tr>
<td valign="top" align="left"><bold>NORMAL</bold></td>
<td valign="top" align="center">F<sub><italic>AD</italic></sub> &#x0002B; F<sub><italic>BD</italic></sub> &#x0002B; F<sub><italic>CD</italic></sub></td>
<td valign="top" align="center">F<sub><italic>DA</italic></sub> &#x0002B; F<sub><italic>DB</italic></sub> &#x0002B; F<sub><italic>DC</italic></sub></td>
</tr></tbody>
</table>
</table-wrap>




<p><bold>Accuracy</bold> represents the model&#x00027;s overall ability to correctly classify the TP and TN classes from all the predicted labels. It can be calculated using <xref ref-type="disp-formula" rid="E9">Equation 9</xref>.</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mstyle class="mbox"><mml:mtext>Accuracy</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mo>&#x000D7;</mml:mo><mml:mn>100</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><bold>Precision</bold> measures the actual TP samples from all the positive predicted samples of the respective class. It can be calculated using <xref ref-type="disp-formula" rid="E10">Equation 10</xref>.</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M13"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mstyle class="mbox"><mml:mtext>Precision</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><bold>Recall</bold> is also commonly known as sensitivity. It measures the actual TP samples from the considered predicted samples of that class. It can be calculated using <xref ref-type="disp-formula" rid="E11">Equation 11</xref>.</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mstyle class="mbox"><mml:mtext>Recall</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi><mml:mi>N</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><bold>Specificity</bold> measures the TN samples from the predicted samples of the classes. It can be calculated using <xref ref-type="disp-formula" rid="E12">Equation 12</xref>.</p>
<disp-formula id="E12"><label>(12)</label><mml:math id="M15"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mstyle class="mbox"><mml:mtext>Specificity</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><bold>F1-Score</bold> describes the actual predicted results obtained through precision and recall. It can be calculated using <xref ref-type="disp-formula" rid="E13">Equation 13</xref>.</p>
<disp-formula id="E13"><label>(13)</label><mml:math id="M16"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mstyle class="mbox"><mml:mtext>score</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mfrac><mml:mrow><mml:mstyle class="mbox"><mml:mtext>Precision</mml:mtext></mml:mstyle><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x000D7;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mstyle class="mbox"><mml:mtext>Recall</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>Precision</mml:mtext></mml:mstyle><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mstyle class="mbox"><mml:mtext>Recall</mml:mtext></mml:mstyle></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></sec>
<sec>
<title>5.1.2 Reliability parameters</title>
<p>To assess the model&#x00027;s reliability on unseen data and its generalization, reliability parameters such as Cohen&#x00027;s kappa coefficient and Matthews correlation coefficient have been calculated.</p>
<p><bold>Cohen&#x00027;s Kappa coefficient</bold></p>
<p>Cohen&#x00027;s Kappa statistic (kappa) assesses a model&#x00027;s performance by measuring the agreement between predicted and actual labels, while accounting for the agreement that could happen by chance. It indicates how accurate the predictions are, or how close they are to the actual value of their respective labels. It is computed using the observed and expected accuracy values extracted from the confusion matrices. Observed accuracy corresponds to the ratio of accurately predicted values to the total number of values in the confusion matrix. <xref ref-type="disp-formula" rid="E14">Equation 14</xref> represents the mathematical representation for observed accuracy. Expected accuracy is defined as a random accuracy, and <xref ref-type="disp-formula" rid="E15">Equation 15</xref> shows the mathematical representation of expected accuracy, where the sum of total values in the predicted row for a respective class is multiplied by the sum of total values of the actual class for the respective class, and <italic>i</italic>&#x02208;0, 1, 2, 3 represents the four classes: 0: CNV, 1: DME, 2: Drusen, and 3: NORMAL used in this study. The combination of observed accuracy versus expected accuracy is known as the kappa statistic, and <xref ref-type="disp-formula" rid="E16">Equation 16</xref> represents the kappa statistic.</p>
<disp-formula id="E14"><label>(14)</label><mml:math id="M17"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mtext>Observed&#x000A0;accuracy</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E15"><label>(15)</label><mml:math id="M18"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mtext>Expected&#x000A0;accuracy</mml:mtext><mml:mo>=</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E16"><label>(16)</label><mml:math id="M20"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mtext>Kappa</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle class="mbox"><mml:mtext>Observed Accuracy</mml:mtext></mml:mstyle><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x02212;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mstyle class="mbox"><mml:mtext>Expected Accuracy</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x02212;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mstyle class="mbox"><mml:mtext>Expected Accuracy</mml:mtext></mml:mstyle></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><bold>Matthews correlation coefficient</bold></p>
<p>The Matthews correlation coefficient (MCC) is used to assess the correlation between predicted and true binary classification labels, taking into account all four confusion matrix categories (TP, TN, FP, FN). The range of MCC lies between &#x02013;1 and 1, where 1 indicates that a model is perfectly positive and capable of classifying positive samples with greater accuracy. Conversely, &#x02013;1 indicates that the model has a negative correlation and, in most cases, will misclassify positive samples. Thus, &#x02013;1 represents the worst-case scenario for a model, which will be unable to classify the samples correctly. <xref ref-type="disp-formula" rid="E17">Equation 17</xref> represents the mathematical formulation of MCC, and <italic>i</italic> &#x02208;0, 1, 2, 3 represents the four classes, namely 0: CNV, 1: DME, 2: Drusen, and 3: NORMAL, used in this study.</p>
<disp-formula id="E17"><label>(17)</label><mml:math id="M21"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mi>M</mml:mi><mml:mi>C</mml:mi><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:mi>T</mml:mi><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>F</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;where,&#x000A0;</mml:mtext><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn><mml:mo>&#x000A0;</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec></sec>
<sec>
<title>5.2 Results and discussion</title>
<p>This section provides the results of the training performance, the key performance parameters, and the reliability parameters of the models. In addition, an ablation study has also been performed on the proposed OculusNet model to validate it.</p>
<sec>
<title>5.2.1 Training and validation results</title>
<p>The training performance of the proposed OculusNet model, as well as the models used for comparison, has been evaluated and reported in terms of validation and test accuracy. The training curves for the proposed OculusNet model, along with the other utilized models, are shown in <xref ref-type="fig" rid="F9">Figure 9</xref>. The curves demonstrate that the models are well-trained, exhibiting no data bias, underfitting, or overfitting. Each model achieved over 90% validation and training accuracy.</p>
<fig id="F9" position="float">
<label>Figure 9</label>
<caption><p>Accuracy and Loss curves. <bold>(a)</bold> DenseNet121. <bold>(b)</bold> MobileNetV2. <bold>(c)</bold> VGG16. <bold>(d)</bold> VGG19. <bold>(e)</bold> OculusNet.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1596726-g0009.tif"/>
</fig>

</sec>
<sec>
<title>5.2.2 Class-wise classification report</title>
<p>The confusion matrices obtained from the validation dataset are shown in <xref ref-type="fig" rid="F10">Figure 10</xref>. In these confusion matrices, the labels on the y-axis represent the actual labels, while the labels on the x-axis represent the predicted number of images for the respective classes. Moreover, <xref ref-type="table" rid="T8">Table 8</xref> provides a classification report for each model. These parameter values are obtained after the training of the model and can be validated using <xref ref-type="disp-formula" rid="E9">Equations 9</xref>&#x02013;<xref ref-type="disp-formula" rid="E13">13</xref>.</p>


<fig id="F10" position="float">
<label>Figure 10</label>
<caption><p>Confusion matrices for the validation dataset for each model. <bold>(a)</bold> DenseNet121. <bold>(b)</bold> VGG16. <bold>(c)</bold> MobileNetV2. <bold>(d)</bold> VGG19. <bold>(e)</bold> OculusNet.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1596726-g0010.tif"/>
</fig>


<table-wrap position="float" id="T8">
<label>Table 8</label>
<caption><p>Performance parameters of the models used for the validation dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#727779;color:#ffffff">
<th valign="top" align="left"><bold>Models</bold></th>
<th valign="top" align="center"><bold>Classes</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>F1-Score</bold></th>
<th valign="top" align="center"><bold>Specificity</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="4">DenseNet121</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.983</td>
<td valign="top" align="center">0.967</td>
<td valign="top" align="center">0.974</td>
<td valign="top" align="center">0.994</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.972</td>
<td valign="top" align="center">0.987</td>
<td valign="top" align="center">0.979</td>
<td valign="top" align="center">0.990</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.932</td>
<td valign="top" align="center">0.943</td>
<td valign="top" align="center">0.937</td>
<td valign="top" align="center">0.979</td>
</tr>
 <tr>
<td valign="top" align="left">Normal</td>
<td valign="top" align="center">0.959</td>
<td valign="top" align="center">0.947</td>
<td valign="top" align="center">0.952</td>
<td valign="top" align="center">0.986</td>
</tr> <tr>
<td valign="top" align="left" rowspan="4">VGG16</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.950</td>
<td valign="top" align="center">0.919</td>
<td valign="top" align="center">0.934</td>
<td valign="top" align="center">0.983</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.955</td>
<td valign="top" align="center">0.951</td>
<td valign="top" align="center">0.952</td>
<td valign="top" align="center">0.985</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.866</td>
<td valign="top" align="center">0.919</td>
<td valign="top" align="center">0.891</td>
<td valign="top" align="center">0.952</td>
</tr>
 <tr>
<td valign="top" align="left">Normal</td>
<td valign="top" align="center">0.921</td>
<td valign="top" align="center">0.899</td>
<td valign="top" align="center">0.909</td>
<td valign="top" align="center">0.974</td>
</tr> <tr>
<td valign="top" align="left" rowspan="4">MobileNetV2</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.983</td>
<td valign="top" align="center">0.979</td>
<td valign="top" align="center">0.98</td>
<td valign="top" align="center">0.994</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.995</td>
<td valign="top" align="center">0.995</td>
<td valign="top" align="center">0.995</td>
<td valign="top" align="center">0.998</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.955</td>
<td valign="top" align="center">0.955</td>
<td valign="top" align="center">0.955</td>
<td valign="top" align="center">0.985</td>
</tr>
 <tr>
<td valign="top" align="left">Normal</td>
<td valign="top" align="center">0.971</td>
<td valign="top" align="center">0.975</td>
<td valign="top" align="center">0.972</td>
<td valign="top" align="center">0.990</td>
</tr> <tr>
<td valign="top" align="left" rowspan="4">VGG19</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.939</td>
<td valign="top" align="center">0.931</td>
<td valign="top" align="center">0.935</td>
<td valign="top" align="center">0.980</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.942</td>
<td valign="top" align="center">0.915</td>
<td valign="top" align="center">0.928</td>
<td valign="top" align="center">0.981</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.863</td>
<td valign="top" align="center">0.887</td>
<td valign="top" align="center">0.875</td>
<td valign="top" align="center">0.953</td>
</tr>
 <tr>
<td valign="top" align="left">Normal</td>
<td valign="top" align="center">0.908</td>
<td valign="top" align="center">0.915</td>
<td valign="top" align="center">0.912</td>
<td valign="top" align="center">0.969</td>
</tr> <tr>
<td valign="top" align="left" rowspan="4">OculusNet</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.987</td>
<td valign="top" align="center">0.987</td>
<td valign="top" align="center">0.987</td>
<td valign="top" align="center">0.995</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.988</td>
<td valign="top" align="center">0.995</td>
<td valign="top" align="center">0.991</td>
<td valign="top" align="center">0.995</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.979</td>
<td valign="top" align="center">0.979</td>
<td valign="top" align="center">0.979</td>
<td valign="top" align="center">0.993</td>
</tr>
 <tr>
<td valign="top" align="left">Normal</td>
<td valign="top" align="center">0.987</td>
<td valign="top" align="center">0.979</td>
<td valign="top" align="center">0.982</td>
<td valign="top" align="center">0.995</td>
</tr></tbody>
</table>
</table-wrap>




<p>Considering the results presented in <xref ref-type="table" rid="T8">Table 8</xref> and the obtained confusion matrices shown in <xref ref-type="fig" rid="F10">Figure 10</xref>, the validation accuracies of each model &#x0201C;VGG19, DenseNet121, MobileNetV2, and VGG16&#x0201D; are 91.03%, 97.18%, 97.48%, and 92.14%, respectively. However, the proposed model, OculusNet, outperformed the pre-trained models, achieving a validation accuracy of 98.59%.</p>
<p>Similarly, the confusion matrices (<xref ref-type="fig" rid="F11">Figure 11</xref>) and classification report (<xref ref-type="table" rid="T9">Table 9</xref>) for the testing dataset have been obtained. The test dataset remains unseen and was not exposed to the models during the training phase.</p>


<fig id="F11" position="float">
<label>Figure 11</label>
<caption><p>Confusion matrices on the test dataset for each model <bold>(a)</bold> DenseNet121. <bold>(b)</bold> VGG16. <bold>(c)</bold> MobileNetV2. <bold>(d)</bold> VGG19. <bold>(e)</bold> OculusNet.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1596726-g0011.tif"/>
</fig>

<table-wrap position="float" id="T9">
<label>Table 9</label>
<caption><p>Performance parameters of the utilized models for the test dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#727779;color:#ffffff">
<th valign="top" align="left"><bold>Models</bold></th>
<th valign="top" align="center"><bold>Classes</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>F1-Score</bold></th>
<th valign="top" align="center"><bold>Specificity</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="4">DenseNet121</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.960</td>
<td valign="top" align="center">0.926</td>
<td valign="top" align="center">0.943</td>
<td valign="top" align="center">0.987</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.956</td>
<td valign="top" align="center">0.971</td>
<td valign="top" align="center">0.963</td>
<td valign="top" align="center">0.985</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.888</td>
<td valign="top" align="center">0.919</td>
<td valign="top" align="center">0.903</td>
<td valign="top" align="center">0.961</td>
</tr>
 <tr>
<td valign="top" align="left">Normal</td>
<td valign="top" align="center">0.925</td>
<td valign="top" align="center">0.910</td>
<td valign="top" align="center">0.917</td>
<td valign="top" align="center">0.975</td>
</tr> <tr>
<td valign="top" align="left" rowspan="4">VGG16</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.946</td>
<td valign="top" align="center">0.852</td>
<td valign="top" align="center">0.896</td>
<td valign="top" align="center">0.984</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.897</td>
<td valign="top" align="center">0.929</td>
<td valign="top" align="center">0.913</td>
<td valign="top" align="center">0.965</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.809</td>
<td valign="top" align="center">0.903</td>
<td valign="top" align="center">0.854</td>
<td valign="top" align="center">0.929</td>
</tr>
 <tr>
<td valign="top" align="left">Normal</td>
<td valign="top" align="center">0.901</td>
<td valign="top" align="center">0.855</td>
<td valign="top" align="center">0.877</td>
<td valign="top" align="center">0.969</td>
</tr> <tr>
<td valign="top" align="left" rowspan="4">MobileNetV2</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.969</td>
<td valign="top" align="center">0.925</td>
<td valign="top" align="center">0.947</td>
<td valign="top" align="center">0.990</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.953</td>
<td valign="top" align="center">0.980</td>
<td valign="top" align="center">0.966</td>
<td valign="top" align="center">0.983</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.908</td>
<td valign="top" align="center">0.929</td>
<td valign="top" align="center">0.918</td>
<td valign="top" align="center">0.968</td>
</tr>
 <tr>
<td valign="top" align="left">Normal</td>
<td valign="top" align="center">0.938</td>
<td valign="top" align="center">0.935</td>
<td valign="top" align="center">0.935</td>
<td valign="top" align="center">0.979</td>
</tr> <tr>
<td valign="top" align="left" rowspan="4">VGG19</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.912</td>
<td valign="top" align="center">0.877</td>
<td valign="top" align="center">0.894</td>
<td valign="top" align="center">0.972</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.930</td>
<td valign="top" align="center">0.912</td>
<td valign="top" align="center">0.921</td>
<td valign="top" align="center">0.974</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.800</td>
<td valign="top" align="center">0.880</td>
<td valign="top" align="center">0.838</td>
<td valign="top" align="center">0.926</td>
</tr>
 <tr>
<td valign="top" align="left">Normal</td>
<td valign="top" align="center">0.885</td>
<td valign="top" align="center">0.848</td>
<td valign="top" align="center">0.866</td>
<td valign="top" align="center">0.963</td>
</tr> <tr>
<td valign="top" align="left" rowspan="4">OculusNet</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.983</td>
<td valign="top" align="center">0.945</td>
<td valign="top" align="center">0.963</td>
<td valign="top" align="center">0.994</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.929</td>
<td valign="top" align="center">0.980</td>
<td valign="top" align="center">0.953</td>
<td valign="top" align="center">0.975</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.950</td>
<td valign="top" align="center">0.938</td>
<td valign="top" align="center">0.943</td>
<td valign="top" align="center">0.983</td>
</tr>
 <tr>
<td valign="top" align="left">Normal</td>
<td valign="top" align="center">0.957</td>
<td valign="top" align="center">0.954</td>
<td valign="top" align="center">0.955</td>
<td valign="top" align="center">0.986</td>
</tr></tbody>
</table>
</table-wrap>





<p>Considering the results from these models as shown in <xref ref-type="table" rid="T9">Table 9</xref> and <xref ref-type="fig" rid="F11">Figure 11</xref>, the test accuracy for each model &#x0201C;VGG19, DenseNet121, MobileNetV2, and VGG16&#x0201D; was 87.98%, 93.15%, 94.19%, and 88.59%, respectively. The proposed model achieved the highest test accuracy of 95.48%. These results indicate that the proposed model outperformed all other pre-trained models with the highest validation and testing accuracies.</p></sec>
<sec>
<title>5.2.3 Reliability parameters results</title>
<p>The results of the reliability parameters are presented in <xref ref-type="table" rid="T10">Table 10</xref>, demonstrating that the proposed OculusNet model outperformed across all classes and metrics, with observed accuracies ranging from 0.989 to 0.995 on the validation set and 0.972 to 0.982 on the test set, along with high Kappa and MCC values. This suggests that OculusNet&#x00027;s architecture is particularly effective for feature extraction and generalization capabilities suited to the specific characteristics of retinal disease images. The Kappa statistic and MCC values across all models and classes were high, indicating a strong agreement between the predicted and actual labels. These metrics highlight the models&#x00027; ability to accurately differentiate among various classes, which is crucial in medical diagnostics, where the stakes are high due to the potential consequences of misdiagnosis. The high Kappa and MCC values also imply that the models were well-trained, providing a reliable assessment of their predictive performance.</p>

<table-wrap position="float" id="T10">
<label>Table 10</label>
<caption><p>Kappa coefficient and Matthews correlation coefficient for all models.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#727779;color:#ffffff">
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>Models</bold></th>
<th valign="top" align="center"><bold>Classes</bold></th>
<th valign="top" align="center"><bold>Expected Acc</bold>.</th>
<th valign="top" align="center"><bold>Observed Acc</bold>.</th>
<th valign="top" align="center"><bold>Kappa</bold></th>
<th valign="top" align="center"><bold>MCC</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="20">Valid</td>
<td valign="top" align="left" rowspan="4">DenseNet-121</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.627</td>
<td valign="top" align="center">0.987</td>
<td valign="top" align="center">0.965</td>
<td valign="top" align="center">0.967</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.622</td>
<td valign="top" align="center">0.989</td>
<td valign="top" align="center">0.970</td>
<td valign="top" align="center">0.973</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.623</td>
<td valign="top" align="center">0.968</td>
<td valign="top" align="center">0.915</td>
<td valign="top" align="center">0.917</td>
</tr>
 <tr>
<td valign="top" align="left">NORMAL</td>
<td valign="top" align="center">0.626</td>
<td valign="top" align="center">0.976</td>
<td valign="top" align="center">0.935</td>
<td valign="top" align="center">0.937</td>
</tr>
 <tr>
<td valign="top" align="left" rowspan="4">VGG16</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.629</td>
<td valign="top" align="center">0.967</td>
<td valign="top" align="center">0.911</td>
<td valign="top" align="center">0.913</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.625</td>
<td valign="top" align="center">0.976</td>
<td valign="top" align="center">0.936</td>
<td valign="top" align="center">0.938</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.617</td>
<td valign="top" align="center">0.944</td>
<td valign="top" align="center">0.853</td>
<td valign="top" align="center">0.855</td>
</tr>
 <tr>
<td valign="top" align="left">NORMAL</td>
<td valign="top" align="center">0.628</td>
<td valign="top" align="center">0.955</td>
<td valign="top" align="center">0.879</td>
<td valign="top" align="center">0.880</td>
</tr>
 <tr>
<td valign="top" align="left" rowspan="4">MobileNetV2</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.625</td>
<td valign="top" align="center">0.990</td>
<td valign="top" align="center">0.973</td>
<td valign="top" align="center">0.975</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.625</td>
<td valign="top" align="center">0.997</td>
<td valign="top" align="center">0.992</td>
<td valign="top" align="center">0.994</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.625</td>
<td valign="top" align="center">0.977</td>
<td valign="top" align="center">0.938</td>
<td valign="top" align="center">0.940</td>
</tr>
 <tr>
<td valign="top" align="left">NORMAL</td>
<td valign="top" align="center">0.624</td>
<td valign="top" align="center">0.986</td>
<td valign="top" align="center">0.962</td>
<td valign="top" align="center">0.965</td>
</tr>
 <tr>
<td valign="top" align="left" rowspan="4">VGG19</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.626</td>
<td valign="top" align="center">0.967</td>
<td valign="top" align="center">0.911</td>
<td valign="top" align="center">0.913</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.628</td>
<td valign="top" align="center">0.964</td>
<td valign="top" align="center">0.903</td>
<td valign="top" align="center">0.905</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.621</td>
<td valign="top" align="center">0.936</td>
<td valign="top" align="center">0.831</td>
<td valign="top" align="center">0.832</td>
</tr>
 <tr>
<td valign="top" align="left">NORMAL</td>
<td valign="top" align="center">0.623</td>
<td valign="top" align="center">0.955</td>
<td valign="top" align="center">0.880</td>
<td valign="top" align="center">0.882</td>
</tr>
 <tr>
<td valign="top" align="left" rowspan="4">OculusNet</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.625</td>
<td valign="top" align="center">0.993</td>
<td valign="top" align="center">0.981</td>
<td valign="top" align="center">0.983</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.623</td>
<td valign="top" align="center">0.995</td>
<td valign="top" align="center">0.986</td>
<td valign="top" align="center">0.989</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.625</td>
<td valign="top" align="center">0.989</td>
<td valign="top" align="center">0.970</td>
<td valign="top" align="center">0.971</td>
</tr>
 <tr>
<td valign="top" align="left">NORMAL</td>
<td valign="top" align="center">0.626</td>
<td valign="top" align="center">0.991</td>
<td valign="top" align="center">0.975</td>
<td valign="top" align="center">0.978</td>
</tr> <tr>
<td valign="top" align="left" rowspan="20"><bold>Test</bold></td>
<td valign="top" align="left" rowspan="4">DenseNet-121</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.629</td>
<td valign="top" align="center">0.971</td>
<td valign="top" align="center">0.921</td>
<td valign="top" align="center">0.924</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.622</td>
<td valign="top" align="center">0.981</td>
<td valign="top" align="center">0.949</td>
<td valign="top" align="center">0.950</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.620</td>
<td valign="top" align="center">0.950</td>
<td valign="top" align="center">0.868</td>
<td valign="top" align="center">0.870</td>
</tr>
 <tr>
<td valign="top" align="left">NORMAL</td>
<td valign="top" align="center">0.627</td>
<td valign="top" align="center">0.958</td>
<td valign="top" align="center">0.887</td>
<td valign="top" align="center">0.889</td>
</tr>
 <tr>
<td valign="top" align="left" rowspan="4">VGG16</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.637</td>
<td valign="top" align="center">0.950</td>
<td valign="top" align="center">0.862</td>
<td valign="top" align="center">0.866</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.620</td>
<td valign="top" align="center">0.955</td>
<td valign="top" align="center">0.881</td>
<td valign="top" align="center">0.883</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.610</td>
<td valign="top" align="center">0.922</td>
<td valign="top" align="center">0.800</td>
<td valign="top" align="center">0.803</td>
</tr>
 <tr>
<td valign="top" align="left">NORMAL</td>
<td valign="top" align="center">0.631</td>
<td valign="top" align="center">0.940</td>
<td valign="top" align="center">0.837</td>
<td valign="top" align="center">0.838</td>
</tr>
 <tr>
<td valign="top" align="left" rowspan="4">MobileNetV2</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.630</td>
<td valign="top" align="center">0.974</td>
<td valign="top" align="center">0.929</td>
<td valign="top" align="center">0.930</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.621</td>
<td valign="top" align="center">0.983</td>
<td valign="top" align="center">0.955</td>
<td valign="top" align="center">0.955</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.622</td>
<td valign="top" align="center">0.958</td>
<td valign="top" align="center">0.888</td>
<td valign="top" align="center">0.891</td>
</tr>
 <tr>
<td valign="top" align="left">NORMAL</td>
<td valign="top" align="center">0.625</td>
<td valign="top" align="center">0.967</td>
<td valign="top" align="center">0.912</td>
<td valign="top" align="center">0.913</td>
</tr>
 <tr>
<td valign="top" align="left" rowspan="4">VGG19</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.629</td>
<td valign="top" align="center">0.948</td>
<td valign="top" align="center">0.859</td>
<td valign="top" align="center">0.860</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.627</td>
<td valign="top" align="center">0.961</td>
<td valign="top" align="center">0.895</td>
<td valign="top" align="center">0.896</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.612</td>
<td valign="top" align="center">0.915</td>
<td valign="top" align="center">0.780</td>
<td valign="top" align="center">0.783</td>
</tr>
 <tr>
<td valign="top" align="left">NORMAL</td>
<td valign="top" align="center">0.630</td>
<td valign="top" align="center">0.934</td>
<td valign="top" align="center">0.821</td>
<td valign="top" align="center">0.823</td>
</tr>
 <tr>
<td valign="top" align="left" rowspan="4">OculusNet</td>
<td valign="top" align="left">CNV</td>
<td valign="top" align="center">0.629</td>
<td valign="top" align="center">0.982</td>
<td valign="top" align="center">0.951</td>
<td valign="top" align="center">0.952</td>
</tr>
 <tr>
<td valign="top" align="left">DME</td>
<td valign="top" align="center">0.618</td>
<td valign="top" align="center">0.976</td>
<td valign="top" align="center">0.937</td>
<td valign="top" align="center">0.939</td>
</tr>
 <tr>
<td valign="top" align="left">Drusen</td>
<td valign="top" align="center">0.626</td>
<td valign="top" align="center">0.972</td>
<td valign="top" align="center">0.925</td>
<td valign="top" align="center">0.926</td>
</tr>
 <tr>
<td valign="top" align="left">NORMAL</td>
<td valign="top" align="center">0.625</td>
<td valign="top" align="center">0.978</td>
<td valign="top" align="center">0.941</td>
<td valign="top" align="center">0.941</td>
</tr></tbody>
</table>
</table-wrap>

</sec>
<sec>
<title>5.2.4 Saliency map results</title>
<p>By utilizing <xref ref-type="disp-formula" rid="E8">Equation 8</xref>, the strongest pixel values are calculated from the given input image and then used to display the saliency map. Moreover, the input images used to generate the saliency map are first rescaled, and their pixel values are normalized to a range between 0 and 1 because our model was trained on rescaled and preprocessed images. The saliency map of each class on the OculusNet model is shown in <xref ref-type="fig" rid="F12">Figure 12</xref>. Furthermore, for a fair comparison, Grad-CAM&#x0002B;&#x0002B; and LIME results have also been presented for each class in <xref ref-type="fig" rid="F12">Figure 12</xref>. From the comparison, it can be observed that the saliency map and Grad-CAM&#x0002B;&#x0002B; show that the model was more focused on the retina layers, unlike LIME. Moreover, it is also a limitation of this study that LIME and Grad-CAM have not been optimized further; thus, as a future direction, these techniques will be explored. Additionally, SHAP analysis for each class is shown in <xref ref-type="fig" rid="F13">Figure 13</xref>. From these heatmaps, it can be observed that the model is trying to focus on the retina layers for decision-making.</p>


<fig id="F12" position="float">
<label>Figure 12</label>
<caption><p>Saliency maps, Gradcam&#x0002B;&#x0002B;, and LIME heatmaps generated by the trained OculusNet for CNV, DME, Drusen, and Normal classes.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1596726-g0012.tif"/>
</fig>

<fig id="F13" position="float">
<label>Figure 13</label>
<caption><p>SHAP heatmaps generated by the trained OculusNet for CNV, DME, Drusen, and normal classes.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1596726-g0013.tif"/>
</fig>

</sec></sec>

<sec>
<title>5.3 Ablation study</title>
<p>In this section, an ablation study was conducted to understand the impact of various architectural components on the performance of OculusNet. An ablation study systematically removes parts of the network to evaluate their contribution to the model&#x00027;s final performance. This approach helps identify the most critical components of the network that significantly affect its accuracy and efficiency. Three different models of OculusNet were evaluated on the test dataset, each with varying configurations and complexities. These models were designed to investigate the influence of specific layers and parameters on the network&#x00027;s ability to process and analyze data. The configurations of these models are detailed in <xref ref-type="table" rid="T11">Table 11</xref>, which outlines the layers and parameters involved in each model.</p>
<table-wrap position="float" id="T11">
<label>Table 11</label>
<caption><p>Models architecture utilized for the ablation study.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#727779;color:#ffffff">
<th valign="top" align="left"><bold>Layers</bold></th>
<th valign="top" align="center"><bold>Model 1</bold></th>
<th valign="top" align="center"><bold>Parameters</bold></th>
<th valign="top" align="center"><bold>Model 2</bold></th>
<th valign="top" align="center"><bold>Parameters</bold></th>
<th valign="top" align="center"><bold>Model 3</bold></th>
<th valign="top" align="center"><bold>Parameters</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">separable_conv2d</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">155</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">155</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">155</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_1</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">1,344</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">1,344</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">1,344</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_2</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">1,344</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">1,344</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">1,344</td>
</tr> <tr>
<td valign="top" align="left">Max_Pooling2d</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">0</td>
</tr> <tr>
<td valign="top" align="left">Batch_Normalization</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">128</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_3</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">2,400</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">2,400</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_4</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">4,763</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">4,763</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
</tr> <tr>
<td valign="top" align="left">Max_Pooling2d_1</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
</tr> <tr>
<td valign="top" align="left">Batch_Normalization_1</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">256</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">256</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_5</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">8,896</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_6</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">17,664</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
</tr> <tr>
<td valign="top" align="left">Max_Pooling2d_2</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
</tr> <tr>
<td valign="top" align="left">Batch_Normalization_2</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">512</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_7</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
</tr> <tr>
<td valign="top" align="left">separable_conv2d_8</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
</tr> <tr>
<td valign="top" align="left">Max_Pooling2d_3</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
</tr> <tr>
<td valign="top" align="left">Batch_Normalization_3</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
</tr> <tr>
<td valign="top" align="left">Flatten</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">0</td>
</tr> <tr>
<td valign="top" align="left">Dense</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">9,437,312</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">22,151,296</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">48,664,704</td>
</tr> <tr>
<td valign="top" align="left">Dense_1</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">8,256</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">8,256</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">8,256</td>
</tr> <tr>
<td valign="top" align="left">Dense_2</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">260</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">260</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">260</td>
</tr></tbody>
</table>
</table-wrap>


<p>Following the detailed layer and parameter configurations, <xref ref-type="table" rid="T12">Table 12</xref> provides a summary of the total parameters, accuracy, and model size in MB for each of the three models. The ablation study emphasizes the significant impact of layer configuration and parameter count on the computational efficiency and model size of OculusNet. By comparing Model 1, Model 2, and Model 3, it becomes clear that increasing the complexity and the number of parameters greatly enlarges the model size, with Model 3 having the largest size of 185.68 MB. Conversely, Model 1, which has the fewest parameters, demonstrates a balance between model size and complexity, suggesting a more efficient architecture for applications where computational resources are limited.</p>


<table-wrap position="float" id="T12">
<label>Table 12</label>
<caption><p>Obtained confusion metrics for the three models.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#727779;color:#ffffff">
<th valign="top" align="left"><bold>Model No</bold>.</th>
<th valign="top" align="center"><bold>Accuracy (%)</bold></th>
<th valign="top" align="center"><bold>Total parameters</bold></th>
<th valign="top" align="center"><bold>Model size (MB)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Model 1</td>
<td valign="top" align="center">83.55</td>
<td valign="top" align="center">9,483,263</td>
<td valign="top" align="center">36.18</td>
</tr> <tr>
<td valign="top" align="left">Model 2</td>
<td valign="top" align="center">80.89</td>
<td valign="top" align="center">22,170,175</td>
<td valign="top" align="center">84.57</td>
</tr> <tr>
<td valign="top" align="left">Model 3</td>
<td valign="top" align="center">75.24</td>
<td valign="top" align="center">48,676,191</td>
<td valign="top" align="center">185.68</td>
</tr></tbody>
</table>
</table-wrap>


<p>As shown in <xref ref-type="table" rid="T12">Table 12</xref>, Model 1 demonstrates the highest accuracy at 83.55% among the three models. With an accuracy of 80.89%, Model 2 exhibits a slight decrease in performance compared to Model 1. This model has improved identification for the CNV category but shows a reduction in accuracy for DME and Drusen conditions. Model 3, with an accuracy of 75.24%, reflects a decline in classification performance. The confusion matrix reveals a significant challenge in distinguishing between all conditions. Compared to Model 1, Model 2, and Model 3, the proposed architecture of the OculusNet model, which contains all layers, achieved the best testing accuracy of 95.48%. The respective confusion metrics for all three models are shown in <xref ref-type="fig" rid="F14">Figure 14</xref>.</p>
<fig id="F14" position="float">
<label>Figure 14</label>
<caption><p>Confusion matrices of the tested models in the ablation study.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1596726-g0014.tif"/>
</fig>


<p>Additioanlly, <xref ref-type="table" rid="T13">Table 13</xref> presents a comparison with recent deep-learning approaches for retinal &#x0201C;disease classification.&#x0201D; In Shin et al. (<xref ref-type="bibr" rid="B26">26</xref>), a semi-automated pipeline was applied to a pig-eye dataset, achieving an accuracy of 83.89%. In contrast, (<xref ref-type="bibr" rid="B29">29</xref>) fine-tuned several pre-trained networks on OCT B-scan images, with Xception achieving the best test accuracy of 92.00%. Moreover, Bhandari et al. (<xref ref-type="bibr" rid="B35">35</xref>) proposed a lightweight CNN with only 983,716 trainable parameters, achieving a test accuracy of 94.29% for classifying CNV, DME, and Drusen. By comparison, the proposed OculusNet achieves a superior accuracy of 95.48%.</p>
<table-wrap position="float" id="T13">
<label>Table 13</label>
<caption><p>Comparison with other studies.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#727779;color:#ffffff">
<th valign="top" align="left"><bold>Reference</bold></th>
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="left"><bold>Approach</bold></th>
<th valign="top" align="left"><bold>Result</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Shin et al. (<xref ref-type="bibr" rid="B26">26</xref>)</td>
<td valign="top" align="left">Pig eye dataset</td>
<td valign="top" align="left">Semi-automated</td>
<td valign="top" align="left">83.89%</td>
</tr> <tr>
<td valign="top" align="left">Bhandari et al. (<xref ref-type="bibr" rid="B35">35</xref>)</td>
<td valign="top" align="left">OCT dataset</td>
<td valign="top" align="left">Lightweight CNN model</td>
<td valign="top" align="left">94.29%</td>
</tr> <tr>
<td valign="top" align="left">Kang et al. (<xref ref-type="bibr" rid="B29">29</xref>)</td>
<td valign="top" align="left">OCT B- scan images</td>
<td valign="top" align="left">Pre-trained models</td>
<td valign="top" align="left">92.00%</td>
</tr> <tr>
<td valign="top" align="left">This study</td>
<td valign="top" align="left">OCT</td>
<td valign="top" align="left">OculusNet</td>
<td valign="top" align="left">95.48%</td>
</tr></tbody>
</table>
</table-wrap>
</sec>

<sec>
<title>5.4 Web deployment</title>
<p>The web deployment of OculusNet was hosted by Streamlit, a platform known for its ease of use and efficiency in deploying data applications. The Streamlit library was employed to build an interactive web application that enables users to upload OCT images and receive classification results based on the pre-trained OculusNet model. As shown in <xref ref-type="fig" rid="F15">Figure 15</xref>, the steps that will be followed to classify OCT images are outlined. Pre-trained weights from the OculusNet model are loaded into the system, ensuring that the web application utilizes the refined and optimized weights derived from extensive training sessions. A background image is set for the web application to enhance the user experience. The title and header are defined to prompt users to upload an OCT image for classification. A file uploader widget is provided for users to upload OCT images in &#x0201C;jpeg,&#x0201D; &#x0201C;jpg,&#x0201D; or &#x0201C;png&#x0201D; formats. Upon uploading, the image is displayed on the web interface. The uploaded image is passed to the classify function, which preprocesses the image and utilizes the model to predict the class of retinal disease. The classification result, along with the confidence score, is presented to the user. The confidence score is formatted to display a percentage, aiding in the interpretability of the result. Upon accessing the Streamlit web application, users encounter a clear and straightforward interface. The process is designed to be intuitive, allowing the user to easily upload an OCT image and wait for the model to classify the retinal condition. The application promptly displays the classification along with a confidence score, providing a valuable tool for preliminary diagnosis or a second opinion in clinical settings.</p>
<fig id="F15" position="float">
<label>Figure 15</label>
<caption><p>Steps for using Web page for OCT image classification. <bold>(a)</bold> Step 1: upload image. <bold>(b)</bold> Step 2: display uploaded image. <bold>(c)</bold> Step 3: prediction.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1596726-g0015.tif"/>
</fig>

</sec></sec>
<sec sec-type="conclusions" id="s6">
<title>6 Conclusion</title>
<p>This study presented an interpretable and web-deployable approach to retinal disease classification, promising to enhance diagnostic capabilities in clinical settings. The use of saliency map visualization as an explainable AI technique improved the interpretation of the decision-making process of the proposed model. This transparency is crucial for clinical adoption, as it fosters trust and understanding among healthcare professionals. Furthermore, an ablation study was conducted on OculusNet to validate the effectiveness and robustness of the chosen architecture. To ensure a fair comparison, transfer learning was applied to four pre-trained models. The results demonstrated the superior performance of the proposed model compared to the pre-trained model, with a test accuracy of 95.48% and a validation accuracy of 98.59%. The model&#x00027;s performance was also evaluated using the Kappa statistic and MCC, both of which confirmed the high reliability and consistency of our model&#x00027;s predictions. For practical deployment, the Streamlit server was utilized to create a user-friendly interface that allows users to upload retinal OCT images and receive instant classification results. This web application has significant potential for integration into ophthalmic departments, providing an accessible and efficient tool for diagnosing retinal diseases.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The datasets analyzed and utilized for this study can be found at DOI: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.17632/rscbjbr9sj.3">10.17632/rscbjbr9sj.3</ext-link>. Further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec sec-type="ethics-statement" id="s8">
<title>Ethics statement</title>
<p>The studies involving humans were approved by the University of California San Diego (Shiley Eye Institute of the University of California San Diego, California Retinal Research Foundation, Medical Center Ophthalmology Associates, Shanghai First People&#x00027;s Hospital, and Beijing Tongren Eye Center) DOI: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.17632/rscbjbr9sj.3">10.17632/rscbjbr9sj.3</ext-link>. The work was conducted in a manner compliant with the United States Health Insurance Portability and Accountability Act (HIPAA) and was adherent to the tenets of the Declaration of Helsinki. Written informed consent for participation was not required from the participants or the participants&#x00027; legal guardians/next of kin in accordance with the national legislation and institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="s9">
<title>Author contributions</title>
<p>MU: Methodology, Software, Writing &#x02013; original draft. JA: Investigation, Validation, Writing &#x02013; review &#x00026; editing. OS: Formal analysis, Funding acquisition, Writing &#x02013; review &#x00026; editing. MA: Funding acquisition, Project administration, Writing &#x02013; review &#x00026; editing. AA: Investigation, Validation, Writing &#x02013; review &#x00026; editing. MH: Resources, Validation, Writing &#x02013; review &#x00026; editing. RU: Data curation, Formal analysis, Writing &#x02013; review &#x00026; editing. MSK: Conceptualization, Methodology, Supervision, Writing &#x02013; original draft.</p>
</sec>
<sec sec-type="funding-information" id="s10">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work is funded by Princess Nourah bint Abdulrahman University Researchers Supporting Project number (PNURSP2025R760), Princess Nourah bint Abdulrahman University, Riyadh, Saudi Arabia. The research team thanks the Deanship of Graduate Studies and Scientific Research at Najran University for supporting the research project through the Nama&#x00027;a program, with the project code NU/GP/SERC/13/352-3.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s11">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p></sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>WJ</given-names></name></person-group>. <source>Vitamin C in Human Health and Disease: Effects, Mechanisms of Action, and New Guidance on Intake</source>. <publisher-loc>Dordrecht</publisher-loc>: <publisher-name>Springer Netherlands</publisher-name> (<year>2019</year>). <pub-id pub-id-type="doi">10.1007/978-94-024-1713-5</pub-id></citation>
</ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Subramanian</surname> <given-names>B</given-names></name> <name><surname>Devishamani</surname> <given-names>C</given-names></name> <name><surname>Raman</surname> <given-names>R</given-names></name> <name><surname>Ratra</surname> <given-names>D</given-names></name></person-group>. <article-title>Association of OCT biomarkers and visual impairment in patients with diabetic macular oedema with vitreomacular adhesion</article-title>. <source>PLoS ONE</source>. (<year>2023</year>) <volume>18</volume>:<fpage>1</fpage>&#x02013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0288879</pub-id><pub-id pub-id-type="pmid">37463157</pub-id></citation></ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Le Boite</surname> <given-names>H</given-names></name> <name><surname>Couturier</surname> <given-names>A</given-names></name> <name><surname>Tadayoni</surname> <given-names>R</given-names></name> <name><surname>Lamard</surname> <given-names>M</given-names></name> <name><surname>Quellec</surname> <given-names>G</given-names></name></person-group>. <article-title>VMseg: Using spatial variance to automatically segment retinal non-perfusion on OCT-angiography</article-title>. <source>PLoS ONE</source>. (<year>2024</year>) <volume>19</volume>:<fpage>1</fpage>&#x02013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0306794</pub-id><pub-id pub-id-type="pmid">39110715</pub-id></citation></ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kermany</surname> <given-names>DS</given-names></name> <name><surname>Goldbaum</surname> <given-names>M</given-names></name> <name><surname>Cai</surname> <given-names>W</given-names></name> <name><surname>Valentim</surname> <given-names>CCS</given-names></name> <name><surname>Liang</surname> <given-names>H</given-names></name> <name><surname>Baxter</surname> <given-names>SL</given-names></name> <etal/></person-group>. <article-title>Identifying medical diagnoses and treatable diseases by image-based deep learning</article-title>. <source>Cell</source>. (<year>2018</year>) <volume>172</volume>:<fpage>1122</fpage>&#x02013;<lpage>1131</lpage>.e9. <pub-id pub-id-type="doi">10.1016/j.cell.2018.02.010</pub-id><pub-id pub-id-type="pmid">29474911</pub-id></citation></ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Farsiu</surname> <given-names>S</given-names></name> <name><surname>Chiu</surname> <given-names>SJ</given-names></name> <name><surname>O&#x00027;Connell</surname> <given-names>RV</given-names></name> <name><surname>Folgar</surname> <given-names>FA</given-names></name> <name><surname>Yuan</surname> <given-names>E</given-names></name> <name><surname>Izatt</surname> <given-names>JA</given-names></name> <etal/></person-group>. <article-title>Quantitative classification of eyes with and without intermediate age-related macular degeneration using optical coherence tomography</article-title>. <source>Ophthalmology</source>. (<year>2014</year>) <volume>121</volume>:<fpage>162</fpage>&#x02013;<lpage>72</lpage>. <pub-id pub-id-type="doi">10.1016/j.ophtha.2013.07.013</pub-id><pub-id pub-id-type="pmid">23993787</pub-id></citation></ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>X</given-names></name> <name><surname>Li</surname> <given-names>X</given-names></name> <name><surname>Tang</surname> <given-names>Q</given-names></name> <name><surname>Zhang</surname> <given-names>Y</given-names></name> <name><surname>Zhang</surname> <given-names>L</given-names></name> <name><surname>Zhang</surname> <given-names>M</given-names></name></person-group>. <article-title>Exploring laser-induced acute and chronic retinal vein occlusion mouse models: Development, temporal in vivo imaging, and application perspectives</article-title>. <source>PLoS ONE</source>. (<year>2024</year>) <volume>19</volume>:<fpage>1</fpage>&#x02013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0305741</pub-id><pub-id pub-id-type="pmid">38885229</pub-id></citation></ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alqudah</surname> <given-names>A</given-names></name> <name><surname>Alqudah</surname> <given-names>AM</given-names></name> <name><surname>AlTantawi</surname> <given-names>M</given-names></name></person-group>. <article-title>Artificial Intelligence Hybrid System for Enhancing Retinal Diseases Classification Using Automated Deep Features Extracted from OCT Images</article-title>. <source>Int J Intell Syst Applic Eng</source>. (<year>2021</year>) <volume>9</volume>:<fpage>91</fpage>&#x02013;<lpage>100</lpage>. <pub-id pub-id-type="doi">10.18201/ijisae.2021.236</pub-id></citation>
</ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kanagasingam</surname> <given-names>Y</given-names></name> <name><surname>Bhuiyan</surname> <given-names>A</given-names></name> <name><surname>Abr&#x000E0;moff</surname> <given-names>MD</given-names></name> <name><surname>Smith</surname> <given-names>RT</given-names></name> <name><surname>Goldschmidt</surname> <given-names>L</given-names></name> <name><surname>Wong</surname> <given-names>TY</given-names></name></person-group>. <article-title>Progress on retinal image analysis for age related macular degeneration</article-title>. <source>Prog Retin Eye Res</source>. (<year>2014</year>) <volume>38</volume>:<fpage>20</fpage>&#x02013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1016/j.preteyeres.2013.10.002</pub-id><pub-id pub-id-type="pmid">24211245</pub-id></citation></ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jayaraman</surname> <given-names>V</given-names></name> <name><surname>Burgner</surname> <given-names>CB</given-names></name> <name><surname>Carter</surname> <given-names>J</given-names></name> <name><surname>Borova</surname> <given-names>I</given-names></name> <name><surname>Bramham</surname> <given-names>N</given-names></name> <name><surname>Lindblad</surname> <given-names>C</given-names></name> <etal/></person-group>. <article-title>Widely tunable electrically pumped 1050nm MEMS-VCSELs for optical coherence tomography</article-title>. In:<person-group person-group-type="editor"><name><surname>Graham</surname> <given-names>LA</given-names></name> <name><surname>Lei</surname> <given-names>C</given-names></name></person-group>, editors. <source>Vertical-Cavity Surface-Emitting Lasers XXIV</source>. International Society for Optics and Photonics, SPIE (<year>2020</year>). p. 113000S. <pub-id pub-id-type="doi">10.1117/12.2543819</pub-id></citation>
</ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Amin</surname> <given-names>J</given-names></name> <name><surname>Sharif</surname> <given-names>M</given-names></name> <name><surname>Rehman</surname> <given-names>A</given-names></name> <name><surname>Raza</surname> <given-names>M</given-names></name> <name><surname>Mufti</surname> <given-names>MR</given-names></name></person-group>. <article-title>Diabetic retinopathy detection and classification using hybrid feature set</article-title>. <source>Microsc Res Tech</source>. (<year>2018</year>) <volume>81</volume>:<fpage>990</fpage>&#x02013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1002/jemt.23063</pub-id><pub-id pub-id-type="pmid">30447130</pub-id></citation></ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alqudah</surname> <given-names>AM</given-names></name> <name><surname>AOCT-NET</surname></name></person-group>. <article-title>a convolutional network automated classification of multiclass retinal diseases using spectral-domain optical coherence tomography images</article-title>. <source>Med Biol Eng Comput</source>. (<year>2020</year>) <volume>58</volume>:<fpage>41</fpage>&#x02013;<lpage>53</lpage>. <pub-id pub-id-type="doi">10.1007/s11517-019-02066-y</pub-id><pub-id pub-id-type="pmid">31728935</pub-id></citation></ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Drexler</surname> <given-names>W</given-names></name> <name><surname>Fujimoto</surname> <given-names>JG</given-names></name></person-group>. <article-title>State-of-the-art retinal optical coherence tomography</article-title>. <source>Prog Retin Eye Res</source>. (<year>2008</year>) <volume>27</volume>:<fpage>45</fpage>&#x02013;<lpage>88</lpage>. <pub-id pub-id-type="doi">10.1016/j.preteyeres.2007.07.005</pub-id><pub-id pub-id-type="pmid">18036865</pub-id></citation></ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Puliafito</surname> <given-names>CA</given-names></name> <name><surname>Hee</surname> <given-names>MR</given-names></name> <name><surname>Lin</surname> <given-names>CP</given-names></name> <name><surname>Reichel</surname> <given-names>E</given-names></name> <name><surname>Schuman</surname> <given-names>JS</given-names></name> <name><surname>Duker</surname> <given-names>JS</given-names></name> <etal/></person-group>. <article-title>Imaging of macular diseases with optical coherence tomography</article-title>. <source>Ophthalmology</source>. (<year>1995</year>) <volume>102</volume>:<fpage>217</fpage>&#x02013;<lpage>29</lpage>. <pub-id pub-id-type="doi">10.1016/S0161-6420(95)31032-9</pub-id></citation>
</ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>D</given-names></name> <name><surname>Swanson</surname> <given-names>EA</given-names></name> <name><surname>Lin</surname> <given-names>CP</given-names></name> <name><surname>Schuman</surname> <given-names>JS</given-names></name> <name><surname>Stinson</surname> <given-names>WG</given-names></name> <name><surname>Chang</surname> <given-names>W</given-names></name> <etal/></person-group>. <article-title>Optical coherence tomography</article-title>. <source>Science</source>. (<year>1991</year>) <volume>254</volume>:<fpage>1178</fpage>&#x02013;<lpage>81</lpage>. <pub-id pub-id-type="doi">10.1126/science.1957169</pub-id><pub-id pub-id-type="pmid">1957169</pub-id></citation></ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Leitgeb</surname> <given-names>R</given-names></name> <name><surname>Hitzenberger</surname> <given-names>CK</given-names></name> <name><surname>Fercher</surname> <given-names>AF</given-names></name></person-group>. <article-title>Performance of Fourier domain vs. time domain optical coherence tomography</article-title>. <source>Optics Expr</source>. (<year>2003</year>) <volume>11</volume>:<fpage>889</fpage>&#x02013;<lpage>94</lpage>. <pub-id pub-id-type="doi">10.1364/OE.11.000889</pub-id></citation>
</ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Golubovic</surname> <given-names>B</given-names></name> <name><surname>Bouma</surname> <given-names>BE</given-names></name> <name><surname>Tearney</surname> <given-names>GJ</given-names></name> <name><surname>Fujimoto</surname> <given-names>JG</given-names></name></person-group>. <article-title>Optical frequency-domain reflectometry using rapid wavelength tuning of a Cr4&#x0002B;:forsterite laser</article-title>. <source>Opt Lett</source>. (<year>1997</year>) <volume>22</volume>:<fpage>1704</fpage>&#x02013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1364/OL.22.001704</pub-id></citation>
</ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Monemian</surname> <given-names>M</given-names></name> <name><surname>Irajpour</surname> <given-names>M</given-names></name> <name><surname>Rabbani</surname> <given-names>H</given-names></name> <name><surname>A</surname></name></person-group>. <article-title>review on texture-based methods for anomaly detection in retinal optical coherence tomography images</article-title>. <source>Optik</source>. (<year>2023</year>) <volume>288</volume>:<fpage>171165</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijleo.2023.171165</pub-id></citation>
</ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ejaz</surname> <given-names>S</given-names></name> <name><surname>Baig</surname> <given-names>R</given-names></name> <name><surname>Ashraf</surname> <given-names>Z</given-names></name> <name><surname>Alnfiai</surname> <given-names>MM</given-names></name> <name><surname>Alnahari</surname> <given-names>MM</given-names></name> <name><surname>Alotaibi</surname> <given-names>RM</given-names></name> <etal/></person-group>. <article-title>deep learning framework for the early detection of multi-retinal diseases</article-title>. <source>PLoS ONE</source>. (<year>2024</year>) <volume>19</volume>:<fpage>1</fpage>&#x02013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0307317</pub-id><pub-id pub-id-type="pmid">39052616</pub-id></citation></ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Patnaik</surname> <given-names>S</given-names></name> <name><surname>Subasi</surname> <given-names>A</given-names></name></person-group>. <article-title>Chapter 12 - Artificial intelligence-based retinal disease classification using optical coherence tomography images</article-title>. In: <source>Applications of Artificial Intelligence in Medical Imaging</source>. Academic Press (<year>2023</year>). p. <fpage>305</fpage>&#x02013;<lpage>319</lpage>. <pub-id pub-id-type="doi">10.1016/B978-0-443-18450-5.00009-8</pub-id></citation>
</ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khalaf</surname> <given-names>NB</given-names></name> <name><surname>Aljobouri</surname> <given-names>HK</given-names></name> <name><surname>Najim</surname> <given-names>MS</given-names></name> <name><surname>&#x000C7;ankaya</surname> <given-names>I</given-names></name></person-group>. <article-title>Simplified convolutional neural network model for automatic classification of retinal diseases from optical coherence tomography images</article-title>. <source>Al-Nahrain J Eng Sci</source>. (<year>2024</year>) <volume>26</volume>:<fpage>314</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.29194/NJES.26040314</pub-id></citation>
</ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Duan</surname> <given-names>J</given-names></name> <name><surname>Tench</surname> <given-names>C</given-names></name> <name><surname>Gottlob</surname> <given-names>I</given-names></name> <name><surname>Proudlock</surname> <given-names>F</given-names></name> <name><surname>Bai</surname> <given-names>L</given-names></name></person-group>. <article-title>New variational image decomposition model for simultaneously denoising and segmenting optical coherence tomography images</article-title>. <source>Phys Med Biol</source>. (<year>2015</year>) <volume>60</volume>:<fpage>8901</fpage>. <pub-id pub-id-type="doi">10.1088/0031-9155/60/22/8901</pub-id><pub-id pub-id-type="pmid">26553577</pub-id></citation></ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Asif</surname> <given-names>S</given-names></name> <name><surname>Amjad</surname> <given-names>K</given-names></name> <name><surname>Qurrat ul</surname> <given-names>A</given-names></name></person-group>. <article-title>Deep residual network for diagnosis of retinal diseases using optical coherence tomography images</article-title>. <source>Interdisc Sci</source>. (<year>2022</year>) <volume>14</volume>:<fpage>906</fpage>&#x02013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1007/s12539-022-00533-z</pub-id><pub-id pub-id-type="pmid">35767116</pub-id></citation></ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Fabritius</surname> <given-names>T</given-names></name> <name><surname>Makita</surname> <given-names>S</given-names></name> <name><surname>Myllyl&#x000E4;</surname> <given-names>R</given-names></name> <name><surname>Yasuno</surname> <given-names>Y</given-names></name></person-group>. <article-title>Automated retinal pigment epithelium identification from optical coherence tomography images</article-title>. In: <source>Proceedings Volume 7168, Optical Coherence Tomography and Coherence Domain Optical Methods in Biomedicine XIII; SPIE BiOS</source>. <publisher-loc>San Jose, California</publisher-loc>: <publisher-name>SPIE</publisher-name> (<year>2009</year>). <pub-id pub-id-type="doi">10.1117/12.808543</pub-id></citation>
</ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Muni Nagamani</surname> <given-names>G</given-names></name> <name><surname>Rayachoti</surname> <given-names>E</given-names></name></person-group>. <article-title>Deep learning network (DL-Net) based classification and segmentation of multi-class retinal diseases using OCT scans</article-title>. <source>Biomed Signal Process Control</source>. (<year>2024</year>) <volume>88</volume>:<fpage>105619</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2023.105619</pub-id></citation>
</ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>M</given-names></name> <name><surname>Zhu</surname> <given-names>W</given-names></name> <name><surname>Yu</surname> <given-names>K</given-names></name> <name><surname>Chen</surname> <given-names>Z</given-names></name> <name><surname>Shi</surname> <given-names>F</given-names></name> <name><surname>Zhou</surname> <given-names>Y</given-names></name> <etal/></person-group>. <article-title>Semi-supervised capsule cGAN for speckle noise reduction in retinal OCT images</article-title>. <source>IEEE Trans Med Imag</source>. (<year>2021</year>) <volume>40</volume>:<fpage>1168</fpage>&#x02013;<lpage>83</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2020.3048975</pub-id><pub-id pub-id-type="pmid">33395391</pub-id></citation></ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shin</surname> <given-names>C</given-names></name> <name><surname>Gerber</surname> <given-names>MJ</given-names></name> <name><surname>Lee</surname> <given-names>YH</given-names></name> <name><surname>Rodriguez</surname> <given-names>M</given-names></name> <name><surname>Pedram</surname> <given-names>SA</given-names></name> <name><surname>Hubschman</surname> <given-names>JP</given-names></name> <etal/></person-group>. <article-title>Semi-automated extraction of lens fragments via a surgical robot using semantic segmentation of OCT images with deep learning - experimental results in ex vivo animal model</article-title>. <source>IEEE Robot Autom Lett</source>. (<year>2021</year>) <volume>6</volume>:<fpage>5261</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2021.3072574</pub-id><pub-id pub-id-type="pmid">34621980</pub-id></citation></ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Adel</surname> <given-names>A</given-names></name> <name><surname>Soliman</surname> <given-names>MM</given-names></name> <name><surname>Khalifa</surname> <given-names>NEM</given-names></name> <name><surname>Mostafa</surname> <given-names>K</given-names></name></person-group>. <article-title>Automatic classification of retinal eye diseases from optical coherence tomography using transfer learning</article-title>. In: <source>2020 16th International Computer Engineering Conference (ICENCO)</source>. <publisher-loc>IEEE</publisher-loc> (<year>2020</year>). p. <fpage>37</fpage>&#x02013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1109/ICENCO49778.2020.9357324</pub-id></citation>
</ref>
<ref id="B28">
<label>28.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Islam</surname> <given-names>KT</given-names></name> <name><surname>Wijewickrema</surname> <given-names>S</given-names></name> <name><surname>O&#x00027;Leary</surname> <given-names>S</given-names></name></person-group>. <article-title>Identifying diabetic retinopathy from OCT images using deep transfer learning with artificial neural networks</article-title>. In: <source>2019 IEEE 32nd International Symposium on Computer-Based Medical Systems (CBMS)</source>. <publisher-loc>IEEE</publisher-loc> (<year>2019</year>). p. <fpage>281</fpage>&#x02013;<lpage>286</lpage>. <pub-id pub-id-type="doi">10.1109/CBMS.2019.00066</pub-id></citation>
</ref>
<ref id="B29">
<label>29.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kang</surname> <given-names>NY</given-names></name> <name><surname>Ra</surname> <given-names>H</given-names></name> <name><surname>Lee</surname> <given-names>K</given-names></name> <name><surname>Lee</surname> <given-names>JH</given-names></name> <name><surname>Lee</surname> <given-names>WK</given-names></name> <name><surname>Baek</surname> <given-names>J</given-names></name></person-group>. <article-title>Classification of pachychoroid on optical coherence tomography using deep learning</article-title>. <source>Graefe&#x00027;s Arch Clin Exper Ophthalmol</source>. (<year>2021</year>) <volume>259</volume>:<fpage>1803</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1007/s00417-021-05104-4</pub-id><pub-id pub-id-type="pmid">33616757</pub-id></citation></ref>
<ref id="B30">
<label>30.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>Q</given-names></name> <name><surname>Zhang</surname> <given-names>B</given-names></name> <name><surname>Hu</surname> <given-names>Y</given-names></name> <name><surname>Liu</surname> <given-names>B</given-names></name> <name><surname>Cao</surname> <given-names>D</given-names></name> <name><surname>Yang</surname> <given-names>D</given-names></name> <etal/></person-group>. <article-title>Detection of morphologic patterns of diabetic macular edema using a deep learning approach based on optical coherence tomography images</article-title>. <source>Retina</source>. (<year>2021</year>) <volume>41</volume>:<fpage>1110</fpage>&#x02013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1097/IAE.0000000000002992</pub-id><pub-id pub-id-type="pmid">33031250</pub-id></citation></ref>
<ref id="B31">
<label>31.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hassan</surname> <given-names>SAE</given-names></name> <name><surname>Akbar</surname> <given-names>S</given-names></name> <name><surname>Gull</surname> <given-names>S</given-names></name> <name><surname>Rehman</surname> <given-names>A</given-names></name> <name><surname>Alaska</surname> <given-names>H</given-names></name></person-group>. <article-title>Deep learning-based automatic detection of central serous retinopathy using optical coherence tomographic images</article-title>. in <source>2021 1st International Conference on Artificial Intelligence and Data Analytics (CAIDA).</source> IEEE (<year>2021</year>). <fpage>206</fpage>&#x02013;<lpage>211</lpage>.</citation>
</ref>
<ref id="B32">
<label>32.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jawad</surname> <given-names>MA</given-names></name> <name><surname>Khursheed</surname> <given-names>F</given-names></name> <name><surname>Nawaz</surname> <given-names>S</given-names></name> <name><surname>Mir</surname> <given-names>AH</given-names></name></person-group>. <article-title>Towards improved fundus disease detection using Swin Transformers</article-title>. <source>Multimed Tools Appl</source>. (<year>2024</year>) <volume>83</volume>:<fpage>78125</fpage>&#x02013;<lpage>59</lpage>. <pub-id pub-id-type="doi">10.1007/s11042-024-18627-9</pub-id></citation>
</ref>
<ref id="B33">
<label>33.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abdul Jawad</surname> <given-names>M</given-names></name> <name><surname>Khursheed</surname> <given-names>F</given-names></name></person-group>. <article-title>Deep and dense convolutional neural network for multi category classification of magnification specific and magnification independent breast cancer histopathological images</article-title>. <source>Biomed Signal Process Control</source>. (<year>2022</year>) <volume>78</volume>:<fpage>103935</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2022.103935</pub-id></citation>
</ref>
<ref id="B34">
<label>34.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abdul Jawad</surname> <given-names>M</given-names></name> <name><surname>Khursheed</surname> <given-names>F</given-names></name></person-group>. <article-title>A novel approach for color-balanced reference image selection for breast histology image normalization</article-title>. <source>Biomed Signal Process Control</source>. (<year>2024</year>) <volume>94</volume>:<fpage>106299</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2024.106299</pub-id></citation>
</ref>
<ref id="B35">
<label>35.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bhandari</surname> <given-names>M</given-names></name> <name><surname>Shahi</surname> <given-names>TB</given-names></name> <name><surname>Neupane</surname> <given-names>A</given-names></name></person-group>. <article-title>Evaluating retinal disease diagnosis with an interpretable lightweight CNN model resistant to adversarial attacks</article-title>. <source>J Imaging</source>. (<year>2023</year>) <volume>9</volume>:<fpage>219</fpage>. <pub-id pub-id-type="doi">10.3390/jimaging9100219</pub-id><pub-id pub-id-type="pmid">37888326</pub-id></citation></ref>
<ref id="B36">
<label>36.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kermany</surname> <given-names>D</given-names></name></person-group>. <source>Labeled optical coherence tomography (OCT) and chest X-ray images for classification</source>. Mendeley data (<year>2018</year>).</citation>
</ref>
<ref id="B37">
<label>37.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Le</surname> <given-names>DN</given-names></name> <name><surname>Parvathy</surname> <given-names>VS</given-names></name> <name><surname>Gupta</surname> <given-names>D</given-names></name> <name><surname>Khanna</surname> <given-names>A</given-names></name> <name><surname>Rodrigues</surname> <given-names>JJPC</given-names></name> <name><surname>Shankar</surname> <given-names>K</given-names></name></person-group>. <article-title>IoT enabled depthwise separable convolution neural network with deep support vector machine for COVID-19 diagnosis and classification</article-title>. <source>Int J Mach Learn Cybern</source>. (<year>2021</year>) <volume>12</volume>:<fpage>3235</fpage>&#x02013;<lpage>48</lpage>. <pub-id pub-id-type="doi">10.1007/s13042-020-01248-7</pub-id><pub-id pub-id-type="pmid">33727984</pub-id></citation></ref>
<ref id="B38">
<label>38.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Q</given-names></name> <name><surname>Ning</surname> <given-names>J</given-names></name> <name><surname>Yuan</surname> <given-names>J</given-names></name> <name><surname>Xiao</surname> <given-names>L</given-names></name></person-group>. <article-title>A depthwise separable dense convolutional network with convolution block attention module for COVID-19 diagnosis on CT scans</article-title>. <source>Comput Biol Med</source>. (<year>2021</year>) <volume>137</volume>:<fpage>104837</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2021.104837</pub-id><pub-id pub-id-type="pmid">34530335</pub-id></citation></ref>
<ref id="B39">
<label>39.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Umair</surname> <given-names>M</given-names></name> <name><surname>Khan</surname> <given-names>MS</given-names></name> <name><surname>Ahmed</surname> <given-names>F</given-names></name> <name><surname>Baothman</surname> <given-names>F</given-names></name> <name><surname>Alqahtani</surname> <given-names>F</given-names></name> <name><surname>Alian</surname> <given-names>M</given-names></name> <etal/></person-group>. <article-title>Detection of COVID-19 using transfer learning and grad-CAM visualization on indigenously collected x-ray dataset</article-title>. <source>Sensors</source>. (<year>2021</year>) <volume>21</volume>:<fpage>5813</fpage>. <pub-id pub-id-type="doi">10.3390/s21175813</pub-id><pub-id pub-id-type="pmid">34502702</pub-id></citation></ref>
<ref id="B40">
<label>40.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>J</given-names></name> <name><surname>Wang</surname> <given-names>G</given-names></name> <name><surname>Xiao</surname> <given-names>X</given-names></name> <name><surname>Bao</surname> <given-names>M</given-names></name> <name><surname>Tian</surname> <given-names>G</given-names></name></person-group>. <article-title>Explainable ensemble learning method for OCT detection with transfer learning</article-title>. <source>PLoS ONE</source>. (<year>2024</year>) <volume>19</volume>:<fpage>1</fpage>&#x02013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0296175</pub-id><pub-id pub-id-type="pmid">38517913</pub-id></citation></ref>
<ref id="B41">
<label>41.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bhandari</surname> <given-names>M</given-names></name> <name><surname>Shahi</surname> <given-names>TB</given-names></name> <name><surname>Siku</surname> <given-names>B</given-names></name> <name><surname>Neupane</surname> <given-names>A</given-names></name></person-group>. <article-title>Explanatory classification of CXR images into COVID-19, Pneumonia and Tuberculosis using deep learning and XAI</article-title>. <source>Comput Biol Med</source>. (<year>2022</year>) <volume>150</volume>:<fpage>106156</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.106156</pub-id><pub-id pub-id-type="pmid">36228463</pub-id></citation></ref>
</ref-list>
</back>
</article>