<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Cell. Infect. Microbiol.</journal-id>
<journal-title>Frontiers in Cellular and Infection Microbiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Cell. Infect. Microbiol.</abbrev-journal-title>
<issn pub-type="epub">2235-2988</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fcimb.2024.1397316</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Cellular and Infection Microbiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>MpoxNet: dual-branch deep residual squeeze and excitation monkeypox classification network with attention mechanism</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Sun</surname>
<given-names>Jingbo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2672681"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Yuan</surname>
<given-names>Baoxi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2169355"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sun</surname>
<given-names>Zhaocheng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2740976"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhu</surname>
<given-names>Jiajun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2741049"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Deng</surname>
<given-names>Yuxin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2740969"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gong</surname>
<given-names>Yi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2740993"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Yuhe</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2740978"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Electronic Information, Xijing University</institution>, <addr-line>Xi&#x2019;an</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Shaanxi Key Laboratory of Integrated and Intelligent Navigation, The 20th Research Institute of China Electronics Technology Group Corporation</institution>, <addr-line>Xi&#x2019;an</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Xi&#x2019;an Key Laboratory of High Precision Industrial Intelligent Vision Measurement Technology, Xijing University</institution>, <addr-line>Xi&#x2019;an</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Muhammad Munir, Lancaster University, United Kingdom</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Krishna Kumar Mohbey, Central University of Rajasthan, India</p>
<p>Chiranjibi Sitaula, The University of Melbourne, Australia</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Baoxi Yuan, <email xlink:href="mailto:ybxbupt@163.com">ybxbupt@163.com</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>07</day>
<month>06</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>14</volume>
<elocation-id>1397316</elocation-id>
<history>
<date date-type="received">
<day>07</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>08</day>
<month>05</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Sun, Yuan, Sun, Zhu, Deng, Gong and Chen</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Sun, Yuan, Sun, Zhu, Deng, Gong and Chen</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>While the world struggles to recover from the devastation wrought by the widespread spread of COVID-19, monkeypox virus has emerged as a new global pandemic threat. In this paper, a high precision and lightweight classification network MpoxNet based on ConvNext is proposed to meet the need of fast and safe detection of monkeypox classification. In this method, a two-branch depth-separable convolution residual Squeeze and Excitation module is designed. This design aims to extract more feature information with two branches, and greatly reduces the number of parameters in the model by using depth-separable convolution. In addition, our method introduces a convolutional attention module to enhance the extraction of key features within the receptive field. The experimental results show that MpoxNet has achieved remarkable results in monkeypox disease classification, the accuracy rate is 95.28%, the precision rate is 96.40%, the recall rate is 93.00%, and the F1-Score is 95.80%. This is significantly better than the current mainstream classification model. It is worth noting that the FLOPS and the number of parameters of MpoxNet are only 30.68% and 31.87% of those of ConvNext-Tiny, indicating that the model has a small computational burden and model complexity while efficient performance.</p>
</abstract>
<kwd-group>
<kwd>monkeypox</kwd>
<kwd>deep learning</kwd>
<kwd>image processing</kwd>
<kwd>artificial intelligence</kwd>
<kwd>feature selection</kwd>
</kwd-group>
<counts>
<fig-count count="10"/>
<table-count count="4"/>
<equation-count count="12"/>
<ref-count count="55"/>
<page-count count="16"/>
<word-count count="7837"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Virus and Host</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>With the 2020 coronavirus pandemic having a profound impact across the globe, reports of the emergence of monkeypox in 2023 reveal the threat of another global virus (<xref ref-type="bibr" rid="B34">McCollum and Damon, 2014</xref>). Monkeypox (Mpox) is a disease caused by Mpox virus and is a viral zoonotic disease of orthopoxvirus, so it can be transmitted from animals to humans through direct close contact (<xref ref-type="bibr" rid="B23">Islam et&#xa0;al., 2022</xref>), as well as human-to-human transmission. Mpox was first discovered in 1958 in a monkey in a laboratory in Copenhagen, Denmark (<xref ref-type="bibr" rid="B30">Ladnyj et&#xa0;al., 1972</xref>) and is known as Mpox due to its similar outbreak symptoms to smallpox.</p>
<p>The Mpox virus caused the first infection in the Congo in 1970. Since then, most cases have occurred in Congo, Central and West Africa, and the number of cases has gradually increased, affecting many people living near tropical regions. As of 2022, the World Health Organization(WHO) reports that several other non-African countries such as Europe and the United States have also reported cases of Mpox virus infection (<xref ref-type="bibr" rid="B5">Alakunle et&#xa0;al., 2020</xref>).</p>
<p>Since the declaration of the eradication of smallpox in 1980 and the subsequent cessation of smallpox vaccination, monkeypox has emerged as the predominant orthopoxvirus. Its symptoms resemble those of smallpox, thus garnering attention in the field of public health (<xref ref-type="bibr" rid="B38">Mohbey et&#xa0;al., 2022</xref>). In 2003, the United States became the first country outside Africa to experience a monkeypox outbreak. According to reports, in September 2018, a Nigerian tourist was infected with monkeypox in Israel; in September 2018, December 2019, May 2021, and May 2022, cases of infection were also reported in Singapore; while in May 2019, July 2021, and November 2021, cases of monkeypox were recorded in the United States. These countries may be located in Southeast Asia. In May 2022, a significant number of monkeypox cases were reported in some countries where the disease does not typically occur (<xref ref-type="bibr" rid="B38">Mohbey et&#xa0;al., 2022</xref>). According to the U.S. Centers for Disease Control and Prevention (CDC), as of December 21, 2022, Mpox cases have been reported in 94 countries worldwide, totaling approximately 83,424 cases. Due to the dire impact of the COVID-19 pandemic, Mpox cases have begun to be closely monitored, showing signs of potential epidemic transmission even though large-scale transmission has not yet occurred (<xref ref-type="bibr" rid="B42">Rogers et&#xa0;al., 2008</xref>; <xref ref-type="bibr" rid="B5">Alakunle et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B16">G&#xfc;rb&#xfc;z and Aydin, 2022</xref>; <xref ref-type="bibr" rid="B46">Sharif et&#xa0;al., 2022</xref>). As a result, deep anxiety and worry among people are gradually spreading.</p>
<p>Monkeypox virus infection is generally divided into two stages: the invasive stage and the skin rash stage. Symptoms during the invasive stage include fever, severe headache, swollen lymph nodes, back pain, muscle aches, and weakness. During the skin rash stage, a rash appears 1&#x2013;3 days after the onset of fever, concentrating on the face and limbs (<xref ref-type="bibr" rid="B2">Adalja and Inglesby, 2022</xref>). The rash progresses from macules to papules, vesicles, pustules, forms scabs, and eventually falls off. These skin lesions are typically quite painful. When the rash appears, the patient becomes contagious. Monkeypox virus can spread through contact with infected individuals or animals. Specifically, when individuals come into contact with the ulcers, scabs, respiratory droplets, or oral fluids of an infected person, it may lead to the transmission of the disease to others (<xref ref-type="bibr" rid="B48">Simpson et&#xa0;al., 2020</xref>). Therefore, timely diagnosis of this disease is essential. According to the guidelines proposed by the WHO, healthcare personnel should wear protective gear when caring for patients. Additionally, patients need to be isolated, and they should maintain distance from others. Mpox exhibits subtle differences from other viruses such as smallpox, chickenpox, and measles, primarily in the inflammatory and rash symptoms induced within the human body. Apart from polymerase chain reaction (PCR), which is an effective diagnostic method (<xref ref-type="bibr" rid="B21">Ibrahim et&#xa0;al., 2021</xref>), non-specialists may find it challenging to differentiate them visually. Moreover, PCR testing is costly and typically requires a considerable amount of time to yield results. Therefore, it has not been widely adopted, posing challenges to rapid diagnosis (<xref ref-type="bibr" rid="B40">Reed et&#xa0;al., 2004</xref>). In order to involve non-specialists in the prevention and control of Mpox virus, researchers are actively seeking effective methods utilizing artificial intelligence to identify cases of Mpox. They are engaged in data collection and research experiments to gain a deeper understanding of this disease (<xref ref-type="bibr" rid="B11">Banerjee et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B27">Khafaga et&#xa0;al., 2022a</xref>). The development of automated identification algorithms could not only aid in the rapid identification of Mpox virus cases but also be utilized for training healthcare professionals who are not specialized in Mpox.</p>
<p>Despite the relatively low fatality rate of Mpox within the range of 1&#x2013;10% (<xref ref-type="bibr" rid="B15">Gong et&#xa0;al., 2022</xref>), the absence of an antiviral therapy for curing monkeypox (<xref ref-type="bibr" rid="B41">Reynolds et&#xa0;al., 2017</xref>) underscores the importance of early detection in preventing its spread. Early identification plays a crucial role in the prevention, diagnosis, and treatment of this disease. With the rapid advancement of artificial intelligence models in the field of medicine, deep learning models for medical image analysis have been proposed for various medical science applications. However, there still exist numerous traditional manual classification methods for identifying Mpox, which suffer from apparent inefficiencies and high costs. The reliance on manual diagnostic classification is susceptible to human factors, leading to diagnostic variability and delayed confirmation of Mpox, depriving patients of timely access to appropriate treatment plans. Therefore, there is an urgent need to introduce advanced technologies to replace simplistic manual classification methods, thereby enhancing the chances of patient recovery and reducing the risk of transmission.</p>
<p>Over the years, deep learning (DL) has achieved remarkable success, exerting a profound impact on the core concepts of machine learning (ML) and artificial intelligence (AI). DL methods have demonstrated outstanding results in various industrial domains, overcoming limitations of traditional approaches. They have become powerful tools in the fields of image analysis and pattern recognition, showing extensive potential applications in disease detection. In particular, Convolutional Neural Networks (CNNs) have emerged as a cornerstone in image recognition, owing to their exceptional capabilities in feature extraction and non-linear representation. CNN, a DL neural network architecture, is commonly employed with image data as input. Through a series of operations, it extracts crucial features and information from input images, facilitating tasks such as classification or other related objectives. Some scholars have made significant progress by applying deep learning techniques to disease classification tasks. Sandeep et&#xa0;al. proposed a low-complexity CNN for identifying skin diseases such as psoriasis, melanoma, lupus, and chickenpox (<xref ref-type="bibr" rid="B45">Sandeep et&#xa0;al., 2022</xref>). Their research found that using existing VGGNet and image analysis techniques could accurately detect 71% of skin diseases. However, their proposed solution demonstrated optimal performance with an accuracy of approximately 78%. Glock et&#xa0;al (<xref ref-type="bibr" rid="B14">Glock et&#xa0;al., 2021</xref>). utilized the ResNet-50 model to develop a transfer learning approach for measles detection, exhibiting good performance on multiple rash image datasets with a sensitivity of 81.7%, specificity of 97.1%, and accuracy of 95.2%. Velasco et&#xa0;al (<xref ref-type="bibr" rid="B52">Velasco et&#xa0;al., 2019</xref>) introduced an intelligent smartphone skin disease recognition method based on MobileNet, reporting an accurate detection of chickenpox symptoms with an accuracy of about 94.4%. Sahin et&#xa0;al. (<xref ref-type="bibr" rid="B43">Sahin et&#xa0;al., 2022</xref>) developed a mobile application on the Android platform that rapidly diagnoses Mpox patients using deep learning technology, achieving an image classification accuracy of 91.11%.</p>    <p>Some scholars have already applied deep learning to the task of Mpox classification, achieving significant results (<xref ref-type="bibr" rid="B37">Mehrotra et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B3">Ahsan et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B7">Almufareh et&#xa0;al., 2023</xref>). A portion of researchers has also conducted relevant studies in the field of Mpox disease identification, as elucidated in the &#x201c;Literature Review.&#x201d; The main objective of this study is to determine and validate the optimal performing model for Mpox classification through transfer learning methods and classification models, with the aim of determining the incidence rate of Mpox. This study introduces a Mpox classification algorithm named MpoxNet, based on ConvNext (<xref ref-type="bibr" rid="B32">Liu et&#xa0;al., 2022</xref>), designed for the task of Mpox disease classification. Improvements to the ConvNext model are as follows: firstly, the introduction of a new dual-branch deep residual Squeeze and Excitation (D<sup>2</sup>RSE) module replaces the original ConvNext. This module has a dual-channel structure, significantly enhancing the model&#x2019;s classification accuracy while achieving better performance with a reduced number of model parameters. By integrating the Convolutional Block Attention Module (CBAM) with ConvNext, we capture the spatial relationships of Mpox features during training, adaptively weighting features based on their importance in different spatial and channel dimensions, and suppressing irrelevant regions, allowing the model to selectively focus on features of different disease categories. For example, Sitaula et&#xa0;al. (<xref ref-type="bibr" rid="B49">Sitaula and Hossain, 2021</xref>)proposed a deep learning model based on attention, which employed the attention mechanism of VGG-16, aiming to more accurately capture the spatial relationships of key regions in chest X-ray images. In this experiment, MpoxNet is compared with other mainstream algorithm models, demonstrating higher classification accuracy in Mpox classification scenarios. The main contributions of this paper are as follows:</p>
<list list-type="order">
<list-item>
<p>The introduction of a dual-branch D<sup>2</sup>RSE module, along&#xa0;with the integration of CBAM to enhance the ConvNext model, facilitates the effective extraction of critical local and global information regions within Mpox images. As a result, the proposed MpoxNet model demonstrates a notable improvement in the accuracy of monkeypox detection.</p>
</list-item>
<list-item>
<p>Conducted comprehensive ablation experiments to independently verify the impactful roles of the D<sup>2</sup>RSE module and CBAM attention mechanism in the model&#x2019;s performance. This contributes to a more profound understanding of Mpox disease, offering valuable insights for future research.</p>
</list-item>
<list-item>
<p>Our proposed method requires fewer parameters compared to other mainstream networks and is trained end-to-end, which adequately validates the superiority of the network in terms of performance.</p>
</list-item>
</list>
<p>This paper is organized into the following sections: The second section outlines existing methodologies in Mpox image classification. The third section provides insights into the Mpox dataset and introduces the proposed enhancements. The fourth section details the performance evaluation and experimental results. Finally, the fifth section offers a comprehensive summary of the paper.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Literature review</title>
<p>Skin lesions are prevalent in clinical practice, making precise detection and diagnosis crucial for accurate patient treatment. In recent years, the emergence of machine learning technologies has provided significant potential to assist in the identification and clinical decision-making for skin lesions (<xref ref-type="bibr" rid="B13">Fraiwan and Faouri, 2022</xref>). In this section, we will present various artificial intelligence techniques, including machine learning, CNNs, and transfer learning algorithms, applied for the detection and diagnosis of Mpox virus.</p>
<p>Iftikhar et&#xa0;al. (<xref ref-type="bibr" rid="B22">Iftikhar et&#xa0;al., 2023</xref>) proposed a novel filtering and ensemble technique for the rapid and accurate prediction of Mpox cases. This approach generates two subsequences, long-term trend sequences, and residual sequences, through filtering, and employs five machine learning models for prediction. Mandal et&#xa0;al. (<xref ref-type="bibr" rid="B33">Mandal et&#xa0;al., 2022</xref>) introduced a clustering method for Mpox cases that combines machine learning and Particle Swarm Optimization (PSO). Bhosale et&#xa0;al. (<xref ref-type="bibr" rid="B12">Bhosale et&#xa0;al., 2022</xref>) conducted similar work, utilizing linear regression, decision trees, random forests, elasticNet, and ARIMA for the prediction of Mpox cases. They further introduced a dataset named Mpox Skin Image Dataset (MSID), widely employed in various studies. For instance, Khafaga et&#xa0;al. (<xref ref-type="bibr" rid="B28">Khafaga et&#xa0;al., 2022b</xref>) introduced a novel framework for the classification of Mpox disease images. They employed the Random Fractal Search (BERSFS) using the Al-Biruni Earth Radius (BER) optimization method for fine-tuning on deep CNN layers. Saleh and Rabie et&#xa0;al. (<xref ref-type="bibr" rid="B44">Saleh and Rabie, 2023</xref>) proposed a Human Mpox Diagnosis (HMD) strategy based on artificial intelligence technology. This strategy comprises two key components: utilizing Improved Binary Chimpanzee Optimization (IBCO) and selecting features with significant value for transfer to the diagnostic model of ensemble learning. Ultimately, HMD achieved an accuracy of 0.98. Ahsan et&#xa0;al. (<xref ref-type="bibr" rid="B3">Ahsan et&#xa0;al., 2023</xref>)developed a Mpox diagnostic model using the Generalization and Regularization-based Transfer Learning Approach (GRA-TLA) for binary and multiclass classification. They tested ten different Convolutional Neural Network (CNN) models, including both binary and multiclass tasks, in three independent studies. In studies one and two, their model combined with Extreme Inception (Xception) achieved accuracies ranging from 77% to 88% in distinguishing individuals with and without Mpox. In study three, the accuracy using the ResNet-101 network ranged from 84% to 99%. Kumar et&#xa0;al. (<xref ref-type="bibr" rid="B29">Kumar, 2022</xref>) employed skin images for Mpox disease diagnosis, using various CNN models and machine learning algorithms. They extracted image features using Vgg16Net and AlexNet and applied classifiers such as Naive Bayes, Decision Trees (DT), K-Nearest Neighbors (KNN), Support Vector Machine (SVM), and Random Forest. The Naive Bayes algorithm combined with Vgg16Net achieved the highest accuracy at 91.11%.</p>
<p>Research on skin lesion image classification through the combination of CNN and transfer learning methods has yielded significant benefits. Ahsan et&#xa0;al. (<xref ref-type="bibr" rid="B4">Ahsan et&#xa0;al., 2022</xref>) conducted an initial investigation into Mpox diagnosis. The authors collected images of patients infected with Mpox from various accessible portals and proposed a VGG16-based model for Mpox diagnosis. They constructed their dataset by gathering images from Google and utilized transfer learning to create a model based on the VGG16 architecture in two separate studies. The first study successfully distinguished Mpox from chickenpox, achieving an accuracy of up to 0.97, while the second study differentiated Mpox from other diseases (chickenpox, measles, and normal skin) with an accuracy of 0.89. Meena et&#xa0;al. (<xref ref-type="bibr" rid="B36">Meena et&#xa0;al., 2023</xref>) proposed a hybrid technique based on Convolutional Neural Networks (CNNs) and Long Short-Term Memory networks (LSTMs) to embed knowledge graphs into various healthcare applications, providing enhanced data representation and knowledge inference. Experimental results demonstrate that the proposed model achieves an accuracy of 94% on the Mpox dataset. Bala et&#xa0;al. (<xref ref-type="bibr" rid="B10">Bala et&#xa0;al., 2023</xref>) presented a Convolutional Neural Network (MonkeyNet) based on an improved DenseNet-201 for Mpox image recognition. They evaluated its performance using the original images from the MSID dataset for training. Their model accurately identified Mpox on both the original and augmented datasets, achieving accuracies of 93.19% and 98.91%, respectively. Jaradat et&#xa0;al. (<xref ref-type="bibr" rid="B24">Jaradat et&#xa0;al., 2023</xref>) evaluated five pre-trained models, including VGG19, VGG16, ResNet50, MobileNetV2, and EfficientB3. Experimental results demonstrated that MobileNetV2 performed the best, achieving an accuracy of 0.98. Model validation across different datasets confirmed MobileNetV2&#x2019;s highest accuracy of 0.94. Altun et&#xa0;al. (<xref ref-type="bibr" rid="B8">Altun et&#xa0;al., 2023</xref>) employed a similar approach and developed a hybrid function learning model incorporating hyperparameter optimization. They utilized custom models, including MobileNetV3-s, EfficientNetV2, ResNet50, Vgg19, DenseNet121, and Xception. Notably, the optimized MobileNetV3-s model exhibited the best performance, achieving an accuracy of 0.96. Uzun Ozsahin et&#xa0;al. (<xref ref-type="bibr" rid="B51">Uzun Ozsahin et&#xa0;al., 2023</xref>) applied deep learning models such as AlexNet, VGG16, and VGG19 for the detection task on Mpox and chickenpox datasets. Through their research methodology, they achieved a highest precision of 0.99. Sitaula et&#xa0;al. (<xref ref-type="bibr" rid="B50">Sitaula and Shahi, 2022</xref>) utilized deep learning techniques for Mpox diagnosis. They compared 13 pre-trained deep learning models, ultimately selecting the most outstanding model to build their system. The results indicated an accuracy of 0.87 for their Mpox diagnostic model. Ali et&#xa0;al. (<xref ref-type="bibr" rid="B6">Ali et&#xa0;al., 2022</xref>) created a dataset named &#x201c;Monkeypox Skin Lesion Dataset (MSLD)&#x201d; designed for automatic detection of Mpox disease from skin lesions. The images were primarily sourced from websites, news portals, and publicly available case reports. They employed pre-trained models such as VGG-16, ResNet50, and InceptionV3 for Mpox classification. To enhance the accuracy of Mpox detection, Sahin et&#xa0;al. employed six deep learning models, including ResNet-18, MobileNet, NasNetMobile, GoogLeNet, EfficientB0, and ShuffleNet, to distinguish Mpox images from images depicting other diseases. Among these models, MobileNet exhibited the best performance, achieving a 91.11% accuracy in image classification. Additionally, Almufareh et&#xa0;al. (<xref ref-type="bibr" rid="B7">Almufareh et&#xa0;al., 2023</xref>) utilized various independent CNN architectures to differentiate Mpox from non-Mpox cases. They validated their models using MSID and MSLD datasets. Javelle et&#xa0;al. (<xref ref-type="bibr" rid="B25">Javelle et&#xa0;al., 2023</xref>) redefined the emerging Mpox disease and designed a self-management questionnaire for case management, contact monitoring, and support for clinical research. Haque et&#xa0;al. (<xref ref-type="bibr" rid="B17">Haque et&#xa0;al., 2022</xref>) employed five deep learning models, VGG19, Xception, DenseNet121, EfficientNetB3, and MobileNetV2, and integrated spatial attention mechanisms for accurate classification of human Mpox. Yasmin et&#xa0;al. (<xref ref-type="bibr" rid="B55">Yasmin et&#xa0;al., 2022</xref>) obtained Mpox images from the Kaggle global dataset and used nine models for predictions. Among them, the optimal predictive model achieved an MSE value of 41922.55, R2 of 0.49, MAPE of 16.82, MAE of 146.29, and RMSE of 204.75. Meena et&#xa0;al. (<xref ref-type="bibr" rid="B35">Meena et&#xa0;al., 2024</xref>) established a deep learning model based on transfer learning to assist in diagnosing whether patients have Mpox. In the experiment, the InceptionV3 model they used achieved an accuracy of 98%.</p>
<p>
<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> summarizes the relevant studies on Mpox diagnosis. However, research on Mpox diagnosis using deep learning remains relatively limited. Some investigations have assessed the potential of deep learning algorithms in identifying this disease. While the results suggest that deep learning could be a valuable tool for the diagnosis and control of Mpox, further research is needed to validate these findings and establish practical applications in real clinical settings.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Summary of literature review.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Methodology</th>
<th valign="middle" align="left">Approaches</th>
<th valign="middle" align="left">Dataset</th>
<th valign="middle" align="left">Best Results</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Sahin et&#xa0;al. (<xref ref-type="bibr" rid="B43">Sahin et&#xa0;al., 2022</xref>)</td>
<td valign="middle" align="left">ResNet18, GoogleNet, EfficientNetB0,<break/>NasnetMobile, ShufeNet, and MobileNetV2</td>
<td valign="middle" align="left">MSLD</td>
<td valign="top" align="left">Accuracy: 0.91</td>
</tr>
<tr>
<td valign="middle" align="left">Ali et&#xa0;al. (<xref ref-type="bibr" rid="B6">Ali et&#xa0;al., 2022</xref>)</td>
<td valign="middle" align="left">VGG16, ResNet50, and InceptionV3</td>
<td valign="middle" align="left">Custom Dataset</td>
<td valign="top" align="left">Accuracy: 0.82</td>
</tr>
<tr>
<td valign="middle" align="left">Ahsan et&#xa0;al. (<xref ref-type="bibr" rid="B4">Ahsan et&#xa0;al., 2022</xref>)</td>
<td valign="middle" align="left">VGG16</td>
<td valign="middle" align="left">Custom Dataset</td>
<td valign="top" align="left">AUC: 0.97</td>
</tr>
<tr>
<td valign="middle" align="left">Kumar et&#xa0;al. (<xref ref-type="bibr" rid="B29">Kumar, 2022</xref>)</td>
<td valign="middle" align="left">CNN models AlexNet, GoogleNet and VGG16Net with Na&#xef;ve Bayes, SVM, KNN,<break/>Random Forest, and Decision Tree</td>
<td valign="middle" align="left">Ali et&#xa0;al. (<xref ref-type="bibr" rid="B6">Ali et&#xa0;al., 2022</xref>)</td>
<td valign="top" align="left">Accuracy: 0.91</td>
</tr>
<tr>
<td valign="middle" align="left">Jaradat et&#xa0;al. (<xref ref-type="bibr" rid="B24">Jaradat et&#xa0;al., 2023</xref>)</td>
<td valign="middle" align="left">Xception, DenseNet</td>
<td valign="middle" align="left">Custom Dataset</td>
<td valign="top" align="left">Accuracy: 0.98<break/>Precision: 0.99<break/>Recall: 0.96<break/>F-score: 0.98</td>
</tr>
<tr>
<td valign="middle" align="left">Altun et&#xa0;al. (<xref ref-type="bibr" rid="B8">Altun et&#xa0;al., 2023</xref>)</td>
<td valign="middle" align="left">CNN model based on MobileNetV3-s, EfficientNetV2, ResNet50, VGG19, DenseNet121, and Xception models</td>
<td valign="middle" align="left">Custom Dataset</td>
<td valign="top" align="left">Accuracy: 0.96</td>
</tr>
<tr>
<td valign="middle" align="left">Sitaula et&#xa0;al. (<xref ref-type="bibr" rid="B50">Sitaula and Shahi, 2022</xref>)</td>
<td valign="middle" align="left">VGG-16, VGG-19,ResNet50, ResNet101,IncepResNetv2, MobileNetV2, InceptionV3, Xception, EfficientNetB0, EfficientNetB1, EfficientNetB2, DenseNet121 and DenseNet169</td>
<td valign="middle" align="left">Ahsan et&#xa0;al. (<xref ref-type="bibr" rid="B4">Ahsan et&#xa0;al., 2022</xref>)</td>
<td valign="top" align="left">Accuracy:<break/>0.85</td>
</tr>
<tr>
<td valign="middle" align="left">Yasmin et&#xa0;al. (<xref ref-type="bibr" rid="B55">Yasmin et&#xa0;al., 2022</xref>)</td>
<td valign="middle" align="left">Polynomial Regression, SVR, Holt&#x2019;s Linear Model AR Model, SARIMA Model ARIMA Model, MA Model, Holt-Winter&#x2019;s Model, and Prophet Model</td>
<td valign="middle" align="left">Custom Dataset</td>
<td valign="top" align="left">MSE: 41,922.55<break/>R2: 0.49<break/>MAPE: 16.82<break/>MAE: 146.29<break/>RMSE: 204.75</td>
</tr>
<tr>
<td valign="middle" align="left">Alwakid et&#xa0;al. (<xref ref-type="bibr" rid="B9">Alwakid et&#xa0;al., 2022</xref>)</td>
<td valign="middle" align="left">ResNet50</td>
<td valign="top" align="left">HAM10000</td>
<td valign="top" align="left">Accuracy: 0.86<break/>Precision: 0.84<break/>Recall: 0.86<break/>F-score: 0.86</td>
</tr>
<tr>
<td valign="middle" align="left">Abdelhamid et&#xa0;al. (<xref ref-type="bibr" rid="B1">Abdelhamid et&#xa0;al., 2022</xref>)</td>
<td valign="middle" align="left">Binary PSOBER algorithm</td>
<td valign="middle" align="left">Bala et&#xa0;al. (<xref ref-type="bibr" rid="B10">Bala et&#xa0;al., 2023</xref>)</td>
<td valign="top" align="left">Accuracy: 0.98</td>
</tr>
<tr>
<td valign="middle" align="left">Proposed Method</td>
<td valign="middle" align="left">ConvNext</td>
<td valign="middle" align="left">Custom Dataset</td>
<td valign="top" align="left">Accuracy: 0.97<break/>Precision: 0.96<break/>Recall: 0.93<break/>F-score: 0.95</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3" sec-type="materials|methods">
<label>3</label>
<title>Materials and proposed methods</title>
  <p>This section encompasses four key stages: data collection, image preprocessing, model selection, and model optimization. In the first stage, we utilized the Monkeypox Skin Image Dataset (MSID), which is freely accessible on the Kaggle platform (<xref ref-type="bibr" rid="B39">Monkeypox Skin Images Dataset (MSID), n.d</xref>).. And due to limited data, we employed data augmentation techniques to generate additional images. During the data preprocessing phase, the collected images underwent operations such as resizing, normalization, and data augmentation, which are crucial for enhancing model performance. We selected eight commonly used models (SqueezeNet, ResNet18, ResNet34, ResNet50, Vgg16, DenseNet121, Swin-Tiny, and ConvNext-Tiny) for comparison to improve the accuracy of the Mpox virus detection model. In the model training phase, the selected models were trained using the preprocessed images, optimizing model performance by providing images and adjusting parameters. Finally, in the evaluation stage, metrics such as accuracy, precision, recall, and F1 score were employed to assess the models, with the best-performing model selected as the final model. Therefore, this approach utilizes deep learning techniques to analyze Mpox images, accurately classifying and diagnosing the disease based on visual features. The algorithm demonstrates high accuracy in disease classification, providing a valuable tool for the rapid and precise diagnosis of Mpox in clinical settings. The proposed method comprises multiple stages and processes, illustrated in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>, which encompass various steps and operations.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Processes diagram of the proposed method.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcimb-14-1397316-g001.tif"/>
</fig>
<sec id="s3_1">
<label>3.1</label>
<title>Data collection</title>
  <p>This study utilized the Mpox Skin Image Dataset (MSID), a freely available resource on the Kaggle platform (<xref ref-type="bibr" rid="B39">Monkeypox Skin Images Dataset (MSID), n.d</xref>)., and categorized the images into two classes: Mpox and non-Mpox. This dataset serves as a crucial resource for in-depth research on the Mpox virus and its impact on human health. The dataset includes high-resolution images that intricately depict the manifestations of Mpox at different stages, as well as the infection symptoms it induces on human skin. The dataset provides detailed visual information on skin lesions, rashes caused by Mpox, and images related to other skin conditions, offering comprehensive visual insights. By encompassing a large number of images covering various aspects of Mpox, researchers are able to delve into the development of Mpox disease and identify crucial features that contribute to accurate diagnosis. Leveraging the outstanding performance of deep learning models in image enhancement, this study fully capitalized on their advantages.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Image preprocessing</title>
<p>The preprocessing stage plays a crucial role in image analysis as it contributes to enhancing data quality and consistency. The preprocessing steps in this study include image scaling and data augmentation. In this research, considering the trade-off between training speed and accuracy, images were resized to 224&#xd7;224 pixels, facilitating easier handling by the model due to the uniformity of image sizes. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> illustrates how data augmentation techniques were applied to enhance the quality of Mpox samples, thereby expanding the dataset and preventing overfitting. Through this augmented dataset, the model&#x2019;s performance was successfully improved, and overfitting was mitigated. The following methods were employed for data augmentation on Mpox images:</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Mpox images enhanced using data augmentation.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcimb-14-1397316-g002.tif"/>
</fig>
<p>Gaussian Noise: Adding Gaussian-distributed noise to the images simulates random signal interference during capture or transmission processes, providing a way to assess the algorithm&#x2019;s performance in real-noise environments.</p>
<p>Random Cropping: Randomly cropping images is a common data augmentation method that increases the variation range of images, enhancing the model&#x2019;s generalization capability and robustness.</p>
<p>Random Rotation (0&#x2013;360 degrees): Randomly rotating images without altering the original image information helps reduce overfitting and improve model efficiency.</p>
<p>Horizontal or Vertical Flipping: Flipping images horizontally or vertically based on probability.</p>
<p>Finally, the Mpox dataset comprises 2000 images after augmentation, with 1400 images used for model training, 400 for testing, and 200 for validation.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Basic architecture selection</title>
<p>In order to enhance the classification performance on the augmented Mpox dataset, this study drew inspiration from algorithms used in other research [23, 29, 48]. However, the relevant literature did not disclose the code in their papers, and there were differences in the datasets used, making it challenging to directly compare the proposed algorithm with theirs. Therefore, this paper evaluated the classification performance of the base models [SqueezeNet (<xref ref-type="bibr" rid="B20">Iandola et&#xa0;al., 2016</xref>), ResNet18 (<xref ref-type="bibr" rid="B18">He et&#xa0;al., 2015</xref>), ResNet34, ResNet50, Vgg16 (<xref ref-type="bibr" rid="B47">Simonyan and Zisserman, 2015</xref>), DenseNet121 (<xref ref-type="bibr" rid="B19">Huang et&#xa0;al., 2018</xref>), Swin-Tiny (<xref ref-type="bibr" rid="B31">Liu et&#xa0;al., 2021</xref>), and ConvNext-Tiny (<xref ref-type="bibr" rid="B32">Liu et&#xa0;al., 2022</xref>)] employed on the Mpox dataset. The reason for selecting these models is that most of them have been widely applied in the literature. Additionally, each model has a unique architecture and strengths, and combining their strengths may contribute to improving overall performance. The experimental results on the test set are presented in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. On the test set, ConvNext-Tiny demonstrated the best generalization performance with an accuracy of 94.33%. Therefore, building upon this foundation, the paper further improved and proposed MpoxNet, which exhibits superior performance. After our enhancements, MpoxNet is much lighter than ConvNext-Tiny, with parameters reduced by less than 68.13%, while achieving a higher accuracy of 0.95%. This makes MpoxNet more suitable for the classification diagnosis of Mpox.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Results of different network on MSID dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left"/>
<th valign="middle" align="left">Accuracy(%)</th>
<th valign="middle" align="left">Precision(%)</th>
<th valign="middle" align="left">Recall(%)</th>
<th valign="middle" align="left">F1-score(%)</th>
<th valign="middle" align="left">Flops(G)</th>
<th valign="middle" align="left">Params(M)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">SqueezeNet</td>
<td valign="middle" align="left">74.52</td>
<td valign="middle" align="left">71.60</td>
<td valign="middle" align="left">71.10</td>
<td valign="middle" align="left">77.10</td>
<td valign="middle" align="left">
<bold>23.44</bold>
</td>
<td valign="middle" align="left">
<bold>0.73</bold>
</td>
</tr>
<tr>
<td valign="middle" align="left">ResNet34</td>
<td valign="middle" align="left">81.76</td>
<td valign="middle" align="left">76.60</td>
<td valign="middle" align="left">85.20</td>
<td valign="middle" align="left">82.70</td>
<td valign="middle" align="left">117.70</td>
<td valign="middle" align="left">21.28</td>
</tr>
<tr>
<td valign="middle" align="left">ResNet50</td>
<td valign="middle" align="left">85.53</td>
<td valign="middle" align="left">88.10</td>
<td valign="middle" align="left">78.20</td>
<td valign="middle" align="left">87.50</td>
<td valign="middle" align="left">132.21</td>
<td valign="middle" align="left">23.51</td>
</tr>
<tr>
<td valign="middle" align="left">ResNet18</td>
<td valign="middle" align="left">88.67</td>
<td valign="middle" align="left">87.90</td>
<td valign="middle" align="left">86.60</td>
<td valign="middle" align="left">89.80</td>
<td valign="middle" align="left">58.35</td>
<td valign="middle" align="left">11.17</td>
</tr>
<tr>
<td valign="middle" align="left">VGG16</td>
<td valign="middle" align="left">90.88</td>
<td valign="middle" align="left">89.00</td>
<td valign="middle" align="left">90.80</td>
<td valign="middle" align="left">91.70</td>
<td valign="middle" align="left">495.04</td>
<td valign="middle" align="left">138.35</td>
</tr>
<tr>
<td valign="middle" align="left">DenseNet121</td>
<td valign="middle" align="left">91.50</td>
<td valign="middle" align="left">90.20</td>
<td valign="middle" align="left">90.80</td>
<td valign="middle" align="left">92.30</td>
<td valign="middle" align="left">92.67</td>
<td valign="middle" align="left">6.95</td>
</tr>
<tr>
<td valign="middle" align="left">Swin-Tiny</td>
<td valign="middle" align="left">92.76</td>
<td valign="middle" align="left">93.40</td>
<td valign="middle" align="left">90.10</td>
<td valign="middle" align="left">93.60</td>
<td valign="middle" align="left">139.88</td>
<td valign="middle" align="left">27.50</td>
</tr>
<tr>
<td valign="middle" align="left">ConvNext-Tiny</td>
<td valign="middle" align="left">94.33</td>
<td valign="middle" align="left">92.50</td>
<td valign="middle" align="left">91.70</td>
<td valign="middle" align="left">94.80</td>
<td valign="middle" align="left">142.55</td>
<td valign="middle" align="left">27.80</td>
</tr>
<tr>
<td valign="middle" align="left">MpoxNet(Our)</td>
<td valign="middle" align="left">
<bold>95.28</bold>
</td>
<td valign="middle" align="left">
<bold>96.40</bold>
</td>
<td valign="middle" align="left">
<bold>93.00</bold>
</td>
<td valign="middle" align="left">
<bold>95.80</bold>
</td>
<td valign="middle" align="left">43.74</td>
<td valign="middle" align="left">8.86</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The best results are highlighted in bold text.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>ConvNext</title>
<p>ConvNext is a pure CNN model proposed by Liu et&#xa0;al (<xref ref-type="bibr" rid="B32">Liu et&#xa0;al., 2022</xref>), designed to eliminate cumbersome operations such as window movement and relative position deviation, providing superior performance and lower computational burden compared to popular transformer networks. The overall structure of ConvNext is based on the ResNet design, incorporating residual blocks and combining various advanced network design techniques to further enhance the overall performance of the network. The detailed structures of ConvNext-Tiny and ConvNext blocks are illustrated in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>
<bold>(A)</bold> ConvNext-Tiny network structure; <bold>(B)</bold> ConvNext Block structure.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcimb-14-1397316-g003.tif"/>
</fig>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>D<sup>2</sup>RSE block</title>
<p>In terms of improving accuracy, increasing the cardinality of the network is more effective than increasing depth or width. This viewpoint was initially proposed in ResNeXt (<xref ref-type="bibr" rid="B54">Xie et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B26">Jiang et&#xa0;al., 2023</xref>). Inspired by this, this paper introduces a Dual-Branch Depthwise Separable Convolution Residual Squeeze and Excitation (D<sup>2</sup>RSE) module in ConvNext. One branch follows a traditional modular design, while the other branch adopts the unique design of two consecutive convolutions in the Wide Residual Network. The dimensions of both branches are set to half of the main branch. Meanwhile, to reduce the number of model parameters, depthwise separable convolutions are used instead of traditional convolutions. Subsequently, by concatenating the output dimensions of the two branches and applying operations such as Batch Normalization and GELU, overfitting of the model is effectively prevented, thereby improving overall performance. Finally, an SE (Squeeze and Excitation) block is added after GELU. It explicitly models interdependencies among convolutional feature channels, allowing adaptive allocation of weights for different channels. This enables the network to perform dynamic channel-wise feature recalibration, enhancing the representation capacity of the network. Through this mechanism, the network learns to selectively emphasize informative features using global information and suppress less useful features. Furthermore, a residual connection is established between the original output and the final layer output, effectively increasing the depth of the model and enhancing its representational capacity and performance. In addition to increasing depth, the residual connection facilitates more effective learning of critical features by providing a mechanism for direct connections across layers, further improving overall performance. The D<sup>2</sup>RSE and SE modules are illustrated in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Configuration of D<sup>2</sup>RSE and SE Block.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcimb-14-1397316-g004.tif"/>
</fig>
<p>Specifically, the input feature map of the D<sup>2</sup>RSE block is processed through depthwise separable convolution and LayerNorm. The feature map captured by depthwise separable convolution and LayerNorm can be represented as shown in <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msup>
<mml:mtext mathvariant="bold-italic">F</mml:mtext>
<mml:mn mathvariant="bold">7</mml:mn>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mtext mathvariant="bold-italic">L</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mtext mathvariant="bold-italic">W</mml:mtext>
<mml:mrow>
<mml:mn mathvariant="bold">7</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn mathvariant="bold">7</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mtext mathvariant="bold-italic">F</mml:mtext>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">input</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the input feature map, <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mn>7</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:mn>7</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> convolution matrix, and <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents Layer-normalization. Subsequently, the feature map is integrated into each branch. To extract valuable target features from the feature map, we first design two convolutional modules to extract features, guiding the network to learn more robust feature representations. The obtained features can be shown as in <xref ref-type="disp-formula" rid="eq2">Equation 2</xref>:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>L</mml:mi>
</mml:msup>
</mml:mstyle>
<mml:mo>=</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mstyle>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mtext mathvariant="bold-italic">F</mml:mtext>
<mml:mn mathvariant="bold">7</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>First, we input the processed combined feature map <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mn>7</mml:mn>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mfrac>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> into the left branch to compress it into a new feature map <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>L</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mfrac>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. Here, <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represent the fully connected layer matrix and the GELU activation operation, respectively. Meanwhile, we use depthwise separable convolution instead of regular convolution to reduce the number of model parameters. Additionally, GELU activation and Batch Normalization are applied and shown in <xref ref-type="disp-formula" rid="eq3">Equation 3</xref>.</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>R</mml:mi>
</mml:msup>
</mml:mstyle>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mtext mathvariant="bold-italic">W</mml:mtext>
<mml:mrow>
<mml:mn mathvariant="bold">3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn mathvariant="bold">3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">B</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mtext mathvariant="bold-italic">W</mml:mtext>
<mml:mrow>
<mml:mn mathvariant="bold">3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn mathvariant="bold">3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mtext mathvariant="bold-italic">F</mml:mtext>
<mml:mn mathvariant="bold">7</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>It is noteworthy that these two-branch maps help us extract more representative feature maps from different scales of receptive fields. The feature maps <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>L</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mfrac>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>R</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mfrac>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are concatenated, followed by BatchNorm and GELU operations. The obtained features can be shown as in <xref ref-type="disp-formula" rid="eq4">Equation 4</xref>:</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>G</mml:mi>
</mml:msup>
</mml:mstyle>
<mml:mo>=</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">B</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>L</mml:mi>
</mml:msup>
</mml:mstyle>
<mml:mo>&#x2225;</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>R</mml:mi>
</mml:msup>
</mml:mstyle>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im11">
<mml:mo>&#x2225;</mml:mo>
</mml:math>
</inline-formula> denotes concatenation along the channel dimension. The processed feature map <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>G</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is then input into the SE block.</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>S</mml:mi>
</mml:msup>
</mml:mstyle>
<mml:mo>=</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>G</mml:mi>
</mml:msup>
</mml:mstyle>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext mathvariant="bold-italic">s</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mtext mathvariant="bold-italic">W</mml:mtext>
<mml:mn mathvariant="bold">2</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mtext mathvariant="bold-italic">W</mml:mtext>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>G</mml:mi>
</mml:msup>
</mml:mstyle>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the <inline-formula>
<mml:math display="inline" id="im14">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> activation function, <inline-formula>
<mml:math display="inline" id="im15">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes global average pooling operation. <xref ref-type="disp-formula" rid="eq5">Equation 5</xref> can be automatically backpropagated for training, adjusting the values of <inline-formula>
<mml:math display="inline" id="im16">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im17">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> &#x200b; through gradient descent to optimize the model&#x2019;s performance. Finally, the output feature map <inline-formula>
<mml:math display="inline" id="im18">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>S</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is added to our input feature map, resulting in the final representation. The obtained features can be shown as in <xref ref-type="disp-formula" rid="eq6">Equation 6</xref>:</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>S</mml:mi>
</mml:msup>
</mml:mstyle>
<mml:mo>=</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>G</mml:mi>
</mml:msup>
</mml:mstyle>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext mathvariant="bold-italic">s</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mtext mathvariant="bold-italic">W</mml:mtext>
<mml:mn mathvariant="bold">2</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mtext mathvariant="bold-italic">W</mml:mtext>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>G</mml:mi>
</mml:msup>
</mml:mstyle>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im19">
<mml:mo>&#x2295;</mml:mo>
</mml:math>
</inline-formula> represents element-wise addition. Subsequently, <inline-formula>
<mml:math display="inline" id="im20">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>D</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> serves as the input for the next stage and is also the output of the entire D<sup>2</sup>RSE module.</p>
</sec>
<sec id="s3_6">
<label>3.6</label>
<title>CBAM</title>
<p>CBAM introduces spatial and channel attention mechanisms, enhancing the performance of convolutional neural networks. The spatial attention dynamically adjusts the importance of different image positions, while the channel attention adaptively adjusts the importance of different channels. This attention mechanism facilitates more effective capture of critical features in Mpox images, thereby improving performance across various visual tasks. CBAM, proposed by Woo et&#xa0;al. (<xref ref-type="bibr" rid="B53">Woo et&#xa0;al., 2018</xref>), is a simple yet effective feedforward convolutional neural network attention module. It infers attention maps in a sequential manner for channel and spatial dimensions independently and then multiplies this map by the input feature map to achieve adaptive feature refinement. Moreover, as CBAM is a lightweight and general-purpose module, it can be seamlessly integrated into any CNN architecture, allowing for end-to-end training alongside the base CNN. <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref> illustrates the structure of CBAM, which includes both channel attention and spatial attention modules.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Convolution Block Attention Module.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcimb-14-1397316-g005.tif"/>
</fig>
<p>As illustrated in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>, the feature map initially undergoes the channel attention module. This module aggregates spatial information of the feature map using average pooling and max pooling operations, generating two distinct spatial context features, namely <inline-formula>
<mml:math display="inline" id="im21">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mi>c</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im22">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mi>c</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. Subsequently, these two features are passed through a shared network consisting of a multilayer perceptron (MLP) and a hidden layer to generate the channel attention map <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:msup>
<mml:mi>M</mml:mi>
<mml:mi>C</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. Finally, the channel attention map is fused with the original feature map through element-wise summation, forming the ultimate feature output. The feature output is then forwarded to the spatial attention module, which employs two pooling operations to aggregate the channel information of the feature map, generating two 2D maps, <inline-formula>
<mml:math display="inline" id="im24">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mi>S</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im25">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mi>S</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. These maps are then concatenated and convolved through a standard convolutional layer to produce the final 2D spatial attention map. The mathematical expressions for the aforementioned operations are given by <xref ref-type="disp-formula" rid="eq7">Equations 7</xref>, <xref ref-type="disp-formula" rid="eq8">8</xref>.</p>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msup>
<mml:mi>M</mml:mi>
<mml:mi>C</mml:mi>
</mml:msup>
</mml:mstyle>
<mml:mo>=</mml:mo>
<mml:mtext mathvariant="bold-italic">s</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mtext mathvariant="bold-italic">W</mml:mtext>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mtext mathvariant="bold-italic">W</mml:mtext>
<mml:mn mathvariant="bold">0</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mi>c</mml:mi>
</mml:msubsup>
</mml:mstyle>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mtext mathvariant="bold-italic">W</mml:mtext>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mtext mathvariant="bold-italic">W</mml:mtext>
<mml:mn mathvariant="bold">0</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mi>c</mml:mi>
</mml:msubsup>
</mml:mstyle>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msup>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
</mml:msup>
</mml:mstyle>
<mml:mo>=</mml:mo>
<mml:mtext mathvariant="bold-italic">s</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mtext mathvariant="bold-italic">W</mml:mtext>
<mml:mrow>
<mml:mn mathvariant="bold">7</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn mathvariant="bold">7</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mi>S</mml:mi>
</mml:msubsup>
</mml:mstyle>
<mml:mo>;</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mi>S</mml:mi>
</mml:msubsup>
</mml:mstyle>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In the study, we further explored the embedding positions of CBAM in the model and designed three variations, as shown in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>: (a) indicates the usage of CBAM after each ConvNext block operation; (b) represents the model using CBAM after each downsampling; (c) denotes the model incorporating CBAM before each downsampling. Experimental results revealed that (c) exhibited superior performance, and consequently, our model adopted the design of embedding CBAM after each downsampling.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>CBAM embedding position design. <bold>(A)</bold> denotes the use of CBAM after each ConvNext block operation; <bold>(B)</bold> denotes the use of CBAM after each downsampling in the model; <bold>(C)</bold> denotes the use of CBAM before each downsampling in the model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcimb-14-1397316-g006.tif"/>
</fig>
</sec>
<sec id="s3_7">
<label>3.7</label>
<title>MpoxNet</title>
<p>Firstly, we devised a dual-branch structure based on the ConvNext Block, named the D2RSE module. Subsequently, we integrated the CBAM module into ConvNext, ultimately constructing a high-precision and lightweight network specifically designed for Mpox classification&#x2014;MpoxNet. Experimental results demonstrated that MpoxNet excelled in the Mpox classification task. The overall network architecture is illustrated in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Explanation of MpoxNet network architecture.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcimb-14-1397316-g007.tif"/>
</fig>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Performance evaluation and experimental results</title>
<sec id="s4_1">
<label>4.1</label>
<title>Parameters and evaluation metrics</title>
<p>All experiments were conducted on a high-performance deep learning server with the following hardware configuration: Intel Xeon Silver 4210 CPU with a clock speed of 2.20GHz, NVIDIA GeForce RTX 2080 Ti graphics processing unit with 11GB of video memory, and 128GB of RAM. The deep learning framework employed was Python 3.8.10, Cuda 10.2, torch 1.8.1, and torchvision 0.9.1. The operating system used was Windows 10. During the experiments, consistent training parameters and configurations were applied to train various models. The size of training images was fixed at 224&#xd7;224, and the batch size was set to 16. Model training utilized the cross-entropy loss function and the AdamW optimizer. In the initial training stage, a warm-up of 1 epoch was performed. The warm-up phase involved gradually updating the learning rate for each iteration using one-dimensional linear interpolation. Following the warm-up, a cosine annealing function was employed to decay the learning rate, starting with an initial learning rate of 0.0005. To ensure fair performance comparisons across different models, no transfer learning was utilized, and each model underwent training for 300 epochs.</p>
<p>Confusion matrix is utilized to calculate performance metrics such as accuracy, precision, recall, and F1 score by comparing predicted labels with actual labels. We employed five widely used performance metrics, including accuracy, precision, recall, and F1-score.</p>
<p>Accuracy is the most commonly used metric, representing the proportion of correctly classified samples out of the total number of samples. The expression is as in <xref ref-type="disp-formula" rid="eq9">Equation 9</xref>.</p>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>A</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
</mml:mstyle>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mstyle>
<mml:mo>+</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mstyle>
<mml:mo>+</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mstyle>
<mml:mo>+</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mstyle>
<mml:mo>+</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mstyle>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Precision represents the proportion of true positive samples among those classified as positive. In this context, it is defined as the ratio of samples correctly predicted as Mpox by the model to all samples predicted as Mpox by the model. The expression is as in <xref ref-type="disp-formula" rid="eq10">Equation 10</xref>:</p>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mstyle>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mstyle>
<mml:mo>+</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mstyle>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Recall, also known as sensitivity or true positive rate, measures the proportion of correctly predicted positive samples among all actual positive samples. The recall formula is as shown in <xref ref-type="disp-formula" rid="eq11">Equation 11</xref>.</p>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mstyle>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mstyle>
<mml:mo>+</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mstyle>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The F1-score, also known as the F-measure, is a metric for classification problems that considers both precision and recall. It is the harmonic mean of precision and recall, providing a balance between the two metrics. The F1-score formula is as shown in <xref ref-type="disp-formula" rid="eq12">Equation 12</xref>.</p>
<disp-formula id="eq12">
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:mtext mathvariant="bold-italic">F</mml:mtext>
<mml:mn mathvariant="bold">1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
</mml:mstyle>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn mathvariant="bold">2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mstyle>
<mml:mo>&#xd7;</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mstyle>
<mml:mo>+</mml:mo>
<mml:mstyle mathvariant="bold-italic">
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mstyle>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>TP (True Positive): Instances where the actual class is Mpox, and the model correctly predicts it as Mpox.</p>
<p>TN (True Negative): Instances where both the actual class and the predicted class are not Mpox.</p>
<p>FP (False Positive): Instances where the actual class is not Mpox, but the model incorrectly predicts it as Mpox.</p>
<p>FN (False Negative): Instances where the actual class is Mpox, but the model incorrectly predicts it as not Mpox.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Basic network results</title>
<p>
<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> presents the performance evaluation of our proposed model compared to other trained models. The evaluation results demonstrate that MpoxNet outperforms other networks significantly in terms of prediction accuracy and parameter efficiency on the MSID dataset, achieving an accuracy of 95.28%, precision of 96.40%, recall of 93.00%, and F1-score of 95.80%. Compared to the ConvNext-Tiny model, this paper not only improves the model&#x2019;s accuracy but also significantly reduces FLOPS and parameter count, being only 30.68% and 31.87% of ConvNext-Tiny, respectively. ConvNext exhibits good fitting performance with an accuracy of 94.33%, and MpoxNet also demonstrates excellent fitting capabilities. The SqueezeNet model has the lowest accuracy at 75.94%, and due to its concise network architecture, the FLOPS and parameter count of this model are only 23.44G and 0.73M, making it unsuitable for practical applications directly. Additionally, com-pared to ResNet18 and ResNet50, ResNet34 performs relatively poorly, with a classification accuracy of only 81.76%. ResNet50 and ResNet18 exhibit similar performance, but ResNet18 has only 44.13% and 47.51% of the FLOPS and model parameters of ResNet50, respectively. Although VGG16 performs well on the test set, its FLOPS and model parameter count are the highest. <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref> illustrates the learning curves of each model during training, including validation accuracy and training loss. In <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8A</bold>
</xref>, the horizontal axis represents the number of training epochs, and the vertical axis shows the corresponding validation accuracy. In <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8B</bold>
</xref>, the horizontal axis similarly represents the number of training epochs, and the vertical axis displays the change in training loss. <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8A</bold>
</xref> indicates that all models begin to converge around 30 epochs and gradually stabilize, reaching complete convergence by 300 epochs. Among these models, MpoxNet achieves the highest validation accuracy at 95.28%. <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8B</bold>
</xref> reveals relatively consistent trends in the training loss curves for CNN architecture models, with MpoxNet demonstrating out-standing fitting capabilities. In contrast, VGG16 exhibits lower convergence, possibly due to its larger FLOPS and model parameter count.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Training and validation of each model on the MSID dataset. <bold>(A)</bold> Trend of the validation accuracy curve of the model with the number of epochs. <bold>(B)</bold> Trend of the training loss curve of the model with the number of epochs.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcimb-14-1397316-g008.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref> displays the confusion matrices for all models tested on the MSID dataset. The confusion matrix is a crucial tool for evaluating the performance of classification problems, and its size depends on the output dimensions. In this binary classification problem, the matrix size is 2&#xd7;2. The confusion matrix compares the target output and the actual predicted output of the classification model, providing a more intuitive way to identify the strengths and weaknesses of the model and offering detailed performance in-sights. The analysis of the confusion matrix leads to the conclusion that most predictions of the model are accurate. The horizontal axis represents the true labels, while the vertical axis represents the model&#x2019;s predicted results. The elements on the diagonal indicate the model&#x2019;s correct predictions, while other elements represent instances of model mispredictions. Nevertheless, due to the high similarity of many images, potential inaccuracies still exist.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Confusion matrix for different networks.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcimb-14-1397316-g009.tif"/>
</fig>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>CBAM embedding position experiment</title>
<p>In Section 3.5, we proposed three different CBAM embedding position schemes, and the corresponding experimental results are detailed in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>. Observation reveals that ConvNext, as an advanced classification network, demonstrates robust performance, and the choice of CBAM embedding position in scheme (c) further enhances the model&#x2019;s performance. Therefore, after each downsampling operation in the network, we opt for the adoption of CBAM.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Comparison of different CBAM embedding position.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Method</th>
<th valign="middle" align="left">Accuracy(%)</th>
<th valign="middle" align="left">Precision(%)</th>
<th valign="middle" align="left">Recall(%)</th>
<th valign="top" align="left">F1-score(%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">ConvNext-Tiny</td>
<td valign="middle" align="left">94.33</td>
<td valign="middle" align="left">92.50</td>
<td valign="middle" align="left">91.70</td>
<td valign="middle" align="left">94.80</td>
</tr>
<tr>
<td valign="middle" align="left">ConvNext with(a)</td>
<td valign="middle" align="left">94.52</td>
<td valign="middle" align="left">92.63</td>
<td valign="middle" align="left">
<bold>94.40</bold>
</td>
<td valign="middle" align="left">94.60</td>
</tr>
<tr>
<td valign="middle" align="left">ConvNext with(b)</td>
<td valign="middle" align="left">94.45</td>
<td valign="middle" align="left">92.64</td>
<td valign="middle" align="left">93.00</td>
<td valign="middle" align="left">94.10</td>
</tr>
<tr>
<td valign="middle" align="left">ConvNext with(c)</td>
<td valign="middle" align="left">
<bold>94.65</bold>
</td>
<td valign="middle" align="left">
<bold>94.30</bold>
</td>
<td valign="middle" align="left">93.70</td>
<td valign="middle" align="left">
<bold>95.20</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The best results are highlighted in bold text.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Ablation experiments</title>
<p>We validated the effectiveness of the D2RSE and CBAM module in MpoxNet through ablation experiments on the test set, and the specific results are presented in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>. Observing the experimental data, the CBAM module improved the classification accuracy by 0.32% with almost no increase in the number of model parameters. Combined with the use of the D2RSE module, MpoxNet achieved a classification accuracy of 95.28%, while significantly reducing the FLOPS and parameter count by approximately 69.31% and 68.13%, respectively. This demonstrates the effectiveness of these two modules in enhancing the recognition performance of MpoxNet for Mpox and achieving lightweight model design.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Comparison of ablation experiments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">D<sup>2</sup>RSE</th>
<th valign="middle" align="left">CBAM</th>
<th valign="middle" align="left">Accuracy(%)</th>
<th valign="middle" align="left">Precision(%)</th>
<th valign="middle" align="left">Recall(%)</th>
<th valign="middle" align="left">F1-score(%)</th>
<th valign="middle" align="left">Flops(G)</th>
<th valign="middle" align="left">Params(M)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">&#x2013;</td>
<td valign="middle" align="left">&#x2013;</td>
<td valign="middle" align="left">94.33</td>
<td valign="middle" align="left">92.50</td>
<td valign="middle" align="left">95.10</td>
<td valign="middle" align="left">94.80</td>
<td valign="middle" align="left">142.55</td>
<td valign="middle" align="left">27.80</td>
</tr>
<tr>
<td valign="middle" align="left">&#x221a;</td>
<td valign="middle" align="left">&#x2013;</td>
<td valign="middle" align="left">95.33</td>
<td valign="middle" align="left">94.70</td>
<td valign="middle" align="left">94.80</td>
<td valign="middle" align="left">95.80</td>
<td valign="middle" align="left">
<bold>43.71</bold>
</td>
<td valign="middle" align="left">
<bold>8.83</bold>
</td>
</tr>
<tr>
<td valign="middle" align="left">&#x2013;</td>
<td valign="middle" align="left">&#x221a;</td>
<td valign="middle" align="left">94.65</td>
<td valign="middle" align="left">94.30</td>
<td valign="middle" align="left">93.70</td>
<td valign="middle" align="left">95.20</td>
<td valign="middle" align="left">142.65</td>
<td valign="middle" align="left">27.90</td>
</tr>
<tr>
<td valign="middle" align="left">&#x221a;</td>
<td valign="middle" align="left">&#x221a;</td>
<td valign="middle" align="left">
<bold>95.28</bold>
</td>
<td valign="middle" align="left">
<bold>96.40</bold>
</td>
<td valign="middle" align="left">
<bold>93.00</bold>
</td>
<td valign="middle" align="left">
<bold>95.80</bold>
</td>
<td valign="middle" align="left">43.74</td>
<td valign="middle" align="left">8.86</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The best results are highlighted in bold text.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s4_5">
<label>4.5</label>
<title>Visual interpretation of the model</title>
<p>To provide a more intuitive analysis of the Mpox classification process, this study introduced the Gradient-weighted Class Activation Mapping (Grad-CAM) method (<xref ref-type="bibr" rid="B53">Woo et&#xa0;al., 2018</xref>). This method generates heatmaps by weighting the model&#x2019;s output with the gradients of a specified class, highlighting image regions that significantly influence classification decisions. In the heatmap, regions with higher weights are displayed in deeper red, emphasizing their greater impact on the model&#x2019;s category discrimination. In contrast, regions with lower weights appear in lighter blue, suggesting a milder influence of the image information in these areas on the classification recognition model. <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref> displays the heatmaps generated by Grad-CAM for each model in Mpox classification.</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Comparison of heat maps generated by Grad-CAM for each model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcimb-14-1397316-g010.tif"/>
</fig>
<p>Through the heatmaps, we can clearly observe the core regions that the models focus on. The heatmaps of the SqueezeNet and ResNet18 models exhibit similar features, concentrating on broader areas, with an issue of inaccurate focus on the infected regions. This suggests that they seem to show a considerable interest in widely distributed areas, which may explain their relatively poorer performance in classification. VGG16 and DenseNet121 demonstrate relatively good precision in locating Mpox infection regions. However, their focus areas seem slightly insufficient, indicating potential for further improvement. Swin-Tiny employs a strategy of layer-wise splitting to obtain multi-scale features. Efficient feature communication is achieved through cross-attention networks for these multi-scale features. However, due to its focus extending to the most extensive areas, there is an issue of imprecise attention to Mpox infection regions. Compared to other networks, ConvNeXt-Tiny exhibits a more precise ability to focus on Mpox infection regions, resulting in superior classification outcomes. In MpoxNet, the introduction of the D2RSE module and CBAM module further optimizes the model&#x2019;s attention to the Mpox region. Therefore, these improvements significantly enhance the classification performance.</p>
</sec>
</sec>
<sec id="s5" sec-type="discussion">
<label>5</label>
<title>Discussions</title>
<p>In <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>, despite slight variations in experimental parameters and training datasets compared to other methods, MpoxNet attains the highest recognition accuracy. Meanwhile, <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> illustrates that MpoxNet outperforms other networks with the lowest Flops and model parameter count, registering at 23.44G and 0.73M respectively. This allows the MpoxNet to not only serve as a real-time assessment tool but also be easily applied on smartphones for real-time identification and prediction of monkeypox cases. While MpoxNet demonstrates excellent performance in identifying monkeypox lesions, its precision in localizing widely scattered cases of monkeypox is somewhat lacking. In <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>, although the heatmap can encompass the entire infected area, it fails to accurately mark the positions of individual infection points, highlighting the need for further improvement in this aspect.</p>
</sec>
<sec id="s6" sec-type="conclusions">
<label>6</label>
<title>Conclusions</title>
<p>This paper proposes a high-precision lightweight classification network, MpoxNet, based on ConvNext, for the diagnostic classification of Mpox. Firstly, a dual-branch depth separable convolution residual Squeeze and Excitation (D2RSE) module is designed. Then, CBAM is introduced to improve diagnostic accuracy and significantly reduce the model&#x2019;s parameter count. Our proposed model is compared with SqueezeNet, ResNet18, ResNet34, ResNet50, VGG16, DenseNet121, Swin-Tiny, and ConvNext-Tiny. The proposed model achieves the highest accuracy, precision, recall, and F1-score among all tested models. In ablation experiments, the effectiveness of the D2RSE and CBAM modules is individually verified. By comparing MpoxNet with baseline networks from related papers using the BUSI dataset, the results clearly demonstrate the significant advantages of MpoxNet in terms of performance. This series of comparative experiments validates the crucial roles of the D2RSE and CBAM modules, providing solid evidence for the performance superiority of MpoxNet.</p>
<p>If the results of this study can be implemented, healthcare professionals may be able to improve patient prognosis and reduce medical costs. The research on deep learning and transfer learning techniques for automated Mpox detection not only provides innovative approaches for disease diagnosis but also opens new avenues for ad-dressing diagnostic challenges of other infectious diseases. Breakthroughs in this field will profoundly impact the medical domain, paving the way for the development of future diagnostic tools and methods.</p>
</sec>
<sec id="s7" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material. Further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s8" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The manuscript presents research on animals that do not require ethical approval for their study. Written informed consent was obtained from the minor(s)&#x2019; legal guardian/next of kin for the publication of any potentially identifiable images or data included in this article.</p>
</sec>
<sec id="s9" sec-type="author-contributions">
<title>Author contributions</title>
<p>JS: Software, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. BY: Methodology, Supervision, Writing &#x2013; review &amp; editing. ZS: Validation, Writing &#x2013; review &amp; editing. JZ: Visualization, Writing &#x2013; review &amp; editing. YD: Resources, Writing &#x2013; review &amp; editing. YG: Data curation, Writing &#x2013; review &amp; editing. YC: Data curation, Writing &#x2013; review &amp; editing.</p>
</sec>
</body>
<back>
<sec id="s10" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This research was funded in part by the Foundation of Shaanxi Key Laboratory of Integrated and Intelligent Navigation under Grant SKLIIN-20190102; in part by the Natural Science Foundation of Shaanxi Province under Grant 2021JM-537, 2019JQ-936, and 2021GY-341; and in part by the Research Foundation for Talented Scholars of Xijing University under Grant XJ20B01, XJ19B01, and XJ17B06.</p>
</sec>
<sec id="s11" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>Authors JS and BY were employed by the company The 20th Research Institute of China Electronics Technology Group Corporation.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s12" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abdelhamid</surname> <given-names>A. A.</given-names>
</name>
<name>
<surname>El-Kenawy</surname> <given-names>E.-S. M.</given-names>
</name>
<name>
<surname>Khodadadi</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Mirjalili</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Khafaga</surname> <given-names>D. S.</given-names>
</name>
<name>
<surname>Alharbi</surname> <given-names>A. H.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Classification of monkeypox images based on transfer learning and the al-biruni earth radius optimization algorithm</article-title>. <source>Mathematics</source> <volume>10</volume>, <elocation-id>3614</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/math10193614</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adalja</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Inglesby</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A novel international monkeypox outbreak</article-title>. <source>Ann. Intern. Med.</source> <volume>175</volume>, <fpage>1175</fpage>&#x2013;<lpage>1176</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.7326/M22&#x2013;1581</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahsan</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Uddin</surname> <given-names>M. R.</given-names>
</name>
<name>
<surname>Ali</surname> <given-names>M. S.</given-names>
</name>
<name>
<surname>Islam</surname> <given-names>M. K.</given-names>
</name>
<name>
<surname>Farjana</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Sakib</surname> <given-names>A. N.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Deep transfer learning approaches for Monkeypox disease diagnosis</article-title>. <source>Expert Syst. Appl.</source> <volume>216</volume>, <elocation-id>119483</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eswa.2022.119483</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahsan</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Uddin</surname> <given-names>M. R.</given-names>
</name>
<name>
<surname>Farjana</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Sakib</surname> <given-names>A. N.</given-names>
</name>
<name>
<surname>Momin</surname> <given-names>K. A.</given-names>
</name>
<name>
<surname>Luna</surname> <given-names>S. A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Image Data collection and implementation of deep learning-based model in detecting Monkeypox disease using modified VGG16</article-title>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2206.01862</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alakunle</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Moens</surname> <given-names>U.</given-names>
</name>
<name>
<surname>Nchinda</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Okeke</surname> <given-names>M. I.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Monkeypox virus in Nigeria: infection biology, epidemiology, and evolution</article-title>. <source>Viruses</source> <volume>12</volume>, <elocation-id>1257</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/v12111257</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Ali</surname> <given-names>S. N.</given-names>
</name>
<name>
<surname>Ahmed</surname> <given-names>M. T.</given-names>
</name>
<name>
<surname>Paul</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Jahan</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Sani</surname> <given-names>S. M. S.</given-names>
</name>
<name>
<surname>Noor</surname> <given-names>N.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>) <source>Monkeypox skin lesion detection using deep learning models: A feasibility study</source>. Available online at: <uri xlink:href="http://arxiv.org/abs/2207.03342">http://arxiv.org/abs/2207.03342</uri> (Accessed <access-date>October 18, 2023</access-date>).</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Almufareh</surname> <given-names>M. F.</given-names>
</name>
<name>
<surname>Tehsin</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Humayun</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Kausar</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A transfer learning approach for clinical detection support of monkeypox skin lesions</article-title>. <source>Diagnostics</source> <volume>13</volume>, <elocation-id>1503</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/diagnostics13081503</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altun</surname> <given-names>M.</given-names>
</name>
<name>
<surname>G&#xfc;r&#xfc;ler</surname> <given-names>H.</given-names>
</name>
<name>
<surname>&#xd6;zkaraca</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Monkeypox detection using CNN with transfer learning</article-title>. <source>Sensors</source> <volume>23</volume>, <elocation-id>1783</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s23041783</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alwakid</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Gouda</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Humayun</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Sama</surname> <given-names>N. U.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Melanoma detection using deep learning-based classifications</article-title>. <source>Healthcare</source> <volume>10</volume>, <elocation-id>2481</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/healthcare10122481</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bala</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Hossain</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Hossain</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Abdullah</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Rahman</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Manavalan</surname> <given-names>B.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>MonkeyNet: A robust deep convolutional neural network for monkeypox disease detection and classification</article-title>. <source>Neural Networks</source> <volume>161</volume>, <fpage>757</fpage>&#x2013;<lpage>775</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.neunet.2023.02.022</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Banerjee</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Robinson</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Sathian</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Global re-emergence of human monkeypox: Population on high alert</article-title>. <source>Nepal J. Epidemiol.</source> <volume>12</volume>, <fpage>1179</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3126/nje.v12i2.45974</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Bhosale</surname> <given-names>Y. H.</given-names>
</name>
<name>
<surname>Zanwar</surname> <given-names>S. R.</given-names>
</name>
<name>
<surname>Jadhav</surname> <given-names>A. T.</given-names>
</name>
<name>
<surname>Ahmed</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Gaikwad</surname> <given-names>V. S.</given-names>
</name>
<name>
<surname>Gandle</surname> <given-names>K. S.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Human monkeypox 2022 virus: machine learning prediction model, outbreak forecasting, visualization with time-series exploratory data analysis</article-title>,&#x201d; in <conf-name>2022 13th International Conference on Computing Communication and Networking Technologies (ICCCNT)</conf-name>. (<publisher-name>IEEE</publisher-name>, <publisher-loc>Kharagpur, India</publisher-loc>). <fpage>1</fpage>&#x2013;<lpage>6</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICCCNT54827.2022.9984237</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fraiwan</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Faouri</surname> <given-names>E.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>On the automatic detection and classification of skin cancer using deep transfer learning</article-title>. <source>Sensors</source> <volume>22</volume>, <elocation-id>4963</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s22134963</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Glock</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Napier</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Gary</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Gupta</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Gigante</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Schaffner</surname> <given-names>W.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). &#x201c;<article-title>Measles rash identification using transfer learning and deep convolutional neural networks</article-title>,&#x201d; in <conf-name>2021 IEEE International Conference on Big Data (Big Data)</conf-name>. (<publisher-name>IEEE</publisher-name>, <publisher-loc>Orlando, FL, USA</publisher-loc>). <fpage>3905</fpage>&#x2013;<lpage>3910</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/BigData52589.2021.9671333</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gong</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Chuai</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Chiu</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Monkeypox virus: a re-emergent threat to humans</article-title>. <source>Virologica Sin.</source> <volume>37</volume>, <fpage>477</fpage>&#x2013;<lpage>482</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.virs.2022.07.006</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>G&#xfc;rb&#xfc;z</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Aydin</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Monkeypox skin lesion detection using deep learning models</article-title>,&#x201d; in <conf-name>2022 International Conference on Computers and Artificial Intelligence Technologies (CAIT)</conf-name>. (<publisher-name>IEEE</publisher-name>, <publisher-loc>Quzhou, China</publisher-loc>). <fpage>66</fpage>&#x2013;<lpage>70</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CAIT56099.2022.10072140</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haque</surname> <given-names>M. E.</given-names>
</name>
<name>
<surname>Ahmed</surname> <given-names>M. R.</given-names>
</name>
<name>
<surname>Nila</surname> <given-names>R. S.</given-names>
</name>
<name>
<surname>Islam</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Classification of human monkeypox disease using deep learning models and attention mechanisms</article-title>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2211.15459</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Deep residual learning for image recognition</article-title>. <fpage>9</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1512.03385</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Huang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>van der Maaten</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Weinberger</surname> <given-names>K. Q.</given-names>
</name>
</person-group> (<year>2018</year>) <source>Densely connected convolutional networks</source>. Available online at: <uri xlink:href="http://arxiv.org/abs/1608.06993">http://arxiv.org/abs/1608.06993</uri> (Accessed <access-date>January 22, 2024</access-date>).</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Iandola</surname> <given-names>F. N.</given-names>
</name>
<name>
<surname>Han</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Moskewicz</surname> <given-names>M. W.</given-names>
</name>
<name>
<surname>Ashraf</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Dally</surname> <given-names>W. J.</given-names>
</name>
<name>
<surname>Keutzer</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>SqueezeNet: AlexNet-level accuracy with 50x fewer parameters and &lt;0.5MB model size</article-title>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1602.07360</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ibrahim</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Mirjalili</surname> <given-names>S.</given-names>
</name>
<name>
<surname>El-Said</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Ghoneim</surname> <given-names>S. S. M.</given-names>
</name>
<name>
<surname>Al-Harthi</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Ibrahim</surname> <given-names>T. F.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Wind speed ensemble forecasting based on deep learning using adaptive dynamic optimization algorithm</article-title>. <source>IEEE Access</source> <volume>9</volume>, <fpage>125787</fpage>&#x2013;<lpage>125804</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2021.3111408</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Iftikhar</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>M. S.</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Short-term forecasting of monkeypox cases using a novel filtering and combining technique</article-title>. <source>Diagnostics</source> <volume>13</volume>, <elocation-id>1923</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/diagnostics13111923</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Islam</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Hussain</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Chowdhury</surname> <given-names>F. U. H.</given-names>
</name>
<name>
<surname>Islam</surname> <given-names>B. M. R.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Can artificial intelligence detect monkeypox from digital skin images</article-title>? <fpage>7</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2022.08.08.503193</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jaradat</surname> <given-names>A. S.</given-names>
</name>
<name>
<surname>Al Mamlook</surname> <given-names>R. E.</given-names>
</name>
<name>
<surname>Almakayeel</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Alharbe</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Almuflih</surname> <given-names>A. S.</given-names>
</name>
<name>
<surname>Nasayreh</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Automated monkeypox skin lesion detection using deep learning and transfer learning techniques</article-title>. <source>Int. J. Environ. Res. Public Health</source> <volume>20</volume>, <elocation-id>4422</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/ijerph20054422</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Javelle</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Ficko</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Savini</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Mura</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Ferraris</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Tournier</surname> <given-names>J. N.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Monkeypox clinical disease: Literature review and a tool proposal for the monitoring of cases and contacts</article-title>. <source>Travel Med. Infect. Dis.</source> <volume>52</volume>, <elocation-id>102559</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.tmaid.2023.102559</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>JujubeNet: A high-precision lightweight jujube surface defect classification network with an attention mechanism</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>. doi: <pub-id pub-id-type="doi">10.3389/fpls.2022.1108437</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khafaga</surname> <given-names>D. S.</given-names>
</name>
<name>
<surname>Alhussan</surname> <given-names>A. A.</given-names>
</name>
<name>
<surname>El-Kenawy</surname> <given-names>E.-S. M.</given-names>
</name>
<name>
<surname>Ibrahim</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Eid</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Abdelhamid</surname> <given-names>A. A.</given-names>
</name>
</person-group> (<year>2022</year>a). <article-title>Solving optimization problems of metamaterial and double&#xa0;T-shape antennas using advanced meta-heuristics algorithms</article-title>. <source>IEEE Access</source> <volume>10</volume>, <fpage>74449</fpage>&#x2013;<lpage>74471</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2022.3190508</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khafaga</surname> <given-names>D. S.</given-names>
</name>
<name>
<surname>Ibrahim</surname> <given-names>A.</given-names>
</name>
<name>
<surname>El-Kenawy</surname> <given-names>E.-S. M.</given-names>
</name>
<name>
<surname>Abdelhamid</surname> <given-names>A. A.</given-names>
</name>
<name>
<surname>Karim</surname> <given-names>F. K.</given-names>
</name>
<name>
<surname>Mirjalili</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>b). <article-title>An al-biruni earth radius optimization-based deep convolutional neural network for classifying monkeypox disease</article-title>. <source>Diagnostics</source> <volume>12</volume>, <elocation-id>2892</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/diagnostics12112892</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname> <given-names>V.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Analysis of CNN features with multiple machine learning classifiers in diagnosis of monkeypox from digital skin images</article-title>. <fpage>4</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2022.09.11.22278797</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ladnyj</surname> <given-names>I. D.</given-names>
</name>
<name>
<surname>Ziegler</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Kima</surname> <given-names>E.</given-names>
</name>
</person-group> (<year>1972</year>). <article-title>A human infection caused by monkeypox virus in Basankusu Territory, Democratic Republic of the Congo</article-title>. <source>Bull. World Health Organ</source> <volume>46</volume>, <fpage>593</fpage>&#x2013;<lpage>597</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>) <source>Swin transformer: hierarchical vision transformer using shifted windows</source>. Available online at: <uri xlink:href="http://arxiv.org/abs/2103.14030">http://arxiv.org/abs/2103.14030</uri> (Accessed <access-date>January 22, 2024</access-date>).</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Mao</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>C.-Y.</given-names>
</name>
<name>
<surname>Feichtenhofer</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Darrell</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>A convNet for the 2020s</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>. (<publisher-loc>New Orleans, LA, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>11976</fpage>&#x2013;<lpage>11986</lpage>. doi: <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.01167</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mandal</surname> <given-names>A. K.</given-names>
</name>
<name>
<surname>Sarma</surname> <given-names>P. K. D.</given-names>
</name>
<name>
<surname>Dehuri</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Machine Learning Approaches and Particle Swarm Optimization Based Clustering for the Human Monkeypox Viruses: A Study</article-title>,&#x201d; in <source>Innovations in Intelligent Computing and Communication</source>. Eds. <person-group person-group-type="editor">
<name>
<surname>Panda</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Dehuri</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Patra</surname> <given-names>M. R.</given-names>
</name>
<name>
<surname>Behera</surname> <given-names>P. K.</given-names>
</name>
<name>
<surname>Tsihrintzis</surname> <given-names>G. A.</given-names>
</name>
<name>
<surname>Cho</surname> <given-names>S.-B.</given-names>
</name>
<etal/>
</person-group> (<publisher-name>Springer International Publishing</publisher-name>, <publisher-loc>Cham</publisher-loc>), <fpage>313</fpage>&#x2013;<lpage>332</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978&#x2013;3-031&#x2013;23233-6_24</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McCollum</surname> <given-names>A. M.</given-names>
</name>
<name>
<surname>Damon</surname> <given-names>I. K.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Human monkeypox</article-title>. <source>Clin. Infect. Dis.</source> <volume>58</volume>, <fpage>260</fpage>&#x2013;<lpage>267</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/cid/cit703</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meena</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Mohbey</surname> <given-names>K. K.</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Monkeypox recognition and prediction from visuals using deep transfer learning-based neural networks</article-title>. <source>Multimed Tools Appl.</source>, <fpage>1</fpage>&#x2013;<lpage>25</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11042&#x2013;024-18437-z</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meena</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Mohbey</surname> <given-names>K. K.</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Lokesh</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A hybrid deep learning approach for detecting sentiment polarities and knowledge graph representation on monkeypox tweets</article-title>. <source>Decision Anal. J.</source> <volume>7</volume>, <elocation-id>100243</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.dajour.2023.100243</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mehrotra</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Ansari</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Agrawal</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Anand</surname> <given-names>R. S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A Transfer Learning approach for AI-based classification of brain tumors</article-title>. <source>Mach. Learn. Appl.</source> <volume>2</volume>, <elocation-id>100003</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.mlwa.2020.100003</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mohbey</surname> <given-names>K. K.</given-names>
</name>
<name>
<surname>Meena</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Lokesh</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A CNN-LSTM-based hybrid deep learning approach to detect sentiment polarities on Monkeypox tweets</article-title>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2208.12019</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="web">
<person-group person-group-type="author">
<collab>Monkeypox Skin Images Dataset (MSID)</collab>
</person-group>. (<year>2023</year>). Available at: <uri xlink:href="https://www.kaggle.com/datasets/dipuiucse/monkeypoxskinimagedataset">https://www.kaggle.com/datasets/dipuiucse/monkeypoxskinimagedataset</uri> (Accessed <access-date>November 10, 2023</access-date>).</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reed</surname> <given-names>K. D.</given-names>
</name>
<name>
<surname>Melski</surname> <given-names>J. W.</given-names>
</name>
<name>
<surname>Graham</surname> <given-names>M. B.</given-names>
</name>
<name>
<surname>Regnery</surname> <given-names>R. L.</given-names>
</name>
<name>
<surname>Sotir</surname> <given-names>M. J.</given-names>
</name>
<name>
<surname>Wegner</surname> <given-names>M. V.</given-names>
</name>
<etal/>
</person-group>. (<year>2004</year>). <article-title>The detection of monkeypox in humans in the western hemisphere</article-title>. <source>N.&#xa0;Engl. J. Med.</source> <volume>350</volume>, <fpage>342</fpage>&#x2013;<lpage>350</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1056/NEJMoa032299</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reynolds</surname> <given-names>M. G.</given-names>
</name>
<name>
<surname>McCollum</surname> <given-names>A. M.</given-names>
</name>
<name>
<surname>Nguete</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Shongo Lushima</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Petersen</surname> <given-names>B. W.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Improving the care and treatment of monkeypox patients in low-resource settings: applying evidence from contemporary biomedical and smallpox biodefense research</article-title>. <source>Viruses</source> <volume>9</volume>, <elocation-id>380</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/v9120380</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rogers</surname> <given-names>J. V.</given-names>
</name>
<name>
<surname>Parkinson</surname> <given-names>C. V.</given-names>
</name>
<name>
<surname>Choi</surname> <given-names>Y. W.</given-names>
</name>
<name>
<surname>Speshock</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Hussain</surname> <given-names>S. M.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>A preliminary assessment of silver nanoparticle inhibition of monkeypox virus plaque formation</article-title>. <source>Nanoscale Res. Lett.</source> <volume>3</volume>, <fpage>129</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11671&#x2013;008-9128&#x2013;2</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sahin</surname> <given-names>V. H.</given-names>
</name>
<name>
<surname>Oztel</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Yolcu Oztel</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Human monkeypox classification from skin lesion images with deep pre-trained network using mobile application</article-title>. <source>J.&#xa0;Med. Syst.</source> <volume>46</volume>, <fpage>79</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10916&#x2013;022-01863&#x2013;7</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saleh</surname> <given-names>A. I.</given-names>
</name>
<name>
<surname>Rabie</surname> <given-names>A. H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Human monkeypox diagnose (HMD) strategy based on data mining and artificial intelligence techniques</article-title>. <source>Comput. Biol. Med.</source> <volume>152</volume>, <elocation-id>106383</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.106383</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sandeep</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Vishal</surname> <given-names>K. P.</given-names>
</name>
<name>
<surname>Shamanth</surname> <given-names>M. S.</given-names>
</name>
<name>
<surname>Chethan</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Diagnosis of visible diseases using CNNs</article-title>,&#x201d; in <source>Proceedings of International Conference on Communication and Artificial Intelligence</source>. Eds. <person-group person-group-type="editor">
<name>
<surname>Goyal</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Gupta</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Mirjalili</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Trivedi</surname> <given-names>A.</given-names>
</name>
</person-group> (<publisher-name>Springer Nature</publisher-name>, <publisher-loc>Singapore</publisher-loc>), <fpage>459</fpage>&#x2013;<lpage>468</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978&#x2013;981-19&#x2013;0976-4_38</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sharif</surname> <given-names>S. M. A.</given-names>
</name>
<name>
<surname>Naqvi</surname> <given-names>R. A.</given-names>
</name>
<name>
<surname>Biswas</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Loh</surname> <given-names>W.-K.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Deep perceptual enhancement for medical image analysis</article-title>. <source>IEEE J. Biomed. Health Inf.</source> <volume>26</volume>, <fpage>4826</fpage>&#x2013;<lpage>4836</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/JBHI.2022.3168604</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Simonyan</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zisserman</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Very deep convolutional networks for large-scale image recognition</article-title>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1409.1556</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Simpson</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Heymann</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Brown</surname> <given-names>C. S.</given-names>
</name>
<name>
<surname>Edmunds</surname> <given-names>W. J.</given-names>
</name>
<name>
<surname>Elsgaard</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Fine</surname> <given-names>P.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Human monkeypox &#x2013; After 40 years, an unintended consequence of smallpox eradication</article-title>. <source>Vaccine</source> <volume>38</volume>, <fpage>5077</fpage>&#x2013;<lpage>5081</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.vaccine.2020.04.062</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sitaula</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Hossain</surname> <given-names>M. B.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Attention-based VGG-16 model for COVID-19 chest X-ray image classification</article-title>. <source>Appl. Intell.</source> <volume>51</volume>, <fpage>2850</fpage>&#x2013;<lpage>2863</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10489&#x2013;020-02055-x</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sitaula</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Shahi</surname> <given-names>T. B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Monkeypox virus detection using pre-trained deep learning-based approaches</article-title>. <source>J. Med. Syst.</source> <volume>46</volume>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10916&#x2013;022-01868&#x2013;2</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uzun Ozsahin</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Mustapha</surname> <given-names>M. T.</given-names>
</name>
<name>
<surname>Uzun</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Duwa</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Ozsahin</surname> <given-names>I.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Computer-aided detection and classification of monkeypox and chickenpox lesion in human subjects using deep learning framework</article-title>. <source>Diagnostics</source> <volume>13</volume>, <elocation-id>292</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/diagnostics13020292</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Velasco</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Pascion</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Alberio</surname> <given-names>J. W.</given-names>
</name>
<name>
<surname>Apuang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Cruz</surname> <given-names>J. S.</given-names>
</name>
<name>
<surname>Gomez</surname> <given-names>M. A.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>A smartphone-based skin disease classification using mobileNet CNN</article-title>. <source>IJATCSE</source>, <fpage>2632</fpage>&#x2013;<lpage>2637</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.30534/ijatcse/2019/116852019</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Woo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Park</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>J.-Y.</given-names>
</name>
<name>
<surname>Kweon</surname> <given-names>I. S.</given-names>
</name>
</person-group> (<year>2018</year>) <source>CBAM: convolutional block attention module</source>. Available online at: <uri xlink:href="https://openaccess.thecvf.com/content_ECCV_2018/html/Sanghyun_Woo_Convolutional_Block_Attention_ECCV_2018_paper.html">https://openaccess.thecvf.com/content_ECCV_2018/html/Sanghyun_Woo_Convolutional_Block_Attention_ECCV_2018_paper.html</uri> (Accessed <access-date>July 7, 2023</access-date>).</citation>
</ref>
<ref id="B54">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Xie</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Dollar</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Tu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2017</year>) <source>Aggregated residual transformations for deep neural networks</source>. Available online at: <uri xlink:href="https://openaccess.thecvf.com/content_cvpr_2017/html/Xie_Aggregated_Residual_Transformations_CVPR_2017_paper.html">https://openaccess.thecvf.com/content_cvpr_2017/html/Xie_Aggregated_Residual_Transformations_CVPR_2017_paper.html</uri> (Accessed <access-date>July 17, 2023</access-date>).</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yasmin</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Hassan</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Zaman</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Aung</surname> <given-names>S. T.</given-names>
</name>
<name>
<surname>Karim</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Azam</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A forecasting prognosis of the monkeypox outbreak based on a comprehensive statistical and regression analysis</article-title>. <source>Computation</source> <volume>10</volume>, <elocation-id>177</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/computation10100177</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>