<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1608837</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Comparative analysis of multimodal architectures for effective skin lesion detection using clinical and image data</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Das</surname> <given-names>Adriteyo</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3031208/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Agarwal</surname> <given-names>Vedant</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Shetty</surname> <given-names>Nisha P.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3025902/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Information and Communication Technology, Manipal Institute of Technology, Manipal Academy of Higher Education</institution>, <addr-line>Manipal</addr-line>, <country>India</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Humanities and Management, Manipal Institute of Technology, Manipal Academy of Higher Education</institution>, <addr-line>Manipal</addr-line>, <country>India</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Herwig Unger, University of Hagen, Germany</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Massimo Salvi, Polytechnic University of Turin, Italy</p>
<p>Jay Lofstead, Sandia National Laboratories (DOE), United States</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Nisha P. Shetty <email>nisha.pshetty&#x00040;manipal.edu</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>18</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1608837</elocation-id>
<history>
<date date-type="received">
<day>09</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Das, Agarwal and Shetty.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Das, Agarwal and Shetty</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Background/Introduction</title>
<p>Skin lesion classification poses a critical diagnostic challenge in dermatology, where early and accurate identification has a direct impact on patient outcomes. While deep learning approaches have shown promise using dermatoscopic images alone, the integration of clinical metadata remains underexplored despite its potential to enhance diagnostic accuracy.</p></sec>
<sec>
<title>Methods</title>
<p>We developed a novel multimodal data fusion framework that systematically integrates dermatoscopic images with clinical metadata for the classification of skin lesions. Using the HAM10000 dataset, we evaluated multiple fusion strategies, including simple concatenation, weighted concatenation, self-attention mechanisms, and cross-attention fusion. Clinical features were processed through a customized Multi-Layer Perceptron (MLP), while images were analyzed using a modified Residual Networks (ResNet) architecture. Model interpretability was enhanced using Gradient-weighted Class Activation Mapping (Grad-CAM) visualization to identify the contribution of clinical attributes to classification decisions.</p></sec>
<sec>
<title>Results</title>
<p>Cross-attention fusion achieved the highest classification accuracy, demonstrating superior performance compared to unimodal approaches and simpler fusion techniques. The multimodal framework significantly outperformed image-only baselines, with cross-attention effectively capturing inter-modal dependencies and contextual relationships between visual and clinical data modalities.</p></sec>
<sec>
<title>Discussion/Conclusions</title>
<p>Our findings demonstrate that integrating clinical metadata with dermatoscopic images substantially improves the accuracy of skin lesion classification. However, challenges, including class imbalance and the computational complexity of advanced fusion methods, require further investigation.</p></sec></abstract>
<kwd-group>
<kwd>skin lesion classification</kwd>
<kwd>multimodal fusion</kwd>
<kwd>dermatoscopic images</kwd>
<kwd>clinical metadata</kwd>
<kwd>cross-attention</kwd>
<kwd>HAM10000</kwd>
<kwd>interpretability</kwd>
<kwd>deep learning</kwd>
</kwd-group>
<counts>
<fig-count count="10"/>
<table-count count="8"/>
<equation-count count="19"/>
<ref-count count="64"/>
<page-count count="19"/>
<word-count count="11344"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Machine Learning and Artificial Intelligence</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Skin cancer ranks as the 5th most widespread cancer type. It is among the most severe variants and is projected to overtake cardiovascular disease as the primary cause of death in humans in the near future (<xref ref-type="bibr" rid="B19">Hasan et al., 2023</xref>). From 1990 to 2017, the incidence of individuals diagnosed with malignant skin melanoma (MEL), squamous cell carcinoma (SCC), and basal cell carcinoma (BCC) surged by 215.7%, 196.8%, and 90.9%, respectively (<xref ref-type="bibr" rid="B25">Kavita et al., 2023</xref>). The predominant forms of skin cancer include MEL, BCC, and SCC, alongside precancerous conditions such as actinic keratosis (AK) (<xref ref-type="bibr" rid="B47">Rogers et al., 2015</xref>). As a pressing global health issue, skin cancer highlights the urgent necessity for early detection to enhance patient prognosis and decrease fatality rates. Timely identification facilitates less aggressive interventions and reduces medical expenses by catching cancers at manageable phases (<xref ref-type="bibr" rid="B23">Jerant et al., 2000</xref>). Existing diagnostic approaches, like skin self-examination (SSE) and clinical skin examination (CSE), depend significantly on visual assessment and tools like the Asymmetry, Border, Color, Diameter, Evolving (ABCDE) criteria. While advanced methods such as dermoscopy and total body photography (TBP) boost precision, conventional techniques are often subjective, labor-intensive, and unavailable in under-resourced regions (<xref ref-type="bibr" rid="B32">Loescher et al., 2013</xref>; <xref ref-type="bibr" rid="B44">Rajput et al., 2021</xref>; <xref ref-type="bibr" rid="B11">Esteva et al., 2017</xref>).</p>
<p>Computerized systems, including Computer-aided Diagnosis (CAD) software and image processing algorithms, offer reproducible, objective, and quick evaluation of skin lesions, less dependent on subjective human judgment. Such systems can process large volumes of data with efficiency, allowing early detection of subtle patterns that would be easily overlooked by conventional techniques (<xref ref-type="bibr" rid="B17">Han et al., 2018</xref>). Automation also enhances accessibility by being incorporated into telemedicine platforms, extending diagnostic abilities to distant and under-served communities. By minimizing the necessity for invasive biopsies and follow-up visits, automation decreases healthcare expenditure without compromising diagnostic efficiency and patient outcome (<xref ref-type="bibr" rid="B58">Tschandl et al., 2019</xref>).</p>
<p>Machine learning (ML) and deep learning (DL) have made tremendous progress over the last few years, fueled by more computational power and large datasets. These technologies have shown remarkable success in a range of medical classification problems, including eye movement-based disease prediction with Decision Trees and Random Forests, automated skin disease classification with k-Nearest Neighbors(KNN) and Support Vector Machines(SVM) with 98.22% accuracy, and brain tumor classification with Convolutional Neural Networks (CNNs) from the Visual Geometry Group(VGG) like VGG16 and VGG19 with accuracies ranging from 92.5% to 97.8% (<xref ref-type="bibr" rid="B3">Ahsan et al., 2022</xref>).</p>
<p>(<xref ref-type="bibr" rid="B16">Haenssle et al. 2018</xref>) tested the diagnostic accuracy of a homegrown Convolutional Neural Network (CNN) constructed on top of Google&#x00027;s Inception v4 model, trained on data from partner dermatologists and the International Skin Imaging Collaboration (ISIC) dermoscopic archive, with a heterogeneous panel of 58 dermatologists from across the globe. The findings indicated that the CNN performed better than most dermatologists with a mean AUC-ROC of 0.86 against 0.79 (<italic>p</italic> &#x0003C; 0.01). This work highlights the possibility of dermatologists adopting CNNs in their practice, thus enhancing the accuracy of diagnosis and ultimately improving patient outcomes such as better prognosis, treatment options, and general well-being.</p>
<p>Multimodal data represents information derived from diverse sources and formats, including images, text, audio, and physiological signals. This integrated approach mirrors human cognitive processes, where multiple sensory modalities contribute to perception and interpretation. In medical contexts, multimodal data combines Medical images, Patient records (demographics, medical history, lab results), Physiological signals, and Patient-reported outcomes.</p>
<p>Recent studies have demonstrated the efficacy of multimodal approaches in medical diagnosis. For example, (<xref ref-type="bibr" rid="B1">Adarsh et al. 2024</xref>) achieved 98.27% accuracy using a Multi-feature Kernel Supervised within-class-similar Discriminative Dictionary Learning (MKSCDDL) algorithm for Alzheimer&#x00027;s Disease classification, (<xref ref-type="bibr" rid="B24">Jiang et al. 2022</xref>) employed multimodal ultrasound data with a CNN, achieving 98.22% accuracy in early breast cancer detection, and (<xref ref-type="bibr" rid="B29">Kumar et al. 2022</xref>) utilized audio and X-ray imaging with a CNN and Deep Uniform Net, obtaining 98.67% accuracy in COVID-19 classification.</p>
<p>In this study, we analyze fusion techniques for skin cancer classification, leveraging multimodal skin images and clinical data. Our research explores how fusion methods can enhance skin lesion classification performance compared to single-modality approaches. We investigate various fusion strategies to integrate diverse data sources and improve model accuracy.</p>
<p>To enhance the interpretability of our multimodal systems, we apply the Gradient-Weighted Class Activation Mapping (Grad-CAM) approach for deep learning explainability and feature relevance assessment. Our contributions aim to advance skin lesion classification by presenting robust fusion strategies that offer high accuracy and clinical interpretability.</p></sec>
<sec id="s2">
<title>2 Related work</title>
<p>The automated detection and classification of skin lesions, especially for the diagnosis of skin cancer, has been an essential area of research in Applied Artificial Intelligence (AI). Current research has investigated various methodologies, from texture-based feature extraction to multimodal deep learning, with the objective of improving the accuracy, efficiency, and explainability of computer-aided diagnosis systems.</p>
<p>In the research conducted by (<xref ref-type="bibr" rid="B6">Arshad et al. 2021</xref>), the skin images were subjected to augmentation, after which important features were extracted using fine-tuned ResNet-50 and ResNet-101. A serial-based fusion approach fused the features, and the selected best features were classified using supervised learning algorithms. As a future scope, the authors propose improvements in feature selection, extraction, and parallel feature fusion.</p>
<p>(<xref ref-type="bibr" rid="B53">Sevli 2021</xref>) employed a customized CNN to classify 7 types of skin lesions in the HAM10000 dataset. They developed a web application to deploy the model and validated its performance with seven dermatologists. The analysis was twofold: evaluating the classification performance of the model with expert feedback and vice versa. The authors attributed the model&#x00027;s success to improvements in training set size and proposed the use of real lesion images to enhance generalizability.</p>
<p>(<xref ref-type="bibr" rid="B61">Wang et al. 2023</xref>) introduced a two-stream neural network architecture for feature extraction. The extracted features were fused using a feature fusion module with a multireceptive field and Generalized Mean Pooling (GeM). As future work, the authors proposed incorporating different imaging modalities and other clinical diagnostic data to enhance the study&#x00027;s scope.</p>
<p>(<xref ref-type="bibr" rid="B2">Adebiyi et al. 2024</xref>) employed the Align Before Fuse (ABEF) framework, which combined image features extracted by a Vision Transformer and text features extracted by Bidirectional Encoder Representations from Transformers(BERT). These features were jointly encoded using a text-image encoder for classification.</p>
<p>(<xref ref-type="bibr" rid="B55">Srivastava et al. 2022</xref>) proposed a median-based quadrant texture feature extraction module, which was combined with a modified CNN architecture for classification. Their advanced texture extraction method outperformed existing models due to its superior noise-handling capabilities.</p>
<p>(<xref ref-type="bibr" rid="B9">Datta et al. 2021</xref>) compared the performance of various transfer learning architectures with and without a soft attention mechanism for skin cancer classification. The soft attention module effectively localized cancerous regions, thereby enhancing model accuracy and interpretability.</p>
<p>(<xref ref-type="bibr" rid="B30">Lan et al. 2022</xref>) enhanced the capsule network architecture by integrating a large-kernel convolution (31 &#x000D7; 31) and a Convolutional Block Attention Module (CBAM). They further included group convolution to reduce parameter overhead and avoid underfitting. The capsule layer was redesigned to improve feature extraction and runtime efficiency. A lightweight variant, FixCaps-DS, was introduced using depthwise separable convolutions to maintain performance while reducing complexity.</p>
<p>(<xref ref-type="bibr" rid="B15">Gessert et al. 2020</xref>) designed a multimodal deep learning system that handled two tasks from the International Skin Imaging Collaboration (ISIC) 2019 challenge: image-only classification and image&#x0002B;metadata classification. For task 1, they used an ensemble of EfficientNet variants and other CNNs for architectural diversity. For task 2, metadata (e.g., age, sex, anatomical site) was processed with a two-layer dense neural network. Features from both modalities were concatenated and passed through additional dense layers before classification.</p>
<p>(<xref ref-type="bibr" rid="B41">Ou et al. 2022</xref>) employed a deep neural network that used inter-modality cross-attention and intra-modality self-attention to classify skin lesions. ResNet-50 was used to extract image features, and a Multi-Layer Perceptron (MLP) encoded the clinical metadata. After applying the attention mechanisms, features were concatenated and passed through a fully connected softmax classifier.</p>
<p>(<xref ref-type="bibr" rid="B46">Restrepo et al. 2024</xref>) evaluated vector embedding-based multimodal fusion methods for low-resource settings, comparing them with traditional raw data processing. They tested unimodal embeddings (DINO v2 for images, LLAMA 2 for text), Vision-Language Models (VLM) like CLIP, and fine-tuned transformers (BERT, ViT) using early and late fusion strategies. A novel alignment method was also introduced to reduce the &#x0201C;cone effect&#x0201D; in embedding space. While promising results were achieved on benchmark datasets like BRSET and HAM10000, domain-specific limitations for dermatology were acknowledged.</p>
<p>To summarize, despite notable progress having been made on both single-modality and multimodal skin cancer classification, some of the shortcomings still remain. Numerous current models are computationally intensive, rendering them impractical to use in real-world applications, particularly where there are constraints on resources. These models also have difficulty generalizing across varied datasets, and their performance becomes suboptimal when they have to handle other unknown classes or metadata. Additionally, while multimodal systems have shown improved accuracy, they still suffer from limitations like overfitting and inadequate training and inference efficiency.</p>
<p>Our method is designed to address these limitations by building on fusion methods that integrate skin images and clinical data, improving classification performance while being efficient. We focus on minimizing computational expenses and storage needs, rendering the system more implementable in resource-constrained environments. In addition, to enhance interpretability, we incorporate the Grad-CAM method so that feature relevance can be better evaluated and the system can be made more clinically usable. Finally, our method aims to deliver a strong, efficient, and interpretable solution for skin lesion classification in real-world, resource-limited settings. <xref ref-type="table" rid="T1">Table 1</xref> consolidates the above papers.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Comparative summary of skin lesion classification approaches.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>References</bold></th>
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="left"><bold>Classifiers/techniques</bold></th>
<th valign="top" align="left"><bold>Performance</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B6">Arshad et al. 2021</xref>)</td>
<td valign="top" align="left">HAM10000</td>
<td valign="top" align="left">ResNet-50, ResNet-101; serial fusion; SVR-based feature selection; supervised ML classifiers</td>
<td valign="top" align="left">95% on fused features (augmented); 91.7% after selection</td>
</tr> <tr>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B53">Sevli 2021</xref>)</td>
<td valign="top" align="left">HAM10000</td>
<td valign="top" align="left">Modified CNN with contrast enhancement and expert validation</td>
<td valign="top" align="left">90.28% accuracy; corrected 11.14% of dermatologist misdiagnoses</td>
</tr> <tr>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B61">Wang et al. 2023</xref>)</td>
<td valign="top" align="left">ISIC 2018</td>
<td valign="top" align="left">Two-stream CNN (DenseNet-121 &#x0002B; improved VGG-16); multi-receptive field module &#x0002B; GeM pooling</td>
<td valign="top" align="left">91.24% accuracy</td>
</tr> <tr>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B2">Adebiyi et al. 2024</xref>)</td>
<td valign="top" align="left">HAM10000</td>
<td valign="top" align="left">Multimodal ALBEF (Vision Transformer &#x0002B; BERT)</td>
<td valign="top" align="left">94.11% accuracy; AUC-ROC 0.9426</td>
</tr> <tr>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B55">Srivastava et al. 2022</xref>)</td>
<td valign="top" align="left">HAM10000, ISIC-UDA11</td>
<td valign="top" align="left">M-QuadLTQP texture encoding &#x0002B; CNN</td>
<td valign="top" align="left">96% average accuracy</td>
</tr> <tr>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B9">Datta et al. 2021</xref>)</td>
<td valign="top" align="left">HAM10000, ISIC-2017</td>
<td valign="top" align="left">Soft attention integrated with IRv2, ResNet, Inception; Grad-CAM visualizations</td>
<td valign="top" align="left">93.4% precision (HAM); 91.6% sensitivity (ISIC-2017)</td>
</tr> <tr>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B30">Lan et al. 2022</xref>)</td>
<td valign="top" align="left">HAM10000</td>
<td valign="top" align="left">FixCaps with CBAM &#x0002B; large-kernel conv; FixCaps-DS with depthwise conv</td>
<td valign="top" align="left">96.49% (FixCaps); 96.13% (lightweight)</td>
</tr> <tr>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B15">Gessert et al. 2020</xref>)</td>
<td valign="top" align="left">ISIC 2019 Challenge</td>
<td valign="top" align="left">Ensemble of CNNs (EfficientNets, ResNeXt, SENet); metadata fusion &#x0002B; heavy augmentation</td>
<td valign="top" align="left">Balanced Acc: 74.2%; Sensitivity: 63%</td>
</tr> <tr>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B41">Ou et al. 2022</xref>)</td>
<td valign="top" align="left">PAD-UPES-20</td>
<td valign="top" align="left">ResNet-50 &#x0002B; MLP; MMF-Net with intra/inter attention fusion</td>
<td valign="top" align="left">76.8% accuracy; BACC: 77.5%; AUC: 94.7%</td>
</tr> <tr>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B46">Restrepo et al. 2024</xref>)</td>
<td valign="top" align="left">HAM10000, BRSET</td>
<td valign="top" align="left">Embedding fusion using CLIP, DINOv2 &#x0002B; LLAMA2; early/late fusion</td>
<td valign="top" align="left">81.8% accuracy (HAM10000); 98.7% (BRSET)</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec sec-type="materials and methods" id="s3">
<title>3 Materials and methods</title>
<sec>
<title>3.1 Dataset</title>
<p>The dataset we have used for our classification problem is the <bold>HAM10000 dataset</bold> (<xref ref-type="bibr" rid="B59">Tschandl et al., 2018</xref>; <xref ref-type="bibr" rid="B49">Scott Mader, 2018</xref>). HAM10000, or &#x0201C;Human Against Machine,&#x0201D; is a curated dataset of multi-source dermatoscopic images of pigmented skin lesions. The final dataset comprises <bold>10,015 dermatoscopic images</bold> collected over 20 years from two principal sources: (1) the Department of Dermatology at the Medical University of Vienna, Austria, a tertiary referral center where diagnoses were established using a combination of histopathology, <italic>in vivo</italic> confocal microscopy, and expert consensus; and (2) a general skin cancer screening practice in Queensland, Australia, operated by Dr. Cliff Rosendahl. This second source provided images acquired in a real-world clinical setting, where lesions were typically triaged and either confirmed via <bold>histopathological examination</bold> or <bold>diagnosed by expert dermatologists</bold> with long-term follow-up. The dataset includes <bold>7 diagnostic categories</bold> representing the most common pigmented skin lesions seen in clinical practice. The distribution of the classes is presented in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Breakdown of classes in the HAM10000 dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Class name</bold></th>
<th valign="top" align="center"><bold>Number of images</bold></th>
<th valign="top" align="left"><bold>Description</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Melanoma (MEL)</td>
<td valign="top" align="center">1,113</td>
<td valign="top" align="left">Melanoma is a malignant tumor of melanin-producing melanocyte cells (<xref ref-type="bibr" rid="B38">National Cancer Institute, 2024a</xref>)</td>
</tr> <tr>
<td valign="top" align="left">Melanocytic nevus (NV)</td>
<td valign="top" align="center">6,705</td>
<td valign="top" align="left">Benign moles of pigment-producing skin cells (<xref ref-type="bibr" rid="B21">James et al., 2006</xref>)</td>
</tr> <tr>
<td valign="top" align="left">Basal cell carcinoma (BCC)</td>
<td valign="top" align="center">514</td>
<td valign="top" align="left">Slow-growing, locally destructive skin cancer derived from the basal cell layer of the epidermis (<xref ref-type="bibr" rid="B39">National Cancer Institute, 2024b</xref>)</td>
</tr> <tr>
<td valign="top" align="left">Actinic keratosis/Bowen&#x00027;s disease (AKIEC)</td>
<td valign="top" align="center">327</td>
<td valign="top" align="left">Precancerous scaly lesions found on sun-damaged skin (<xref ref-type="bibr" rid="B45">Reinehr and Bakos, 2019</xref>)</td>
</tr> <tr>
<td valign="top" align="left">Benign keratosis (BKL)</td>
<td valign="top" align="center">1,099</td>
<td valign="top" align="left">Common benign skin lesions with sharply demarcated borders, homogenous brown pigmentation, and fine scaling that include Seborrheic keratosis (SK), lichen planus-like keratosis (LPLK), and solar lentigo (SL) (<xref ref-type="bibr" rid="B50">Scott and Oakley, 2023</xref>)</td>
</tr> <tr>
<td valign="top" align="left">Dermatofibroma (DF)</td>
<td valign="top" align="center">115</td>
<td valign="top" align="left">Benign skin nodules of soft tissue (<xref ref-type="bibr" rid="B37">Myers and Fillman, 2024</xref>)</td>
</tr> <tr>
<td valign="top" align="left">Vascular lesions (VASC)</td>
<td valign="top" align="center">142</td>
<td valign="top" align="left">Lesions involving blood vessels, such as angiomas (<xref ref-type="bibr" rid="B56">Steiner and Drolet, 2017</xref>)</td>
</tr></tbody>
</table>
</table-wrap>
<p>In addition to images, the dataset also contains metadata for every patient, including clinical data like age, gender, and location of the lesion. With an <monospace>image_id</monospace> column, the patient&#x00027;s metadata can be associated with its respective lesion image. Such metadata is detailed in <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Clinical features in the HAM10000 dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Clinical feature</bold></th>
<th valign="top" align="left"><bold>Distribution</bold></th>
<th valign="top" align="left"><bold>Description</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Diagnosis (dx)</td>
<td valign="top" align="left">- Nevus (NV): 67% - Melanoma (MEL): 11% - Other types: 22%</td>
<td valign="top" align="left">Medical diagnosis of the skin lesion</td>
</tr> <tr>
<td valign="top" align="left">Diagnostic method</td>
<td valign="top" align="left">- Histopathology: 53% - Follow-up: 37% - Other methods: 10%</td>
<td valign="top" align="left">Method used to confirm the diagnosis</td>
</tr> <tr>
<td valign="top" align="left">Patient age</td>
<td valign="top" align="left">- Range: 0&#x02013;85 years - Divided into 10-year intervals</td>
<td valign="top" align="left">Age of the patient at the time of diagnosis</td>
</tr> <tr>
<td valign="top" align="left">Gender</td>
<td valign="top" align="left">- Male: 54% - Female: 45% - Unspecified: 1%</td>
<td valign="top" align="left">Patient&#x00027;s gender identification</td>
</tr> <tr>
<td valign="top" align="left">Anatomical location</td>
<td valign="top" align="left">- Back: 22% - Lower extremity: 21% - Other locations: 57%</td>
<td valign="top" align="left">Body location where the lesion was found</td>
</tr></tbody>
</table>
</table-wrap></sec>
<sec>
<title>3.2 Preprocessing pipeline</title>
<p>The preprocessing pipeline for the HAM10000 dataset was designed to ensure consistent, high-quality input data for our multimodal fusion model. This involved careful handling of image data, metadata, and class imbalance. The key steps are outlined below.</p>
<sec>
<title>3.2.1 Data splitting</title>
<p>Prior to any augmentation or preprocessing, the full dataset was randomly split into a <bold>70&#x02013;30 ratio</bold> for training and testing. No separate validation set was used. This ensured that augmented samples derived from the training set did not leak into the evaluation pipeline.</p></sec>
<sec>
<title>3.2.2 Class balancing</title>
<p>The HAM10000 dataset exhibits significant class imbalance, with some classes (e.g., DF) containing as few as 115 images and others (e.g., NV) containing over 6,000. To mitigate this, we applied data augmentation exclusively on the <bold>training set</bold> using the following techniques:</p>
<list list-type="bullet">
<list-item><p><bold>Replication</bold>: Duplicated samples from minority classes.</p></list-item>
<list-item><p><bold>Jittering</bold>: Added random noise to pixel intensities.</p></list-item>
<list-item><p><bold>Geometric transformations</bold>: Horizontal/vertical flips, rotations, and scaling.</p></list-item>
<list-item><p><bold>Random undersampling</bold>: Reduced samples from majority classes to avoid overwhelming the model.</p></list-item>
</list>
<p>After augmentation, each class in the training set contained 6,000 images, resulting in a balanced training dataset of 42,000 images.</p></sec>
<sec>
<title>3.2.3 Image preprocessing</title>
<p>Each dermatoscopic image underwent the following preprocessing steps:</p>
<list list-type="bullet">
<list-item><p><bold>Resizing</bold>: All images were resized to a uniform resolution of 256 &#x000D7; 256 pixels.</p></list-item>
<list-item><p><bold>Normalization</bold>: Pixel values were scaled to the range [0, 1] by dividing by 255.</p></list-item>
</list></sec>
<sec>
<title>3.2.4 Metadata preprocessing</title>
<p>Metadata was preprocessed in the following stages:</p>
<list list-type="bullet">
<list-item><p><bold>Handling missing values</bold>: For numerical features (e.g., age), missing values were imputed using the <bold>median</bold>, which is robust to outliers. For categorical features (e.g., gender, lesion location), missing entries were imputed with the <bold>mode</bold>. This imputation preserved the statistical structure of the data: using the median reduced sensitivity to skewed distributions, while mode imputation maintained categorical class balance with 98% fidelity.</p></list-item>
<list-item><p><bold>Encoding</bold>: After imputation, all categorical features were <bold>one-hot encoded</bold>. One-hot encoding was applied <italic>after</italic> mode imputation to enable the model to process these discrete attributes in a format suitable for neural networks.</p></list-item>
<list-item><p><bold>Normalization</bold>: Numerical features were standardized to have zero mean and unit variance.</p></list-item>
</list></sec>
<sec>
<title>3.2.5 Data alignment</title>
<p>To ensure proper alignment between images and metadata, we used the <monospace>image_id</monospace> column as the primary key to merge records. This guaranteed that each dermatoscopic image was correctly paired with its corresponding metadata during training and evaluation.</p>
<p>The complete preprocessing workflow is visualized in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>Overview of the preprocessing pipeline.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1608837-g0001.tif">
<alt-text>Flowchart for preprocessing the HAM10000 dataset. It starts with class balancing techniques like replication and random sampling. Image data is resized and normalized. Clinical data undergoes missing data imputation and one-hot encoding. Both branches align using image ID. The dataset is then split into validation (15%), training (70%), and testing sets (15%).</alt-text>
</graphic>
</fig></sec></sec>
<sec>
<title>3.3 Model pipeline process</title>
<p>The multimodal fusion pipeline consists of three main components: (1) a custom Weighted ResNet for extracting features from dermatoscopic images that we define as DermiResNet, (2) a Clinical MLP for processing clinical metadata, and (3) a fusion module that combines the extracted features for final classification. The pipeline operates as follows:</p>
<list list-type="bullet">
<list-item><p><bold>Input</bold>: Dermatoscopic images and clinical metadata are preprocessed and fed into the pipeline.</p></list-item>
<list-item><p><bold>Feature extraction</bold>: DermiResNet processes the images, while the Clinical MLP processes the metadata.</p></list-item>
<list-item><p><bold>Fusion</bold>: The extracted features are combined using one of several fusion techniques that are further discussed in the paper.</p></list-item>
<list-item><p><bold>Classification</bold>: The fused feature vector is passed through a Feed-forward Neural Network (FFN) to predict the skin lesion class.</p></list-item>
</list>
<p>The details of all networks has been provided in the architecture section and the pipeline has been illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>Overview of the multimodal fusion pipeline.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1608837-g0002.tif">
<alt-text>Diagram showing a machine learning model for skin condition prediction. Dermoscopic images and clinical metadata input into DermiResNet and Clinical MLP models. Outputs are fused and processed by a feedforward neural network (FFN) to generate predictions.</alt-text>
</graphic>
</fig></sec>
<sec>
<title>3.4 Architecture</title>
<sec>
<title>3.4.1 Clinical MLP</title>
<p>A <bold>Multi-Layer Perceptron (MLP)</bold> is a class of feedforward artificial neural networks composed of fully connected layers and nonlinear activation functions. MLPs are particularly well-suited for processing <bold>structured tabular data</bold> such as patient metadata, which lacks spatial structure and does not benefit from convolutional operations (<xref ref-type="bibr" rid="B48">Rumelhart et al., 1986</xref>).</p>
<p>In this study, we utilize a lightweight yet effective multi-layer perceptron (MLP) to process structured clinical metadata, including variables such as age, sex, and anatomical site of the lesion. The input is a 1D vector representation of all available metadata fields, which is first mapped to a 128-dimensional latent space via a fully connected layer with ReLU activation. This is followed by a second fully connected layer that expands the representation to 256 dimensions. This compact architecture is designed to extract meaningful latent embeddings from clinical features while being computationally efficient.</p>
<p><xref ref-type="fig" rid="F3">Figure 3</xref> delineates the architectural pathway of the Clinical MLP, emphasizing its role in encoding patient metadata into latent space.</p>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>Architecture of the clinical MLP.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1608837-g0003.tif">
<alt-text>Flowchart illustrating a process where a tabular input vector passes through an FC layer of 128 dimensions, a ReLU activation, another FC layer of 256 dimensions, and results in extracted clinical features.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<title>3.4.2 DermiResNet</title>
<p>Residual Networks (ResNets), introduced to mitigate the degradation problem in deep architectures, extend conventional convolutional networks by learning residual mappings instead of direct functions (<xref ref-type="bibr" rid="B20">He et al., 2016</xref>). Rather than learning an unreferenced function, ResNets learn a residual mapping, which simplifies optimization and enables very deep architectures. Unlike AlexNet, ResNet is conceptually derived from the simpler and deeper VGG networks (<xref ref-type="bibr" rid="B54">Simonyan and Zisserman, 2015</xref>), but it introduces residual or skip connections that alleviate vanishing gradient issues during training.</p>
<p>A typical residual block includes two convolutional layers. If <italic>x</italic> is the input to a block, and <italic>W</italic><sub>1</sub>, <italic>W</italic><sub>2</sub> are convolution kernels with non-linear activation &#x003C3;, then the residual output is given by:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>F</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>This results in the following expression for the block&#x00027;s output:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mi>F</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>x</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><monospace>DermiResNet</monospace> extends this formulation by introducing a learnable weight &#x003B1; for the skip connection:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mi>F</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mi>x</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Here, &#x003B1; &#x02208; &#x0211D; is a learnable scalar parameter that determines the relative importance of the shortcut connection, enabling the model to adaptively weigh the residual and identity paths during training (<xref ref-type="bibr" rid="B62">Xu et al., 2024</xref>).</p>
<p>The architecture begins with a primary convolutional module (<monospace>conv1</monospace>), after which the network progresses through four sequential stages. Each stage consists of a downsampling convolutional unit and a corresponding residual unit. Across these stages, the number of feature channels is gradually increased (64 &#x02192; 128 &#x02192; 256 &#x02192; 512), while spatial dimensions are reduced through stride-2 convolutions. Within each residual unit, two convolutional layers are used, each followed by batch normalization and LeakyReLU activation. To mitigate overfitting, dropout is applied in the later residual units (<monospace>res3</monospace> and <monospace>res4</monospace>).</p>
<p>The final classification head includes an adaptive average pooling layer, flattening, a two-layer fully connected network, and a softmax activation to output a 512-dimensional feature vector. The complete architecture is visualized in <xref ref-type="fig" rid="F4">Figures 4</xref>, <xref ref-type="fig" rid="F5">5</xref>.</p>
<fig position="float" id="F4">
<label>Figure 4</label>
<caption><p>DermiResNet.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1608837-g0004.tif">
<alt-text>Flowchart depicting the processing of dermoscopic images through a series of layers. It begins with dermoscopic images followed by convolutional layers Conv1 to Conv5, interspersed with residual layers Res1 to Res4. Circles labeled &#x0201C;W&#x0201D; appear after Res1, Res2, and Res3, indicating weights or connections. The final layer leads to the output feature space.</alt-text>
</graphic>
</fig>
<fig position="float" id="F5">
<label>Figure 5</label>
<caption><p>Blocks in DermiResNet.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1608837-g0005.tif">
<alt-text>Diagram of the proposed DermiResNet architecture with three blocks: Conv Block, Res Block, and Final Block. Conv Block includes Conv2D, LeakyReLU, Conv2D, and BatchNorm2D. Res Block consists of Conv2D, LeakyReLU, Dropout, and BatchNorm2D. Final Block contains AvgPool2D, Flatten and Linear, LeakyReLU, and Linear. Each block is vertically arranged.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<title>3.4.3 Fusion block</title>
<p>In this layer, we fuse the extracted features from the DermiResNet (image features) and the Clinical MLP (clinical metadata features). The fusion process combines these multimodal features into a unified representation, which is then passed to a simple classifier for final prediction. The classifier consists of fully connected layers with ReLU activation functions, reducing the fused feature space to 7 output classes corresponding to the skin lesion types.</p>
<p>We explore several fusion techniques to combine the features effectively, including:</p>
<list list-type="bullet">
<list-item><p>Simple concatenation</p></list-item>
<list-item><p>Weighted concatenation</p></list-item>
<list-item><p>Hadamard product</p></list-item>
<list-item><p>Tensor fusion</p></list-item>
<list-item><p>Bilinear fusion</p></list-item>
<list-item><p>Gated fusion</p></list-item>
<list-item><p>Self-attention</p></list-item>
<list-item><p>Cross-attention</p></list-item>
</list>
<p>Detailed descriptions of these fusion techniques, including their mathematical formulations and implementation, are provided in Section 3.5</p></sec></sec>
<sec>
<title>3.5 Fusion details</title>
<p>Fusion involves integrating information from different sources or data modalities into a single, cohesive representation. This approach is especially valuable in <bold>multimodal learning</bold>, where inputs such as images, metadata, or text are combined to enhance model accuracy and robustness. In this study, we combine features produced by the <bold>DermiResNet</bold> (which yields 512-dimensional image embeddings) and the <bold>Clinical MLP</bold> (which produces 256-dimensional clinical metadata embeddings). The resulting fused representation serves as the input to the classification layer. The following subsections present and analyze several fusion strategies explored in our experiments, highlighting their relative strengths and performance.</p>
<sec>
<title>3.5.1 Simple concatenation</title>
<p>Simple concatenation involves merging features from different modalities by <bold>stacking</bold> them along a specific dimension. Given two feature vectors <bold>x</bold> &#x02208; &#x0211D;<sup>512</sup> (from the image model) and <bold>y</bold> &#x02208; &#x0211D;<sup>256</sup> (from the clinical model), the concatenated feature vector <bold>z</bold> &#x02208; &#x0211D;<sup>768</sup> is:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><xref ref-type="disp-formula" rid="E4">Equation 4</xref> shows the simple concatenation of two feature vectors.</p></sec>
<sec>
<title>3.5.2 Weighted concatenation</title>
<p><bold>Weighted concatenation</bold> improves on simple concatenation by applying modality-specific weights. Instead of blindly stacking features, we scale each vector by a <bold>learnable</bold> scalar weight to reflect its importance (<xref ref-type="bibr" rid="B40">Ngiam et al., 2011</xref>; <xref ref-type="bibr" rid="B26">Kiela et al., 2018</xref>). Let <italic>w</italic><sub>1</sub>, <italic>w</italic><sub>2</sub> &#x02208; &#x0211D; be scalar weights applied to the 512D image vector <bold>x</bold> and 256D clinical vector <bold>y</bold>, respectively. The fused vector <bold>z</bold> &#x02208; &#x0211D;<sup>768</sup> is:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></sec>
<sec>
<title>3.5.3 Hadamard product fusion</title>
<p>The <bold>Hadamard product</bold>, or element-wise multiplication, combines two modalities by interacting their elements multiplicatively. Since this requires equal dimensions, we first project both <bold>x</bold> &#x02208; &#x0211D;<sup>512</sup> and <bold>y</bold> &#x02208; &#x0211D;<sup>256</sup> into a common latent space &#x0211D;<sup>256</sup>. The fused representation <bold>z</bold> &#x02208; &#x0211D;<sup>256</sup> is given by:</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02299;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Here, <inline-formula><mml:math id="M7"><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">Linear</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>512</mml:mn><mml:mo>&#x02192;</mml:mo><mml:mn>256</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, and &#x02299; represents element-wise multiplication (<xref ref-type="bibr" rid="B27">Kim et al., 2017</xref>).</p></sec>
<sec>
<title>3.5.4 Tensor fusion</title>
<p><bold>Tensor fusion</bold> captures all possible interactions between modalities using an outer product, resulting in a second-order tensor (<xref ref-type="bibr" rid="B63">Zadeh et al., 2017</xref>). For <bold>x</bold> &#x02208; &#x0211D;<sup>512</sup> and <bold>y</bold> &#x02208; &#x0211D;<sup>256</sup>, the fused tensor <bold>Z</bold> &#x02208; &#x0211D;<sup>512 &#x000D7; 256</sup> is:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold"><mml:mtext>Z</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>&#x02297;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>This allows pairwise modeling of every feature from one modality with every feature from the other, at the cost of increased dimensionality.</p></sec>
<sec>
<title>3.5.5 Bilinear fusion</title>
<p><bold>Bilinear fusion</bold> is a feature interaction mechanism that combines information from two different modalities by explicitly modeling the <bold>pairwise multiplicative interactions</bold> between their respective features. Unlike simple concatenation or element-wise operations, bilinear fusion generates a richer and more expressive representation by learning how every feature from one modality interacts with every feature from the other. Conceptually, bilinear fusion captures second-order statistics between modalities&#x02014;unlike first-order techniques such as concatenation, which only represent raw values. This is particularly valuable in multimodal tasks where the interplay between modalities is non-trivial and nonlinear (<xref ref-type="bibr" rid="B13">Fukui et al., 2016</xref>).</p>
<p>Given an image feature vector <bold>x</bold> &#x02208; &#x0211D;<sup>512</sup> and a clinical metadata feature vector <bold>y</bold> &#x02208; &#x0211D;<sup>256</sup>, the bilinear fusion mechanism applies a set of bilinear mappings to produce a fused feature vector <bold>z</bold> &#x02208; &#x0211D;<sup>256</sup>. Each dimension <italic>z</italic><sub><italic>i</italic></sub> of the output vector is computed using a learnable bilinear interaction:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:mrow></mml:msup><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mtext class="textrm" mathvariant="normal">for&#x000A0;</mml:mtext><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:mn>256</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Here, each <inline-formula><mml:math id="M10"><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mn>512</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mn>256</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> is a slice of the 3D learnable weight tensor <inline-formula><mml:math id="M11"><mml:mrow><mml:mrow><mml:mi mathvariant="script">W</mml:mi></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mn>256</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mn>512</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mn>256</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, such that:</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:mrow></mml:msup><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:mrow></mml:msup><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>256</mml:mn></mml:mrow></mml:msub><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>This results in a final fused representation <bold>z</bold> &#x02208; &#x0211D;<sup>256</sup>, which is then passed through a fully connected layer for classification.</p></sec>
<sec>
<title>3.5.6 Gated fusion</title>
<p><bold>Gated fusion</bold> introduces a dynamic weighting mechanism that assigns different attention to modalities based on input features (<xref ref-type="bibr" rid="B5">Arevalo et al., 2020</xref>). The gating vector <bold>g</bold> &#x02208; &#x0211D;<sup>256</sup> is derived from a sigmoid function applied to a learnable affine combination of inputs:</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M13"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold"><mml:mtext>g</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The final fusion is:</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>g</mml:mtext></mml:mstyle><mml:mo>&#x02299;</mml:mo><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>g</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02299;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Here, <inline-formula><mml:math id="M15"><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">Linear</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>512</mml:mn><mml:mo>&#x02192;</mml:mo><mml:mn>256</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, and &#x02299; denotes element-wise multiplication. This fusion strategy adaptively prioritizes modalities at an instance level.</p></sec>
<sec>
<title>3.5.7 Self-attention fusion</title>
<p><bold>Self-attention</bold> enables the model to attend to the most relevant parts of a single modality. Applied independently on each modality, it transforms the sequence of features <bold>X</bold> &#x02208; &#x0211D;<sup><italic>n</italic>&#x000D7;<italic>d</italic></sup> via attention weights:</p>
<disp-formula id="E12"><label>(12)</label><mml:math id="M16"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold"><mml:mtext>Z</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">softmax</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>Q</mml:mtext></mml:mstyle><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>K</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>V</mml:mtext></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where:</p>
<disp-formula id="E13"><mml:math id="M17"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>Q</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>Q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mstyle mathvariant="bold"><mml:mtext>K</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mstyle mathvariant="bold"><mml:mtext>V</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></disp-formula>
<p>Here, <inline-formula><mml:math id="M18"><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>Q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> are learnable weight matrices. In this context, <bold>Q</bold> (Query) represents the features for which we want to find contextual relevance, <bold>K</bold> (Key) encodes the features to be compared against, and <bold>V</bold> (Value) holds the actual information to be aggregated. Self-attention enables each position in the feature sequence to attend to all positions, allowing the model to learn intra-modality dependencies. It is used before fusion to enhance modality-specific representations (<xref ref-type="bibr" rid="B60">Vaswani et al., 2017</xref>).</p></sec>
<sec>
<title>3.5.8 Cross-attention fusion</title>
<p><bold>Cross-attention</bold> aligns features between modalities by using one modality as <bold>query</bold> and the other as <bold>key-value pairs</bold>. For image features <bold>X</bold> &#x02208; &#x0211D;<sup>1 &#x000D7; 512</sup> and metadata features <bold>Y</bold> &#x02208; &#x0211D;<sup>1 &#x000D7; 256</sup>, the query is derived from image features and the key/value from metadata:</p>
<disp-formula id="E14"><label>(13)</label><mml:math id="M19"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">softmax</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>Q</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>K</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:mrow></mml:msubsup></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>V</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where:</p>
<disp-formula id="E15"><mml:math id="M20"><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>Q</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>Q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>K</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>V</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></disp-formula>
<p>In cross-attention, <bold>Q</bold><sub><italic>x</italic></sub> represents the query derived from the primary modality (e.g., image features), which is seeking relevant complementary information. <bold>K</bold><sub><italic>y</italic></sub> and <bold>V</bold><sub><italic>y</italic></sub> are the key and value vectors derived from the secondary modality (e.g., clinical metadata), where the key determines alignment and the value contributes the corresponding context. This fusion allows modality <bold>X</bold> to selectively attend to modality <bold>Y</bold>, creating cross-modal representations that are aligned and context-aware (<xref ref-type="bibr" rid="B33">Lu et al., 2019</xref>).</p></sec></sec>
<sec>
<title>3.6 Experimental set-up</title>
<sec>
<title>3.6.1 Hardware and software configuration</title>
<p>The hardware and software specifications used for the experiments are summarized in <xref ref-type="table" rid="T4">Table 4</xref>.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Hardware and software configuration.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Component</bold></th>
<th valign="top" align="left"><bold>Specification</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">GPU</td>
<td valign="top" align="left">NVIDIA GeForce RTX 4060</td>
</tr> <tr>
<td valign="top" align="left">Processor</td>
<td valign="top" align="left">AMD Ryzen 7 7800X3D</td>
</tr> <tr>
<td valign="top" align="left">Memory</td>
<td valign="top" align="left">16GB DDR4 RAM</td>
</tr> <tr>
<td valign="top" align="left">Operating System</td>
<td valign="top" align="left">Linux Mint 21.1</td>
</tr> <tr>
<td valign="top" align="left">CUDA</td>
<td valign="top" align="left">Enabled</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<title>3.6.2 Training configuration</title>
<p>The specific training configuration, which outlines hyperparameters and other details, is documented in <xref ref-type="table" rid="T5">Table 5</xref>. Each model was trained for an estimated duration of 2 h.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Training configuration.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Parameter</bold></th>
<th valign="top" align="left"><bold>Value</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Batch size</td>
<td valign="top" align="left">64</td>
</tr> <tr>
<td valign="top" align="left">Number of epochs</td>
<td valign="top" align="left">100</td>
</tr> <tr>
<td valign="top" align="left">Learning rate</td>
<td valign="top" align="left">0.001</td>
</tr> <tr>
<td valign="top" align="left">Optimizer</td>
<td valign="top" align="left">Adam</td>
</tr> <tr>
<td valign="top" align="left">Loss function</td>
<td valign="top" align="left">Cross-entropy loss</td>
</tr></tbody>
</table>
</table-wrap></sec></sec>
<sec>
<title>3.7 Evaluation metrics</title>
<p>The performance of the model was evaluated using Accuracy, Precision, Recall, F1-Score, and AUC-ROC.</p></sec></sec>
<sec id="s4">
<title>4 Result analysis and discussion</title>
<sec>
<title>4.1 Ablation studies</title>
<p>To assess the contribution of each modality to the overall model performance, we first evaluated the individual models trained separately on clinical metadata and dermatoscopic images. The results of these experiments are summarized in <xref ref-type="table" rid="T6">Table 6</xref>.</p>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Performance of individual models in ablation studies.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>F1-Score</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Clinical MLP (metadata only)</td>
<td valign="top" align="center">77.0%</td>
<td valign="top" align="center">0.76</td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.76</td>
</tr> <tr>
<td valign="top" align="left">DermiResNet (image only)</td>
<td valign="top" align="center">92.0%</td>
<td valign="top" align="center">0.91</td>
<td valign="top" align="center">0.92</td>
<td valign="top" align="center">0.91</td>
</tr></tbody>
</table>
</table-wrap>
<p>The Clinical MLP achieved an accuracy of <bold>77.0%</bold>, indicating that clinical metadata alone offers moderate predictive capability. However, the DermiResNet, trained exclusively on dermatoscopic images, achieved a significantly higher accuracy of <bold>92.0%</bold>, showcasing the superior discriminative power of visual data for skin lesion classification. This result aligns with the diagnostic process commonly employed by dermatologists, where visual inspection of skin lesions is typically prioritized over metadata analysis for accurate classification (<xref ref-type="bibr" rid="B10">Dinnes et al., 2018</xref>).</p></sec>
<sec>
<title>4.2 Multimodal fusion performance</title>
<p>We evaluated the performance of various fusion techniques, including simple concatenation, weighted concatenation, Hadamard product, tensor fusion, bilinear fusion, gated fusion, self-attention, and cross-attention. The results are summarized in <xref ref-type="table" rid="T7">Table 7</xref>.</p>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>Performance of multimodal fusion techniques.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Fusion technique</bold></th>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Sensitivity (recall)</bold></th>
<th valign="top" align="center"><bold>F1-score</bold></th>
<th valign="top" align="center"><bold>Specificity</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Simple concatenation</td>
<td valign="top" align="center">96.5%</td>
<td valign="top" align="center">0.93</td>
<td valign="top" align="center">0.93</td>
<td valign="top" align="center">0.93</td>
<td valign="top" align="center">0.97</td>
</tr> <tr>
<td valign="top" align="left">Weighted concatenation</td>
<td valign="top" align="center">97.15%</td>
<td valign="top" align="center">0.97</td>
<td valign="top" align="center">0.97</td>
<td valign="top" align="center">0.97</td>
<td valign="top" align="center">0.98</td>
</tr> <tr>
<td valign="top" align="left">Hadamard product</td>
<td valign="top" align="center">98.85%</td>
<td valign="top" align="center">0.97</td>
<td valign="top" align="center">0.97</td>
<td valign="top" align="center">0.97</td>
<td valign="top" align="center">0.99</td>
</tr> <tr>
<td valign="top" align="left">Tensor fusion</td>
<td valign="top" align="center">96.52%</td>
<td valign="top" align="center">0.96</td>
<td valign="top" align="center">0.96</td>
<td valign="top" align="center">0.96</td>
<td valign="top" align="center">0.97</td>
</tr> <tr>
<td valign="top" align="left">Bilinear fusion</td>
<td valign="top" align="center">98.76%</td>
<td valign="top" align="center">0.98</td>
<td valign="top" align="center">0.98</td>
<td valign="top" align="center">0.98</td>
<td valign="top" align="center">0.99</td>
</tr> <tr>
<td valign="top" align="left">Gated fusion</td>
<td valign="top" align="center">93.0%</td>
<td valign="top" align="center">0.93</td>
<td valign="top" align="center">0.93</td>
<td valign="top" align="center">0.92</td>
<td valign="top" align="center">0.91</td>
</tr> <tr>
<td valign="top" align="left">Self-attention</td>
<td valign="top" align="center">92.70%</td>
<td valign="top" align="center">0.92</td>
<td valign="top" align="center">0.93</td>
<td valign="top" align="center">0.92</td>
<td valign="top" align="center">0.90</td>
</tr> <tr>
<td valign="top" align="left">Cross-attention</td>
<td valign="top" align="center">98.86%</td>
<td valign="top" align="center">0.98</td>
<td valign="top" align="center">0.98</td>
<td valign="top" align="center">0.98</td>
<td valign="top" align="center">0.99</td>
</tr></tbody>
</table>
</table-wrap>
<p>The comparison of different <bold>multimodal fusion methods</bold> provides profound insights into the design of AI-powered medical diagnostic systems. From among the methods, two high-performance methods are notable: <bold>Cross-attention</bold> and the <bold>Hadamard product</bold>, both of which deliver near state-of-the-art performance with respective accuracies of <bold>98.86%</bold> and <bold>98.85%</bold>.</p>
<p><bold>Cross-attention</bold>, one of the high-performance and intricate methods, performs exceptionally well by dynamically weighting and matching modalities&#x00027; features. Through the computation of attention scores between dermatoscopic image features and clinical metadata, it dynamically concentrates on the most discriminative data for every input. This imitates the subtle diagnosis reasoning of skilled clinicians to a great extent. For example, when visual features are uncertain (e.g., look-alike lesions), Cross-Attention uses contextual information like patient age or lesion site to sharpen its prediction. This capacity to represent high-grained, adaptive interactions makes it particularly useful for sophisticated diagnostic tasks such as skin lesion classification (<xref ref-type="bibr" rid="B41">Ou et al., 2022</xref>).</p>
<p><bold>Hadamard product</bold>, another high-performance method but relatively less complex in design, combines features through element-wise multiplication. It extracts localized interactions across modalities, for example, texture-lesion vs. age correlations, with remarkable performance. Nevertheless, in contrast to Cross-Attention, it does not have the dynamic feature importance adaptation capability that may limit its effectiveness in extremely ambiguous or nonlinear situations.</p>
<p>Conversely, certain <bold>sophisticated techniques performed poorly</bold>. <bold>Gated fusion</bold> (<bold>93.0% accuracy</bold>) and <bold>Self-attention fusion</bold> (<bold>92.70% accuracy</bold>) performed poorly despite their complex architectural design. Gated Fusion adds more learnable parameters in the form of gating mechanisms, which can lead to heightened overfitting risks and tougher training, particularly with small data sizes. Self-Attention is incredibly strong in a single modality but might miss key inter-modality dependencies when utilized stand-alone, thus reducing its multimodal performance.</p>
<p>Surprisingly, also simple approaches can perform well. <bold>Weighted concatenation</bold>, a low-complexity method, performed <bold>97.15% accuracy</bold>. By using static or learnable weights for every modality prior to concatenation, it is good at balancing interpretability, stability, and performance. Without modeling high-level feature interactions, its simplicity and stability make it a very practical solution for most clinical scenarios.</p>
<p>These findings highlight an important point: <bold>algorithmic complexity is no guarantee of better performance</bold>. The selection of fusion strategy must be informed by the particular properties of the dataset and clinical application, striking a balance between accuracy, interpretability, and computational cost.</p>
<p>In conclusion, while sophisticated fusion mechanisms such as <bold>Cross-attention</bold> deliver state-of-the-art performance by dynamically aligning visual and clinical modalities, our results demonstrate that in some scenarios, simpler techniques like Weighted Concatenation are highly competitive offering comparable accuracy with significantly lower computational overhead. This makes them well-suited for practical deployment in real-world medical settings, particularly where resources are constrained or rapid inference is essential. Cross-attention, on the other hand, proves most valuable in cases involving complex diagnostic patterns&#x02014;such as ambiguous lesions or subtle correlations between metadata and visual features, where adaptive modality interaction becomes critical.</p>
<p>Ultimately, the success of a multimodal fusion approach lies not merely in algorithmic sophistication but in its ability to replicate the holistic, context-aware diagnostic reasoning employed by clinicians. The choice of fusion strategy should thus be guided by the specific clinical setting, data complexity, and resource constraints, striking a balance between accuracy, interpretability, and efficiency.</p></sec>
<sec>
<title>4.3 Confusion matrix analysis</title>
<p>To further analyze the performance of the best-performing fusion model (Cross-attention), we present its confusion matrix in <xref ref-type="fig" rid="F6">Figure 6</xref>. The matrix shows the number of correct and incorrect predictions for each class.</p>
<fig position="float" id="F6">
<label>Figure 6</label>
<caption><p>Visualization of the confusion matrix for the best-performing Cross-attention model. This heatmap helps identify specific misclassifications and overall model behavior across classes.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1608837-g0006.tif">
<alt-text>Confusion matrix for a classification model showing true labels versus predicted labels for seven categories: BKL, BCC, DF, MEL, NV, VASC, and AKIEC. Diagonal values represent correct predictions, with high numbers for each category, indicating strong performance. Off-diagonal values indicate misclassifications, which are relatively low.</alt-text>
</graphic>
</fig>
<p>The confusion matrix highlights the performance of the model across various skin cancer classes. Notably, the diagonal entries, which represent correctly classified instances, dominate, demonstrating strong overall performance.</p>
<p>The model achieves near-perfect classification for classes like <bold>DF</bold>, <bold>VASC</bold>, and <bold>AKIEC</bold>, with minimal or no misclassifications. This suggests that these classes are well-represented in the training data and exhibit distinct features, allowing the model to identify them with high confidence.</p>
<p>However, there are minor misclassifications, particularly between similar-looking lesion types such as <bold>BKL</bold> and <bold>MEL</bold> or <bold>NV</bold>. For instance, 7 cases of <bold>BKL</bold> are misclassified as <bold>MEL</bold>, and 5 as <bold>AKIEC</bold>, indicating some overlap in visual or clinical features. Similarly, 4 cases of <bold>NV</bold> are mistaken for <bold>MEL</bold>, which is expected given their subtle differences and shared features in certain instances.</p>
<p>The small number of misclassifications in <bold>BCC</bold> and <bold>MEL</bold> classes suggests the model handles malignant lesions well but still requires improvements to reduce errors in high-stakes scenarios.</p>
<p>Overall, the model demonstrates high classification accuracy with room for improvement in distinguishing between lesion types with overlapping visual or clinical characteristics.</p></sec>
<sec>
<title>4.4 ROC-AUC curve</title>
<p>The ROC-AUC scores for different classes are presented in the figure below. The model achieved a score of 0.99 for MEL, 0.98 for NV, and 1.0 for the remaining classes, as shown in the ROC curve (<xref ref-type="fig" rid="F7">Figure 7</xref>).</p>
<fig position="float" id="F7">
<label>Figure 7</label>
<caption><p>ROC-AUC curve for melanoma, nevi, and other classes.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1608837-g0007.tif">
<alt-text>Receiver Operating Characteristic (ROC) curve showing the true positive rate versus false positive rate for various skin conditions. Conditions include benign lesions, basal cell carcinoma, dermatofibroma, melanoma, melanocytic nevi, vascular lesions, and actinic keratoses, with respective area under the curve (AUC) values ranging from 0.98 to 1.00. A diagonal dashed line represents random chance.</alt-text>
</graphic>
</fig>
<p>These high AUC values suggest that the model is capable of effectively distinguishing between the classes. The results indicate that there is no sign of underfitting or overfitting, with the model generalizing well and avoiding excessive fitting to noise in the training data.</p></sec>
<sec>
<title>4.5 Explainability</title>
<p>Explainability is a foundation of applying machine learning models in healthcare. Although deep learning models tend to exhibit state-of-the-art performance, their &#x0201C;black-box&#x0201D; nature is highly problematic in the healthcare domain. Physicians and medical experts need interpretable models for them to comprehend the rationale of predictions, making diagnoses not only correct but also clinically reasonable. Lack of transparency can translate to mistrust, preventing the implementation of AI systems into actual healthcare workflows.</p>
<p>Black-box models that give no clue about their reasoning are especially troubling in high-risk areas such as dermatology. For example, a model can be highly accurate by leveraging spurious correlations or data artifacts instead of clinically significant features, leading to serious failures when deployed in more diverse or unseen clinical environments (<xref ref-type="bibr" rid="B64">Zech et al., 2018</xref>). Explainability closes this gap by illuminating how a model comes to a decision, allowing doctors to verify its reasoning and spot potential errors or biases (<xref ref-type="bibr" rid="B4">Amann et al., 2020</xref>).</p>
<p>This research emphasize explainability so that the multimodal fusion model is not just precise but also reliable and comprehensible. Through the use of visual explanations (Grad-CAM) coupled with relevance analysis of clinical features, an integrated understanding of the decision-making process of the model is presented.</p>
<sec>
<title>4.5.1 Grad-CAM</title>
<p>Grad-CAM is a powerful technique for visualizing the regions of an image that are most influential in a model&#x00027;s decision. It extends the Class Activation Mapping (CAM) approach by using gradient information to weight the importance of feature maps, making it applicable to a wider range of architectures, including those without global average pooling layers.</p>
<p><bold>Mathematical formulation</bold>: Let <italic>A</italic><sup><italic>k</italic></sup> be the activation map of the <italic>k</italic>-th channel in the target convolutional layer, and let <italic>y</italic><sup><italic>c</italic></sup> be the score for class <italic>c</italic>. The weight <inline-formula><mml:math id="M21"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> for the <italic>k</italic>-th channel is computed as the global average of the gradients of <italic>y</italic><sup><italic>c</italic></sup> with respect to <italic>A</italic><sup><italic>k</italic></sup>:</p>
<disp-formula id="E16"><mml:math id="M22"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>Z</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msup><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>where <italic>Z</italic> is the number of pixels in the activation map. These weights capture the importance of each feature map for the target class.</p>
<p>The Grad-CAM heatmap <italic>L</italic><sup><italic>c</italic></sup> is then obtained by a weighted combination of the activation maps, followed by a ReLU function:</p>
<disp-formula id="E17"><mml:math id="M23"><mml:mrow><mml:msup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">ReLU</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:msup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula>
<p>The ReLU function ensures that only positive influences are considered, as negative values are not relevant for the target class (<xref ref-type="bibr" rid="B51">Selvaraju et al., 2017</xref>).</p>
<p>Grad-CAM indicates areas of the image that the model considers most significant for its prediction. For instance, in images obtained with dermatoscopy, these areas may be lesion boundaries, texture, or color transitions. By projecting these areas, Grad-CAM opens a window to the model&#x00027;s &#x0201C;thinking process,&#x0201D; allowing physicians to ensure that the AI system is concentrating on clinically relevant features (<xref ref-type="bibr" rid="B22">Jaworek-Korjakowska et al., 2021</xref>).</p></sec>
<sec>
<title>4.5.2 Analysis of grad-CAM results for cross-attention</title>
<p>To understand the decision-making process of the top-performing model, which utilizes a cross-attention fusion approach, Grad-CAM visualizations were employed. This technique facilitates the interpretation of the model&#x00027;s predictions by identifying key areas in dermatoscopic images that most strongly influence the results. Such <italic>post-hoc</italic> explainability is especially valuable in medical contexts, as it allows clinicians to confirm that the model&#x00027;s predictions are grounded in clinically significant features.</p>
<p>The cross-attention fusion mechanism enhances the interpretability of the model by dynamically aligning and integrating features from different modalities or scales. Unlike simpler fusion techniques, cross-attention computes attention weights between modalities, enabling the model to emphasize the most diagnostically relevant interactions. While Grad-CAM visualizations help interpret the image-based decision process by highlighting regions of importance in dermatoscopic inputs, they are limited to visual modalities. To gain a holistic understanding of the model&#x00027;s reasoning, particularly for clinical metadata, we later discuss Clinical Feature Relevance Scores, which complement Grad-CAM by providing insight into how non-image features influence predictions. This combined interpretability offers a more comprehensive explanation of the model&#x00027;s diagnostic behavior.</p>
<p><xref ref-type="fig" rid="F8">Figure 8</xref> provides an illustrative example of a Grad-CAM heatmap overlaid on an input dermatoscopic image. The model predicts the lesion as BCC with a confidence score of 1, focusing primarily on the lesion&#x00027;s irregular borders and regions of heterogeneous pigmentation. These features are clinically significant as BCC are characterized by asymmetry, border irregularity, and color variation (<xref ref-type="bibr" rid="B42">Puckett et al., 2025</xref>).</p>
<fig position="float" id="F8">
<label>Figure 8</label>
<caption><p>Grad-CAM visualization for a dermatoscopic image classified as Basal Cell Carcinoma. The left panel shows the original image, while the right panel highlights clinically significant regions contributing to the model&#x00027;s prediction. Red regions indicate high importance, and blue regions indicate low importance.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1608837-g0008.tif">
<alt-text>Original image of a skin lesion identified as basal cell carcinoma next to a Grad-CAM visualization. The Grad-CAM highlights the lesion area with vibrant colors, indicating the model&#x00027;s focus, predicting basal cell carcinoma with full confidence.</alt-text>
</graphic>
</fig>
<p>In this specific example, the heatmap demonstrates that the model successfully identifies features associated with malignancy while ignoring irrelevant artifacts, such as hair strands and uniform skin areas. The model&#x00027;s ability to focus on clinically relevant features is a direct outcome of the cross-attention fusion mechanism, which dynamically aligns and weights features from different modalities. By doing so, the model achieves a higher level of alignment with dermatological practices, where lesion borders and internal variations are critical for diagnosis.</p>
<p>As discussed earlier, the model occasionally misclassifies BKL as MEL due to overlapping visual characteristics. One contributing factor is that both lesion types can exhibit irregular pigmentation, asymmetric structures, and varying border definitions, which are also key diagnostic markers for MEL. This confusion is particularly evident in cases where keratosis presents with darker pigmentation and irregular borders, mimicking features of malignant lesions.</p>
<p><xref ref-type="fig" rid="F9">Figure 9</xref> illustrates such a misclassification, where a BKL lesion has been incorrectly predicted as MEL with a confidence score of 0.46. The left panel shows the original image, while the right panel presents the corresponding Grad-CAM heatmap, highlighting the regions that influenced the model&#x00027;s decision.</p>
<fig position="float" id="F9">
<label>Figure 9</label>
<caption><p>Grad-CAM visualization of a misclassified Benign Keratosis case. The model incorrectly predicts this lesion as Melanocytic (MEL) with a confidence of 0.46.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1608837-g0009.tif">
<alt-text>Skin shows two dark brown lesions on the left, labeled &#x0201C;Original Image.&#x0201D; On the right, a Grad-CAM visualization highlights regions of interest with warm colors, indicating a confidence prediction of 0.46 for melanocytic labels. The true diagnosis is benign keratosis lesions.</alt-text>
</graphic>
</fig>
<p>From the heatmap, it is evident that the model assigns high importance (red/yellow regions) to darker pigmented areas and irregular structures within the lesion. This suggests that the model&#x00027;s decision boundary between benign and malignant lesions is influenced primarily by pigmentation and border irregularity, which are not always exclusive to melanoma.</p>
<p>Beyond this example, Grad-CAM visualizations across a wide range of test cases reveal consistent patterns in the model&#x00027;s behavior. The model often prioritizes irregular lesion borders and regions of color variation, which are crucial for distinguishing malignant lesions from benign ones. In cases of BCC, the heatmaps highlight central regions with ulceration or shiny surfaces, further reinforcing the model&#x00027;s alignment with clinical indicators. This interpretability is invaluable for real-world deployment, as it allows clinicians to confirm that the model&#x00027;s focus areas correspond to meaningful diagnostic features.</p>
<p>The insights provided by Grad-CAM are directly correlated with the model&#x00027;s strong quantitative performance. For example, the regions identified by the heatmaps frequently align with features responsible for the model&#x00027;s high sensitivity and specificity, particularly in challenging cases of melanoma and basal cell carcinoma. This combination of performance metrics and visual interpretability underscores the potential of the cross-attention fusion-based model as a trustworthy diagnostic aid.</p>
<p>Additional Grad-CAM visualizations for Cross-attention are provided in <xref ref-type="fig" rid="F10">Figure 10</xref>, showcasing further examples of model attention across different lesion types and cases.</p>
<fig position="float" id="F10">
<label>Figure 10</label>
<caption><p>Composite visualization of Grad-CAM outputs for selected classification cases. For each sample, the left panel shows the Grad-CAM heatmap overlaid on the dermatoscopic image, highlighting regions the model attended to during prediction. The right panel displays the original dermatoscopic image with the ground truth label indicated. This layout enables intuitive interpretation of model focus and correctness of attention alignment.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1608837-g0010.tif">
<alt-text>Comparison of skin condition images using Grad-CAM and original views. The first row shows Actinic Keratosis; Grad-CAM highlights hot spots while the original displays a textured lesion. The second row features Basal Cell Carcinoma; Grad-CAM highlights different regions than the mottled texture shown in the original. The third row shows Melanocytic Nevus; Grad-CAM emphasizes a specific area, contrasting the dark spot in the original. The fourth row displays Vascular Lesion with Grad-CAM highlighting multiple regions not as visible in the original, which shows a detailed, vascular pattern.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<title>4.5.3 Clinical feature relevance</title>
<p>To assess the instance-specific relevance of clinical features, we employed an attribution-based approach using Integrated Gradients (IG) (<xref ref-type="bibr" rid="B57">Sundararajan et al., 2017</xref>). IG is a path-based attribution method that quantifies feature importance by computing the integral of gradients along an interpolation path from a baseline input to the actual input.</p>
<p>Given an input clinical feature vector <italic>x</italic> &#x02208; &#x0211D;<sup><italic>d</italic></sup> and a baseline vector <italic>x</italic>&#x02032; &#x02208; &#x0211D;<sup><italic>d</italic></sup>, the integrated gradient for the <italic>i</italic><sup><italic>th</italic></sup> feature is computed as:</p>
<disp-formula id="E18"><label>(14)</label><mml:math id="M24"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>I</mml:mi><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x003B1;</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mstyle><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>-</mml:mo><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mi>d</mml:mi><mml:mi>&#x003B1;</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>f</italic>(&#x000B7;) represents the model&#x00027;s prediction score for the target class. The integral is approximated using a summation over discrete steps:</p>
<disp-formula id="E19"><label>(15)</label><mml:math id="M25"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>I</mml:mi><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02248;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:mfrac><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:mfrac><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>-</mml:mo><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>S</italic> is the number of interpolation steps.</p>
<p>For each instance, we set the baseline <italic>x</italic>&#x02032; as a zero vector, representing the absence of clinical features. The integrated gradients were computed over <italic>S</italic> &#x0003D; 50 steps, and the absolute values of the attributions were taken as feature importance scores. These scores were then normalized to the range [0, 1] to facilitate interpretability.</p>
<p>The resultant feature relevance scores highlight the contribution of individual clinical attributes to the model&#x00027;s prediction. Features with higher attributions indicate stronger influence on the classification decision, thereby providing insights into the model&#x00027;s reliance on clinical metadata.</p></sec>
<sec>
<title>4.5.4 Clinical feature relevance analysis</title>
<p>For a specific instance of <bold>NV</bold>, we analyzed the relative importance of clinical features using our best-performing Cross-Attention fusion model. As shown in <xref ref-type="table" rid="T8">Table 8</xref>, the top features with high importance include <bold>localization</bold> (scalp, foot, neck), and <bold>diagnostic methods</bold> (consensus, follow-up, confocal). These features likely play a significant role due to the distinct characteristics of lesions in these locations and the reliability of the diagnostic methods. Features with moderate importance, such as localization (face, genital) and diagnosis type (histopathology), contribute to predictions but are less critical. Interestingly, <bold>age</bold> and certain locations (e.g., ear, lower extremity) show low importance, suggesting they have minimal influence on the model&#x00027;s predictions for this instance. The balanced impact of <bold>sex</bold> (both male and female) indicates that while it influences outcomes, it is not among the strongest predictors. Overall, localization emerges as the most relevant feature, aligning with real-world clinical practice where lesion location is a key diagnostic factor.</p>
<table-wrap position="float" id="T8">
<label>Table 8</label>
<caption><p>Instance-specific clinical feature importance for a melanocytic nevus instance.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Clinical feature</bold></th>
<th valign="top" align="center"><bold>Relative importance</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Dx type: follow-up</td>
<td valign="top" align="center">1.00</td>
</tr> <tr>
<td valign="top" align="left">Localization: hand</td>
<td valign="top" align="center">0.80</td>
</tr> <tr>
<td valign="top" align="left">Localization: scalp</td>
<td valign="top" align="center">0.75</td>
</tr> <tr>
<td valign="top" align="left">Localization: neck</td>
<td valign="top" align="center">0.70</td>
</tr> <tr>
<td valign="top" align="left">Localization: acral</td>
<td valign="top" align="center">0.65</td>
</tr> <tr>
<td valign="top" align="left">Localization: lower extremity</td>
<td valign="top" align="center">0.60</td>
</tr> <tr>
<td valign="top" align="left">Localization: chest</td>
<td valign="top" align="center">0.55</td>
</tr> <tr>
<td valign="top" align="left">Localization: unknown</td>
<td valign="top" align="center">0.50</td>
</tr> <tr>
<td valign="top" align="left">Localization: abdomen</td>
<td valign="top" align="center">0.45</td>
</tr> <tr>
<td valign="top" align="left">Localization: genital</td>
<td valign="top" align="center">0.40</td>
</tr> <tr>
<td valign="top" align="left">Sex: male</td>
<td valign="top" align="center">0.35</td>
</tr> <tr>
<td valign="top" align="left">Dx type: consensus</td>
<td valign="top" align="center">0.30</td>
</tr> <tr>
<td valign="top" align="left">Sex: unknown</td>
<td valign="top" align="center">0.25</td>
</tr> <tr>
<td valign="top" align="left">Localization: upper extremity</td>
<td valign="top" align="center">0.20</td>
</tr> <tr>
<td valign="top" align="left">Dx type: confocal</td>
<td valign="top" align="center">0.15</td>
</tr> <tr>
<td valign="top" align="left">Localization: trunk</td>
<td valign="top" align="center">0.12</td>
</tr> <tr>
<td valign="top" align="left">Localization: face</td>
<td valign="top" align="center">0.10</td>
</tr> <tr>
<td valign="top" align="left">Sex: female</td>
<td valign="top" align="center">0.08</td>
</tr> <tr>
<td valign="top" align="left">Localization: back</td>
<td valign="top" align="center">0.06</td>
</tr> <tr>
<td valign="top" align="left">Localization: ear</td>
<td valign="top" align="center">0.04</td>
</tr> <tr>
<td valign="top" align="left">Localization: foot</td>
<td valign="top" align="center">0.03</td>
</tr> <tr>
<td valign="top" align="left">Age</td>
<td valign="top" align="center">0.02</td>
</tr> <tr>
<td valign="top" align="left">Dx type: histopathology</td>
<td valign="top" align="center">0.01</td>
</tr></tbody>
</table>
</table-wrap>
<p>Across various lesion types, we observed that <bold>localization</bold> consistently emerged as the most important clinical feature, underscoring its critical role in skin lesion diagnosis. This aligns with real-world clinical practice, where the anatomical location of a lesion is a key diagnostic factor due to its correlation with sun exposure, skin type, and lesion characteristics.</p>
<p>In cases of <bold>MEL</bold>, <bold>localization</bold> (e.g., back, face) was the top predictor, reflecting the higher prevalence of malignant lesions in sun-exposed areas. Additionally, <bold>age</bold> and <bold>gender</bold> showed moderate influence, consistent with their established roles as risk factors for melanoma. Older patients and males exhibited a higher likelihood of malignancy, further validating the model&#x00027;s alignment with epidemiological trends (<xref ref-type="bibr" rid="B43">Raimondi et al., 2020</xref>; <xref ref-type="bibr" rid="B14">Garbe et al., 2021</xref>).</p>
<p>For <bold>BCC</bold>, the model again prioritized <bold>localization</bold> (e.g., face, neck), as these areas are most susceptible to UV damage (<xref ref-type="bibr" rid="B34">Marzuka and Book, 2015</xref>). Diagnostic methods such as <bold>histopathology</bold> also played a significant role, as BCC is often confirmed through biopsy. Interestingly, <bold>age</bold> showed a stronger influence for BCC compared to other lesion types, likely due to the cumulative effect of UV exposure over time (<xref ref-type="bibr" rid="B52">Seretis et al., 2025</xref>).</p>
<p>In cases of <bold>AKIEC</bold>, <bold>localization</bold> (e.g., face, scalp) remained the most important feature, as AKIEC lesions are strongly associated with chronic sun exposure (<xref ref-type="bibr" rid="B45">Reinehr and Bakos, 2019</xref>). The model also highlighted the importance of <bold>diagnostic methods</bold> (e.g., follow-up, confocal microscopy), reflecting the need for repeated evaluations to monitor these precancerous lesions.</p>
<p>For <bold>DF</bold>, a benign lesion, <bold>localization</bold> (e.g., lower extremities) was again the dominant feature, while <bold>age</bold> and <bold>gender</bold> had minimal influence. This suggests that the visual appearance and location of DF lesions are more critical for diagnosis than demographic factors (<xref ref-type="bibr" rid="B18">Han et al., 2011</xref>).</p>
<p>Finally, in cases of <bold>VASC</bold>, <bold>localization</bold> (e.g., face, trunk) was the most influential feature, as these lesions often appear in specific anatomical regions (<xref ref-type="bibr" rid="B35">Mulligan et al., 2014</xref>). The model also relied heavily on <bold>dermoscopic examination</bold>, highlighting the importance of visual data for diagnosing vascular lesions.</p></sec></sec></sec>
<sec sec-type="conclusions" id="s5">
<title>5 Conclusion</title>
<p>The effectiveness of multimodal fusion methods in improving skin lesion classification performance is well illustrated through this study. By combining clinical metadata with dermatoscopic images, the model reached state-of-the-art accuracy, with the <bold>Cross-attention</bold> and <bold>Hadamard product</bold> methods reaching almost <bold>99% accuracy</bold>. Importantly, these accuracies were reached with a basic laptop setup, without the necessity for specialized Neurap Processing Units (NPUs) or workstations, making the proposed method efficient and accessible. This is especially important for resource-limited settings, where sophisticated computational facilities might not be easily accessible.</p>
<p>The strength of these fusion methods rests in their capacity to merge complementary information from visual and clinical data, emulating the comprehensive diagnostic reasoning that seasoned clinicians often exhibit. The <bold>Cross-attention</bold> method, specifically, performed exceptionally well by dynamically correlating and weighting features from both modalities to allow the model to concentrate on the most discriminative information per input. Likewise, the <bold>Hadamard product</bold> also performed well by appropriating local modality relationships through element-wise multiplication of feature vectors.</p>
<p>These results highlight the need for carefully choosing fusion techniques depending on the specific characteristics of the dataset and clinical context. Though more sophisticated methods like Cross-Attention present better performance in challenging scenarios, simpler approaches such as <bold>Weighted concatenation</bold> offer a useful trade-off between accuracy and computational expense.</p>
<sec>
<title>5.1 Limitations and future directions</title>
<p>Although the findings of this study highlight the capability of multimodal fusion to improve skin lesion classification, it is essential to recognize various limitations requiring future work and improvement.</p>
<list list-type="bullet">
<list-item><p><bold>Computational resource requirements</bold>: While the adopted methodologies illustrate feasibility on common computing hardware, the inherent computational intensity of sophisticated fusion methods, specifically tensor fusion and cross-attention, poses a significant computational burden. Future studies should focus on developing and utilizing optimization techniques. Model pruning, quantization, and knowledge distillation are some techniques that need to be explored to reduce computational overhead without compromising diagnostic performance.</p></list-item>
<list-item><p><bold>Generalizability across diverse populations</bold>: The use of the HAM10000 dataset, as comprehensive as it is, may not perfectly capture the heterogeneity of skin lesions observed in real-world clinical practice. Therefore, the generalizability of the model to diverse patient populations remains an important challenge. Future research should include external validation by evaluating performance on datasets such as ISIC and Pedro Hispano-2 (PH2). Additionally, expanding training data with a broader range of clinical and demographic variables is crucial to enhancing the model&#x00027;s robustness and validity.</p></list-item>
<list-item><p><bold>Integration of extensive clinical data</bold>: The predictive capability of the present model is constrained by the limited range of clinical features available in the HAM10000 dataset. To address this, future research should focus on integrating more comprehensive clinical data. This includes patient medical histories, genetic predisposition, and laboratory test results, which collectively contribute to a more holistic diagnostic analysis.</p></list-item>
<list-item><p><bold>Improving model interpretability for clinical trust</bold>: Clinical adoption of AI-based diagnostic tools is contingent on their interpretability. Although Grad-CAM provides a visual interpretation of feature importance, a deeper understanding of the model&#x00027;s decision-making process is necessary. Future studies should explore advanced Explainable AI (XAI) techniques, such as combining Grad-CAM with clinical feature relevance analysis or developing hybrid models that provide both visual and textual explanations.</p></list-item>
<list-item><p><bold>Refinement of fusion methodologies</bold>: The success of multimodal fusion depends on the selection and optimization of appropriate techniques. Future research should explore adaptive fusion methods that dynamically adjust based on input features. Additionally, investigating ensemble fusion techniques that leverage the strengths of multiple fusion strategies could lead to significant improvements in diagnostic accuracy.</p></list-item>
<list-item><p><bold>Validation in real-world clinical environments</bold>: To assess the practical effectiveness of the proposed system, rigorous validation in real-world clinical settings is essential. Future studies should emphasize real-time deployment of the model in diagnostic workflows, ensuring close collaboration with dermatologists and healthcare professionals to address implementation challenges and optimize the system based on real-world feedback.</p></list-item>
<list-item><p><bold>Handling class imbalance and rare phenotypes</bold>: The misclassification of rare transitions, such as melanoma being classified as nevus or benign keratosis, underscores the need for better handling of class imbalances and subtle feature variations. Beyond basic augmentation, future research should investigate more advanced strategies, such as:</p></list-item></list>
<list list-type="simple">
<list-item><p>&#x02022; <italic>Focal Loss</italic>, which down-weights easy examples and focuses training on hard negatives, improving detection of minority classes (<xref ref-type="bibr" rid="B31">Lin et al., 2020</xref>).</p></list-item>
<list-item><p>&#x02022; <italic>Synthetic oversampling using GANs</italic>, such as Deep Convolutional Architectures for Image Synthesis (DCGAN) or Style-Based Image Generation Networks (StyleGAN2), to generate realistic lesion images for underrepresented classes like dermatofibroma or vascular lesions (<xref ref-type="bibr" rid="B12">Frid-Adar et al., 2018</xref>; <xref ref-type="bibr" rid="B36">Mutepfe et al., 2021</xref>).</p></list-item></list>
<list list-type="bullet">
<list-item><p><bold>Improving preprocessing resilience</bold>: The tendency of the model to focus on non-essential regions, such as hair strands or glossy skin, highlights the need for more robust preprocessing techniques. Enhancing hair removal algorithms and contrast normalization strategies is crucial to eliminating distractions and improving the model&#x00027;s reliability in assessing key lesion features.</p></list-item>
<list-item><p><bold>Color constancy and harmonization</bold>: Variations in lighting and acquisition devices can lead to inconsistent image appearance. Future work should explore <italic>color constancy algorithms</italic> to normalize illumination conditions across samples. Techniques like <italic>Shades of Gray, Gray World</italic>, and <italic>Learning-Based Color Constancy</italic> could significantly reduce lighting-induced variance (<xref ref-type="bibr" rid="B8">Bianco and Cusano, 2019</xref>; <xref ref-type="bibr" rid="B7">Barnard et al., 2002</xref>). This harmonization is critical for enhancing cross-device robustness in clinical deployment.</p></list-item>
<list-item><p><bold>Cross-referencing segmentation with Grad-CAM</bold>: While Grad-CAM provides valuable insights into model attention, it does not guarantee alignment with lesion boundaries. Future work should involve cross-referencing Grad-CAM heatmaps with lesion segmentation masks [e.g., generated using the <bold>Segment Anything Model (SAM)</bold> (<xref ref-type="bibr" rid="B28">Kirillov et al., 2023</xref>)] to verify that the model is focusing on diagnostically relevant regions. This integration could improve both model explainability and diagnostic reliability.</p></list-item>
</list>
<p>This research lays a strong foundation for applying multimodal fusion in skin lesion classification. By addressing the identified limitations and exploring the proposed future directions, we can advance the development of precise, efficient, and clinically viable AI-driven diagnostic systems, ultimately leading to improved outcomes for patients with dermatological conditions.</p></sec></sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com/datasets/kmader/skin-cancer-mnist-ham10000">https://www.kaggle.com/datasets/kmader/skin-cancer-mnist-ham10000</ext-link> (<xref ref-type="bibr" rid="B49">Scott Mader, 2018</xref>).</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>AD: Writing &#x02013; review &#x00026; editing, Writing &#x02013; original draft, Project administration, Visualization, Formal analysis, Data curation, Validation, Methodology. VA: Writing &#x02013; review &#x00026; editing, Investigation, Visualization. NS: Methodology, Project administration, Supervision, Conceptualization, Investigation, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. The funding for the publication of this article was provided by Manipal Academy of Higher Education, Manipal, India.</p>
</sec>
<ack><p>The authors would like to thank the Department of Information and Communication Technology, Manipal Institute of Technology for providing the necessary research infrastructure. We also acknowledge Cryptonite Student Project&#x00027;s Research Division for their valuable insights during the initial phases of this study.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p></sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Adarsh</surname> <given-names>V.</given-names></name> <name><surname>Gangadharan</surname> <given-names>G. R.</given-names></name> <name><surname>Fiore</surname> <given-names>U.</given-names></name> <name><surname>Zanetti</surname> <given-names>P.</given-names></name></person-group> (<year>2024</year>). <article-title>Multimodal classification of alzheimer&#x00027;s disease and mild cognitive impairment using custom mkscddl kernel over CNN with transparent decision-making for explainable diagnosis</article-title>. <source>Sci. Rep</source>. <volume>14</volume>:<fpage>1774</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-024-52185-2</pub-id><pub-id pub-id-type="pmid">38245656</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Adebiyi</surname> <given-names>A.</given-names></name> <name><surname>Abdalnabi</surname> <given-names>N.</given-names></name> <name><surname>Smith</surname> <given-names>E. H.</given-names></name> <name><surname>Hirner</surname> <given-names>J.</given-names></name> <name><surname>Simoes</surname> <given-names>E. J.</given-names></name> <name><surname>Becevic</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Accurate skin lesion classification using multimodal learning on the ham10000 dataset</article-title>. <source>medRxiv. Preprint</source>. <pub-id pub-id-type="doi">10.1101/2024.05.30.24308213</pub-id><pub-id pub-id-type="pmid">40561935</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ahsan</surname> <given-names>M. M.</given-names></name> <name><surname>Luna</surname> <given-names>S. A.</given-names></name> <name><surname>Siddique</surname> <given-names>Z.</given-names></name></person-group> (<year>2022</year>). <article-title>Machine-learning-based disease diagnosis: a comprehensive review</article-title>. <source>Healthcare</source> <volume>10</volume>:<fpage>541</fpage>. <pub-id pub-id-type="doi">10.3390/healthcare10030541</pub-id><pub-id pub-id-type="pmid">35327018</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Amann</surname> <given-names>J.</given-names></name> <name><surname>Blasimme</surname> <given-names>A.</given-names></name> <name><surname>Vayena</surname> <given-names>E.</given-names></name> <name><surname>Frey</surname> <given-names>D.</given-names></name> <name><surname>Madai</surname> <given-names>V. I.</given-names></name> <name><surname>consortium</surname> <given-names>P.</given-names></name></person-group> (<year>2020</year>). <article-title>Explainability for artificial intelligence in healthcare: a multidisciplinary perspective</article-title>. <source>BMC Med. Inform. Decis. Mak</source>. <volume>20</volume>:<fpage>310</fpage>. <pub-id pub-id-type="doi">10.1186/s12911-020-01332-6</pub-id><pub-id pub-id-type="pmid">33256715</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Arevalo</surname> <given-names>J.</given-names></name> <name><surname>Solorio</surname> <given-names>T.</given-names></name> <name><surname>Montes-y G&#x000F3;mez</surname> <given-names>M.</given-names></name> <name><surname>Gonz&#x000E1;lez</surname> <given-names>F. A.</given-names></name></person-group> (<year>2020</year>). <article-title>Gated multimodal networks</article-title>. <source>Neural Comput. Appl</source>. <volume>32</volume>, <fpage>10209</fpage>&#x02013;<lpage>10228</lpage>. <pub-id pub-id-type="doi">10.1007/s00521-019-04559-1</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Arshad</surname> <given-names>M.</given-names></name> <name><surname>Khan</surname> <given-names>M. A.</given-names></name> <name><surname>Tariq</surname> <given-names>U.</given-names></name> <name><surname>Armghan</surname> <given-names>A.</given-names></name> <name><surname>Alenezi</surname> <given-names>F.</given-names></name> <name><surname>Younus Javed</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>A computer-aided diagnosis system using deep learning for multiclass skin lesion classification</article-title>. <source>Comput. Intell. Neurosci</source>. <volume>2021</volume>:<fpage>9619079</fpage>. <pub-id pub-id-type="doi">10.1155/2021/9619079</pub-id><pub-id pub-id-type="pmid">34912449</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barnard</surname> <given-names>K.</given-names></name> <name><surname>Cardei</surname> <given-names>V.</given-names></name> <name><surname>Funt</surname> <given-names>B.</given-names></name></person-group> (<year>2002</year>). <article-title>A comparison of computational color constancy algorithms. I: Methodology and experiments with synthesized data</article-title>. <source>IEEE Trans. Image Proc</source>. <volume>11</volume>, <fpage>972</fpage>&#x02013;<lpage>984</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2002.802531</pub-id><pub-id pub-id-type="pmid">18249720</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bianco</surname> <given-names>S.</given-names></name> <name><surname>Cusano</surname> <given-names>C.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Quasi-unsupervised color constancy,&#x0201D;</article-title> in <source>2019 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</source>, 12204&#x02013;12213. <pub-id pub-id-type="doi">10.1109/CVPR.2019.01249</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Datta</surname> <given-names>S. K.</given-names></name> <name><surname>Shaikh</surname> <given-names>M. A.</given-names></name> <name><surname>Srihari</surname> <given-names>S. N.</given-names></name> <name><surname>Gao</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Soft-attention improves skin cancer classification performance,&#x0201D;</article-title> in <source>International Workshop on Interpretability of Machine Intelligence in Medical Image Computing</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>13</fpage>&#x02013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-87444-5_2</pub-id><pub-id pub-id-type="pmid">39795627</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dinnes</surname> <given-names>J.</given-names></name> <name><surname>Deeks</surname> <given-names>J. J.</given-names></name> <name><surname>Grainge</surname> <given-names>M. J.</given-names></name> <name><surname>Chuchu</surname> <given-names>N.</given-names></name> <name><surname>Ferrante di Ruffano</surname> <given-names>L.</given-names></name> <name><surname>Matin</surname> <given-names>R. N.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Visual inspection for diagnosing cutaneous melanoma in adults</article-title>. <source>Cochr. Datab. System. Rev</source>. <volume>12</volume>:<fpage>CD013194</fpage>. <pub-id pub-id-type="doi">10.1002/14651858.CD013194</pub-id><pub-id pub-id-type="pmid">30521684</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Esteva</surname> <given-names>A.</given-names></name> <name><surname>Kuprel</surname> <given-names>B.</given-names></name> <name><surname>Novoa</surname> <given-names>R. A.</given-names></name> <name><surname>Ko</surname> <given-names>J.</given-names></name> <name><surname>Swetter</surname> <given-names>S. M.</given-names></name> <name><surname>Blau</surname> <given-names>H. M.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Dermatologist-level classification of skin cancer with deep neural networks</article-title>. <source>Nature</source> <volume>542</volume>, <fpage>115</fpage>&#x02013;<lpage>118</lpage>. <pub-id pub-id-type="doi">10.1038/nature21056</pub-id><pub-id pub-id-type="pmid">28117445</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Frid-Adar</surname> <given-names>M.</given-names></name> <name><surname>Diamant</surname> <given-names>I.</given-names></name> <name><surname>Klang</surname> <given-names>E.</given-names></name> <name><surname>Amitai</surname> <given-names>M.</given-names></name> <name><surname>Goldberger</surname> <given-names>J.</given-names></name> <name><surname>Greenspan</surname> <given-names>H.</given-names></name></person-group> (<year>2018</year>). <article-title>Gan-based synthetic medical image augmentation for increased cnn performance in liver lesion classification</article-title>. <source>Neurocomputing</source> <volume>321</volume>, <fpage>321</fpage>&#x02013;<lpage>331</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2018.09.013</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fukui</surname> <given-names>A.</given-names></name> <name><surname>Park</surname> <given-names>D. H.</given-names></name> <name><surname>Yang</surname> <given-names>D.</given-names></name> <name><surname>Rohrbach</surname> <given-names>A.</given-names></name> <name><surname>Darrell</surname> <given-names>T.</given-names></name> <name><surname>Rohrbach</surname> <given-names>M.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Multimodal compact bilinear pooling for visual question answering and visual grounding,&#x0201D;</article-title> in <source>Proceedings of the 2016 Conference on Empirical Methods in Natural Language Processing</source>, eds. J. Su, K. Duh, X. Carreras (Austin, TX: Association for Computational Linguistics), <fpage>457</fpage>&#x02013;<lpage>468</lpage>. <pub-id pub-id-type="doi">10.18653/v1/D16-1044</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Garbe</surname> <given-names>C.</given-names></name> <name><surname>Keim</surname> <given-names>U.</given-names></name> <name><surname>Gandini</surname> <given-names>S.</given-names></name> <name><surname>Amaral</surname> <given-names>T.</given-names></name> <name><surname>Katalinic</surname> <given-names>A.</given-names></name> <name><surname>Hollezcek</surname> <given-names>B.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Epidemiology of cutaneous melanoma and keratinocyte cancer in white populations 1943&#x02013;2036</article-title>. <source>Eur. J. Cancer</source> <volume>152</volume>, <fpage>18</fpage>&#x02013;<lpage>25</lpage>. <pub-id pub-id-type="doi">10.1016/j.ejca.2021.04.029</pub-id><pub-id pub-id-type="pmid">34062483</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gessert</surname> <given-names>N.</given-names></name> <name><surname>Nielsen</surname> <given-names>M.</given-names></name> <name><surname>Shaikh</surname> <given-names>M.</given-names></name> <name><surname>Werner</surname> <given-names>R.</given-names></name> <name><surname>Schlaefer</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>Skin lesion classification using ensembles of multi-resolution efficientNets with meta data</article-title>. <source>MethodsX</source> <volume>7</volume>:<fpage>100864</fpage>. <pub-id pub-id-type="doi">10.1016/j.mex.2020.100864</pub-id><pub-id pub-id-type="pmid">32292713</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Haenssle</surname> <given-names>H.</given-names></name> <name><surname>Fink</surname> <given-names>C.</given-names></name> <name><surname>Schneiderbauer</surname> <given-names>R.</given-names></name> <name><surname>Toberer</surname> <given-names>F.</given-names></name> <name><surname>Buhl</surname> <given-names>T.</given-names></name> <name><surname>Blum</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Man against machine: diagnostic performance of a deep learning convolutional neural network for dermoscopic melanoma recognition in comparison to 58 dermatologists</article-title>. <source>Ann. Oncol</source>. <volume>29</volume>, <fpage>1836</fpage>&#x02013;<lpage>1842</lpage>. <pub-id pub-id-type="doi">10.1093/annonc/mdy166</pub-id><pub-id pub-id-type="pmid">29846502</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han</surname> <given-names>S. S.</given-names></name> <name><surname>Kim</surname> <given-names>M. S.</given-names></name> <name><surname>Lim</surname> <given-names>W.</given-names></name> <name><surname>Park</surname> <given-names>G. H.</given-names></name> <name><surname>Park</surname> <given-names>I.</given-names></name> <name><surname>Chang</surname> <given-names>S. E.</given-names></name></person-group> (<year>2018</year>). <article-title>Classification of the clinical images for benign and malignant cutaneous tumors using a deep learning algorithm</article-title>. <source>J. Invest. Dermatol</source>. <volume>138</volume>, <fpage>1529</fpage>&#x02013;<lpage>1538</lpage>. <pub-id pub-id-type="doi">10.1016/j.jid.2018.01.028</pub-id><pub-id pub-id-type="pmid">29428356</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han</surname> <given-names>T. Y.</given-names></name> <name><surname>Chang</surname> <given-names>H. S.</given-names></name> <name><surname>Lee</surname> <given-names>J. H.</given-names></name> <name><surname>Lee</surname> <given-names>W. M.</given-names></name> <name><surname>Son</surname> <given-names>S. J.</given-names></name></person-group> (<year>2011</year>). <article-title>A clinical and histopathological study of 122 cases of dermatofibroma (benign fibrous histiocytoma)</article-title>. <source>Ann. Dermatol</source>. <volume>23</volume>, <fpage>185</fpage>&#x02013;<lpage>192</lpage>. <pub-id pub-id-type="doi">10.5021/ad.2011.23.2.185</pub-id><pub-id pub-id-type="pmid">21747617</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hasan</surname> <given-names>N.</given-names></name> <name><surname>Nadaf</surname> <given-names>A.</given-names></name> <name><surname>Imran</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Skin cancer: understanding the journey of transformation from conventional to advanced treatment approaches</article-title>. <source>Mol. Cancer</source> <volume>22</volume>:<fpage>168</fpage>. <pub-id pub-id-type="doi">10.1186/s12943-023-01854-3</pub-id><pub-id pub-id-type="pmid">37803407</pub-id></citation></ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Deep residual learning for image recognition,&#x0201D;</article-title> in <source>2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</source>, 770&#x02013;778. <pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>James</surname> <given-names>W. D.</given-names></name> <name><surname>Berger</surname> <given-names>T. G.</given-names></name> <name><surname>Elston</surname> <given-names>D. M.</given-names></name></person-group> (<year>2006</year>). <source>Andrews&#x00027; Diseases of the Skin: Clinical Dermatology</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Saunders Elsevier, 12th edition</publisher-name>.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jaworek-Korjakowska</surname> <given-names>J.</given-names></name> <name><surname>Brodzicki</surname> <given-names>A.</given-names></name> <name><surname>Cassidy</surname> <given-names>B.</given-names></name> <name><surname>Kendrick</surname> <given-names>C.</given-names></name> <name><surname>Yap</surname> <given-names>M. H.</given-names></name></person-group> (<year>2021</year>). <article-title>Interpretability of a deep learning based approach for the classification of skin lesions into main anatomic body sites</article-title>. <source>Cancers</source> <volume>13</volume>:<fpage>6048</fpage>. <pub-id pub-id-type="doi">10.3390/cancers13236048</pub-id><pub-id pub-id-type="pmid">34885158</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jerant</surname> <given-names>A. F.</given-names></name> <name><surname>Johnson</surname> <given-names>J. T.</given-names></name> <name><surname>Sheridan</surname> <given-names>C. D.</given-names></name> <name><surname>Caffrey</surname> <given-names>T. J.</given-names></name></person-group> (<year>2000</year>). <article-title>Early detection and treatment of skin cancer</article-title>. <source>Am Fam Phys</source>. <volume>62</volume>, <fpage>357</fpage>&#x02013;<lpage>382</lpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>M.</given-names></name> <name><surname>Lei</surname> <given-names>S.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Hou</surname> <given-names>L.</given-names></name> <name><surname>Meixiang</surname> <given-names>Z.</given-names></name> <name><surname>Luo</surname> <given-names>Y.</given-names></name></person-group> (<year>2022</year>). <article-title>Multimodal imaging of target detection algorithm under artificial intelligence in the diagnosis of early breast cancer</article-title>. <source>J. Healthc. Eng</source>. <volume>2022</volume>, <fpage>1</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1155/2022/9322937</pub-id><pub-id pub-id-type="pmid">35047160</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kavita</surname> <given-names>A.</given-names></name> <name><surname>Thakur</surname> <given-names>J.</given-names></name> <name><surname>Narang</surname> <given-names>T.</given-names></name></person-group> (<year>2023</year>). <article-title>The burden of skin diseases in india: Global burden of disease study 2017</article-title>. <source>Indian J. Dermatol. Venereol. Leprol</source>. <volume>89</volume>, <fpage>421</fpage>&#x02013;<lpage>425</lpage>. <pub-id pub-id-type="doi">10.25259/IJDVL_978_20</pub-id><pub-id pub-id-type="pmid">34877854</pub-id></citation></ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kiela</surname> <given-names>D.</given-names></name> <name><surname>Conneau</surname> <given-names>A.</given-names></name> <name><surname>Jabri</surname> <given-names>A.</given-names></name> <name><surname>Nickel</surname> <given-names>M.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Learning visually grounded sentence representations,&#x0201D;</article-title> in <source>Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long Papers)</source>, eds. M. Walker, H. Ji, A. Stent (New Orleans, Louisiana: Association for Computational Linguistics), <fpage>408</fpage>&#x02013;<lpage>418</lpage>. <pub-id pub-id-type="doi">10.18653/v1/N18-1038</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>J.-H.</given-names></name> <name><surname>On</surname> <given-names>K.-W.</given-names></name> <name><surname>Lim</surname> <given-names>W.</given-names></name> <name><surname>Kim</surname> <given-names>J.</given-names></name> <name><surname>Ha</surname> <given-names>J.-W.</given-names></name> <name><surname>Zhang</surname> <given-names>B.-T.</given-names></name></person-group> (<year>2017</year>). <article-title>Hadamard product for low-rank bilinear pooling</article-title>. <source>arXiv preprint arXiv:1610.04325</source>.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kirillov</surname> <given-names>A.</given-names></name> <name><surname>Mintun</surname> <given-names>E.</given-names></name> <name><surname>Ravi</surname> <given-names>N.</given-names></name> <name><surname>Mao</surname> <given-names>H.</given-names></name> <name><surname>Rolland</surname> <given-names>C.</given-names></name> <name><surname>Gustafson</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>&#x0201C;Segment anything,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF International Conference on Computer Vision</source>, 4015&#x02013;4026. <pub-id pub-id-type="doi">10.1109/ICCV51070.2023.00371</pub-id></citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kumar</surname> <given-names>S.</given-names></name> <name><surname>Gupta</surname> <given-names>S. K.</given-names></name> <name><surname>Kumar</surname> <given-names>V.</given-names></name> <name><surname>Kumar</surname> <given-names>M.</given-names></name> <name><surname>Chaube</surname> <given-names>M. K.</given-names></name> <name><surname>Naik</surname> <given-names>N. S.</given-names></name></person-group> (<year>2022</year>). <article-title>Ensemble multimodal deep learning for early diagnosis and accurate classification of covid-19</article-title>. <source>Comput. Electr. Eng</source>. <volume>103</volume>:<fpage>108396</fpage>. <pub-id pub-id-type="doi">10.1016/j.compeleceng.2022.108396</pub-id><pub-id pub-id-type="pmid">36160764</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lan</surname> <given-names>Z.</given-names></name> <name><surname>Cai</surname> <given-names>S.</given-names></name> <name><surname>He</surname> <given-names>X.</given-names></name> <name><surname>Wen</surname> <given-names>X.</given-names></name></person-group> (<year>2022</year>). <article-title>Fixcaps: An improved capsules network for diagnosis of skin cancer</article-title>. <source>IEEE Access</source> <volume>10</volume>, <fpage>76261</fpage>&#x02013;<lpage>76267</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2022.3181225</pub-id></citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>T.-Y.</given-names></name> <name><surname>Goyal</surname> <given-names>P.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Doll&#x000E1;r</surname> <given-names>P.</given-names></name></person-group> (<year>2020</year>). <article-title>Focal loss for dense object detection</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>42</volume>, <fpage>318</fpage>&#x02013;<lpage>327</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2018.2858826</pub-id><pub-id pub-id-type="pmid">30040631</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Loescher</surname> <given-names>L. J.</given-names></name> <name><surname>Janda</surname> <given-names>M.</given-names></name> <name><surname>Soyer</surname> <given-names>H. P.</given-names></name> <name><surname>Shea</surname> <given-names>K.</given-names></name> <name><surname>Curiel-Lewandrowski</surname> <given-names>C.</given-names></name></person-group> (<year>2013</year>). <article-title>Advances in skin cancer early detection and diagnosis</article-title>. <source>Semin. Oncol. Nurs</source>. <volume>29</volume>, <fpage>170</fpage>&#x02013;<lpage>181</lpage>. <pub-id pub-id-type="doi">10.1016/j.soncn.2013.06.003</pub-id><pub-id pub-id-type="pmid">23958215</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lu</surname> <given-names>J.</given-names></name> <name><surname>Batra</surname> <given-names>D.</given-names></name> <name><surname>Parikh</surname> <given-names>D.</given-names></name> <name><surname>Lee</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Vilbert: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks,&#x0201D;</article-title> in <source>Proceedings of the 33rd International Conference on Neural Information Processing Systems</source> (<publisher-loc>Red Hook, NY, USA</publisher-loc>: <publisher-name>Curran Associates Inc.</publisher-name>).</citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Marzuka</surname> <given-names>A. G.</given-names></name> <name><surname>Book</surname> <given-names>S. E.</given-names></name></person-group> (<year>2015</year>). <article-title>Basal cell carcinoma: pathogenesis, epidemiology, clinical features, diagnosis, histopathology, and management</article-title>. <source>Yale J. Biol. Med</source>. <volume>88</volume>, <fpage>167</fpage>&#x02013;<lpage>179</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mulligan</surname> <given-names>P. R.</given-names></name> <name><surname>Prajapati</surname> <given-names>H. J.</given-names></name> <name><surname>Martin</surname> <given-names>L. G.</given-names></name> <name><surname>Patel</surname> <given-names>T. H.</given-names></name></person-group> (<year>2014</year>). <article-title>Vascular anomalies: classification, imaging characteristics and implications for interventional radiology treatment approaches</article-title>. <source>Br. J. Radiol</source>. <volume>87</volume>:<fpage>20130392</fpage>. <pub-id pub-id-type="doi">10.1259/bjr.20130392</pub-id><pub-id pub-id-type="pmid">24588666</pub-id></citation></ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mutepfe</surname> <given-names>F.</given-names></name> <name><surname>Kalejahi</surname> <given-names>B. K.</given-names></name> <name><surname>Meshgini</surname> <given-names>S.</given-names></name> <name><surname>Danishvar</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>Generative adversarial network image synthesis method for skin lesion generation and classification</article-title>. <source>J. Med. Signals Sens</source>. <volume>11</volume>, <fpage>237</fpage>&#x02013;<lpage>252</lpage>. <pub-id pub-id-type="doi">10.4103/jmss.JMSS_53_20</pub-id><pub-id pub-id-type="pmid">34820296</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Myers</surname> <given-names>D. J.</given-names></name> <name><surname>Fillman</surname> <given-names>E. P.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;Dermatofibroma,&#x0201D;</article-title> in StatPearls [Internet]. Treasure Island (FL): StatPearls Publishing.</citation>
</ref>
<ref id="B38">
<citation citation-type="web"><person-group person-group-type="author"><collab>National Cancer Institute</collab></person-group> (<year>2024a</year>). <source>Melanoma treatment (pdq<sup>&#x000AE;</sup>)-health professional version</source>. National Cancer Institute. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.cancer.gov/types/skin/hp/melanoma-treatment-pdq">https://www.cancer.gov/types/skin/hp/melanoma-treatment-pdq</ext-link> (Accessed January 9, 2025).</citation>
</ref>
<ref id="B39">
<citation citation-type="web"><person-group person-group-type="author"><collab>National Cancer Institute</collab></person-group> (<year>2024b</year>). <source>Skin cancer treatment (pdq<sup>&#x000AE;</sup>)-health professional version</source>. National Cancer Institute. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.cancer.gov/types/skin/hp/skin-treatment-pdq&#x00023;_177_toc">https://www.cancer.gov/types/skin/hp/skin-treatment-pdq&#x00023;_177_toc</ext-link> (Accessed January 9, 2025).</citation>
</ref>
<ref id="B40">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ngiam</surname> <given-names>J.</given-names></name> <name><surname>Khosla</surname> <given-names>A.</given-names></name> <name><surname>Kim</surname> <given-names>M.</given-names></name> <name><surname>Nam</surname> <given-names>J.</given-names></name> <name><surname>Lee</surname> <given-names>H.</given-names></name> <name><surname>Ng</surname> <given-names>A. Y.</given-names></name></person-group> (<year>2011</year>). <article-title>&#x0201C;Multimodal deep learning,&#x0201D;</article-title> in <source>Proceedings of the 28th International Conference on International Conference on Machine Learning, ICML&#x00027;11</source> (<publisher-loc>Madison, WI, USA</publisher-loc>: <publisher-name>Omnipress</publisher-name>), <fpage>689</fpage>&#x02013;<lpage>696</lpage>.</citation>
</ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ou</surname> <given-names>C.</given-names></name> <name><surname>Zhou</surname> <given-names>S.</given-names></name> <name><surname>Yang</surname> <given-names>R.</given-names></name> <name><surname>Jiang</surname> <given-names>W.</given-names></name> <name><surname>He</surname> <given-names>H.</given-names></name> <name><surname>Gan</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>A deep learning based multimodal fusion model for skin lesion diagnosis using smartphone collected clinical images and metadata</article-title>. <source>Front. Surg</source>. <volume>9</volume>:<fpage>1029991</fpage>. <pub-id pub-id-type="doi">10.3389/fsurg.2022.1029991</pub-id><pub-id pub-id-type="pmid">36268206</pub-id></citation></ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Puckett</surname> <given-names>Y.</given-names></name> <name><surname>Wilson</surname> <given-names>A. M.</given-names></name> <name><surname>Farci</surname> <given-names>F.</given-names></name> <name><surname>Thevenin</surname> <given-names>C.</given-names></name></person-group> (<year>2025</year>). <source>Melanoma Pathology</source>. StatPearls [Internet]. StatPearls Publishing, Treasure Island (FL).</citation>
</ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Raimondi</surname> <given-names>S.</given-names></name> <name><surname>Suppa</surname> <given-names>M.</given-names></name> <name><surname>Gandini</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>Melanoma epidemiology and sun exposure</article-title>. <source>Acta Dermato-Venereol</source>. <volume>100</volume>:<fpage>adv00136</fpage>. <pub-id pub-id-type="doi">10.2340/00015555-3491</pub-id><pub-id pub-id-type="pmid">32346751</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rajput</surname> <given-names>G.</given-names></name> <name><surname>Agrawal</surname> <given-names>S.</given-names></name> <name><surname>Raut</surname> <given-names>G.</given-names></name> <name><surname>Vishvakarma</surname> <given-names>S. K.</given-names></name></person-group> (<year>2021</year>). <article-title>An accurate and noninvasive skin cancer screening based on imaging technique</article-title>. <source>Int. J. Imaging Syst. Technol</source>. <volume>32</volume>, <fpage>354</fpage>&#x02013;<lpage>368</lpage>. <pub-id pub-id-type="doi">10.1002/ima.22616</pub-id></citation>
</ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Reinehr</surname> <given-names>C. P.</given-names></name> <name><surname>Bakos</surname> <given-names>R. M.</given-names></name></person-group> (<year>2019</year>). <article-title>Actinic keratoses: review of clinical, dermoscopic, and therapeutic aspects</article-title>. <source>An. Bras. Dermatol</source>. <volume>94</volume>, <fpage>637</fpage>&#x02013;<lpage>657</lpage>. <pub-id pub-id-type="doi">10.1016/j.abd.2019.10.004</pub-id><pub-id pub-id-type="pmid">31789244</pub-id></citation></ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Restrepo</surname> <given-names>D.</given-names></name> <name><surname>Wu</surname> <given-names>C.</given-names></name> <name><surname>Cajas</surname> <given-names>S. A.</given-names></name> <name><surname>Nakayama</surname> <given-names>L. F.</given-names></name> <name><surname>Celi</surname> <given-names>L. A.</given-names></name> <name><surname>L&#x000F3;pez</surname> <given-names>D. M.</given-names></name></person-group> (<year>2024</year>). <article-title>Multimodal deep learning for low-resource settings: a vector embedding alignment approach for healthcare applications</article-title>. <source>medRxiv</source>. <pub-id pub-id-type="doi">10.1101/2024.06.03.24308401</pub-id></citation>
</ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rogers</surname> <given-names>H. W.</given-names></name> <name><surname>Weinstock</surname> <given-names>M. A.</given-names></name> <name><surname>Feldman</surname> <given-names>S. R.</given-names></name> <name><surname>Coldiron</surname> <given-names>B. M.</given-names></name></person-group> (<year>2015</year>). <article-title>Incidence estimate of nonmelanoma skin cancer (keratinocyte carcinomas) in the us population, 2012</article-title>. <source>JAMA Dermatol</source>. <volume>151</volume>, <fpage>1081</fpage>&#x02013;<lpage>1086</lpage>. <pub-id pub-id-type="doi">10.1001/jamadermatol.2015.1187</pub-id><pub-id pub-id-type="pmid">25928283</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rumelhart</surname> <given-names>D. E.</given-names></name> <name><surname>Hinton</surname> <given-names>G. E.</given-names></name> <name><surname>Williams</surname> <given-names>R. J.</given-names></name></person-group> (<year>1986</year>). <article-title>Learning representations by back-propagating errors</article-title>. <source>Nature</source> <volume>323</volume>, <fpage>533</fpage>&#x02013;<lpage>536</lpage>. <pub-id pub-id-type="doi">10.1038/323533a0</pub-id></citation>
</ref>
<ref id="B49">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Scott Mader</surname> <given-names>K.</given-names></name></person-group> (<year>2018</year>). <source>Skin cancer mnist: Ham10000. Kaggle</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com/datasets/kmader/skin-cancer-mnist-ham10000/">https://www.kaggle.com/datasets/kmader/skin-cancer-mnist-ham10000/</ext-link> (Accessed January 9, 2025).<pub-id pub-id-type="pmid">39891245</pub-id></citation></ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Scott</surname> <given-names>R.</given-names></name> <name><surname>Oakley</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>Benign keratosis: a useful term?</article-title> <source>Dermatol. Pract. Concept</source>. <volume>13</volume>:<fpage>e2023115</fpage>. <pub-id pub-id-type="doi">10.5826/dpc.1302a115</pub-id><pub-id pub-id-type="pmid">37196289</pub-id></citation></ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Selvaraju</surname> <given-names>R. R.</given-names></name> <name><surname>Cogswell</surname> <given-names>M.</given-names></name> <name><surname>Das</surname> <given-names>A.</given-names></name> <name><surname>Vedantam</surname> <given-names>R.</given-names></name> <name><surname>Parikh</surname> <given-names>D.</given-names></name> <name><surname>Batra</surname> <given-names>D.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Grad-cam: visual explanations from deep networks via gradient-based localization,&#x0201D;</article-title> in <source>2017 IEEE International Conference on Computer Vision (ICCV)</source>, 618&#x02013;626. <pub-id pub-id-type="doi">10.1109/ICCV.2017.74</pub-id></citation>
</ref>
<ref id="B52">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Seretis</surname> <given-names>K.</given-names></name> <name><surname>Bounas</surname> <given-names>N.</given-names></name> <name><surname>Rapti</surname> <given-names>E.</given-names></name> <name><surname>Lampri</surname> <given-names>E.</given-names></name> <name><surname>Moschovos</surname> <given-names>V.</given-names></name> <name><surname>Lykoudis</surname> <given-names>E. G.</given-names></name></person-group> (<year>2025</year>). <article-title>Basal cell carcinoma in patients over 80 years presenting for surgical excision: Clinical characteristics and surgical outcomes</article-title>. <source>Curr. Oncol</source>. <volume>32</volume>:<fpage>120</fpage>. <pub-id pub-id-type="doi">10.3390/curroncol32030120</pub-id><pub-id pub-id-type="pmid">40136323</pub-id></citation></ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sevli</surname> <given-names>O.</given-names></name></person-group> (<year>2021</year>). <article-title>A deep convolutional neural network-based pigmented skin lesion classification application and experts evaluation</article-title>. <source>Neural Comput. Applic</source>. <volume>33</volume>, <fpage>12039</fpage>&#x02013;<lpage>12050</lpage>. <pub-id pub-id-type="doi">10.1007/s00521-021-05929-4</pub-id></citation>
</ref>
<ref id="B54">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Simonyan</surname> <given-names>K.</given-names></name> <name><surname>Zisserman</surname> <given-names>A.</given-names></name></person-group> (<year>2015</year>). <article-title>Very deep convolutional networks for large-scale image recognition</article-title>. <source>arXiv preprint arXiv:1409.1556</source>.</citation>
</ref>
<ref id="B55">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Srivastava</surname> <given-names>V.</given-names></name> <name><surname>Kumar</surname> <given-names>D.</given-names></name> <name><surname>Roy</surname> <given-names>S.</given-names></name></person-group> (<year>2022</year>). <article-title>A median based quadrilateral local quantized ternary pattern technique for the classification of dermatoscopic images of skin cancer</article-title>. <source>Comput. Electr. Eng</source>. <volume>102</volume>:<fpage>108259</fpage>. <pub-id pub-id-type="doi">10.1016/j.compeleceng.2022.108259</pub-id></citation>
</ref>
<ref id="B56">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Steiner</surname> <given-names>J. E.</given-names></name> <name><surname>Drolet</surname> <given-names>B. A.</given-names></name></person-group> (<year>2017</year>). <article-title>Classification of vascular anomalies: an update</article-title>. <source>Semin. Intervent. Radiol</source>. <volume>34</volume>, <fpage>225</fpage>&#x02013;<lpage>232</lpage>. <pub-id pub-id-type="doi">10.1055/s-0037-1604295</pub-id><pub-id pub-id-type="pmid">28955111</pub-id></citation></ref>
<ref id="B57">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Sundararajan</surname> <given-names>M.</given-names></name> <name><surname>Taly</surname> <given-names>A.</given-names></name> <name><surname>Yan</surname> <given-names>Q.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Axiomatic attribution for deep networks,&#x0201D;</article-title> in <source>Proceedings of the 34th International Conference on Machine Learning</source> - <italic>Volume 70, ICML&#x00027;17</italic> (<ext-link ext-link-type="uri" xlink:href="https://JMLR.org">JMLR.org</ext-link>), <fpage>3319</fpage>&#x02013;<lpage>3328</lpage>.</citation>
</ref>
<ref id="B58">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tschandl</surname> <given-names>P.</given-names></name> <name><surname>Codella</surname> <given-names>N.</given-names></name> <name><surname>Akay</surname> <given-names>B. N.</given-names></name> <name><surname>Argenziano</surname> <given-names>G.</given-names></name> <name><surname>Braun</surname> <given-names>R. P.</given-names></name> <name><surname>Cabo</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Comparison of the accuracy of human readers versus machine-learning algorithms for pigmented skin lesion classification: an open, web-based, international, diagnostic study</article-title>. <source>Lancet Oncol</source>. <volume>20</volume>, <fpage>938</fpage>&#x02013;<lpage>947</lpage>. <pub-id pub-id-type="doi">10.1016/S1470-2045(19)30333-X</pub-id><pub-id pub-id-type="pmid">31201137</pub-id></citation></ref>
<ref id="B59">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tschandl</surname> <given-names>P.</given-names></name> <name><surname>Rosendahl</surname> <given-names>C.</given-names></name> <name><surname>Kittler</surname> <given-names>H.</given-names></name></person-group> (<year>2018</year>). <article-title>The HAM10000 dataset, a large collection of multi-source dermatoscopic images of common pigmented skin lesions</article-title>. <source>Sci. Data</source> <volume>5</volume>:<fpage>180161</fpage>. <pub-id pub-id-type="doi">10.1038/sdata.2018.161</pub-id><pub-id pub-id-type="pmid">30106392</pub-id></citation></ref>
<ref id="B60">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaswani</surname> <given-names>A.</given-names></name> <name><surname>Shazeer</surname> <given-names>N.</given-names></name> <name><surname>Parmar</surname> <given-names>N.</given-names></name> <name><surname>Uszkoreit</surname> <given-names>J.</given-names></name> <name><surname>Jones</surname> <given-names>L.</given-names></name> <name><surname>Gomez</surname> <given-names>A. N.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>&#x0201C;Attention is all you need,&#x0201D;</article-title> in <source>Proceedings of NeurIPS 2017</source>.</citation>
</ref>
<ref id="B61">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>G.</given-names></name> <name><surname>Yan</surname> <given-names>P.</given-names></name> <name><surname>Tang</surname> <given-names>Q.</given-names></name> <name><surname>Yang</surname> <given-names>L.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name></person-group> (<year>2023</year>). <article-title>Multiscale feature fusion for skin lesion classification</article-title>. <source>Biomed Res. Int</source>. <volume>2023</volume>:<fpage>5146543</fpage>. <pub-id pub-id-type="doi">10.1155/2023/5146543</pub-id><pub-id pub-id-type="pmid">36644161</pub-id></citation></ref>
<ref id="B62">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>G.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Wu</surname> <given-names>X.</given-names></name> <name><surname>Leng</surname> <given-names>X.</given-names></name> <name><surname>Xu</surname> <given-names>Y.</given-names></name></person-group> (<year>2024</year>). <article-title>Development of skip connection in deep neural networks for computer vision and medical image analysis: A survey</article-title>. <source>ArXiv, abs/2405.01725</source>.</citation>
</ref>
<ref id="B63">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zadeh</surname> <given-names>A.</given-names></name> <name><surname>Chen</surname> <given-names>M.</given-names></name> <name><surname>Poria</surname> <given-names>S.</given-names></name> <name><surname>Cambria</surname> <given-names>E.</given-names></name> <name><surname>Morency</surname> <given-names>L.-P.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Tensor fusion network for multimodal sentiment analysis,&#x0201D;</article-title> in <source>Proceedings of the 2017 Conference on Empirical Methods in Natural Language Processing</source>, eds. M. Palmer, R. Hwa, and S. Riedel (Copenhagen, Denmark: Association for Computational Linguistics), <fpage>1103</fpage>&#x02013;<lpage>1114</lpage>. <pub-id pub-id-type="doi">10.18653/v1/D17-1115</pub-id><pub-id pub-id-type="pmid">35111209</pub-id></citation></ref>
<ref id="B64">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zech</surname> <given-names>J. R.</given-names></name> <name><surname>Badgeley</surname> <given-names>M. A.</given-names></name> <name><surname>Liu</surname> <given-names>M.</given-names></name> <name><surname>Costa</surname> <given-names>A. B.</given-names></name> <name><surname>Titano</surname> <given-names>J. J.</given-names></name> <name><surname>Oermann</surname> <given-names>E. K.</given-names></name></person-group> (<year>2018</year>). <article-title>Variable generalization performance of a deep learning model to detect pneumonia in chest radiographs: a cross-sectional study</article-title>. <source>PLoS Med</source>. <volume>15</volume>:<fpage>e1002683</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pmed.1002683</pub-id><pub-id pub-id-type="pmid">30399157</pub-id></citation></ref>
</ref-list>
</back>
</article>