<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1646176</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Enhancing accessibility: a multi-level platform for visual question answering in diabetic retinopathy for individuals with disabilities</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Alotaibi</surname> <given-names>Sarah</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3100582/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Al-Hadhrami</surname> <given-names>Suheer</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3096974/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Al-Ahmadi</surname> <given-names>Saad</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2416077/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Computer Science, College of Computer and Information Sciences, King Saud University</institution>, <addr-line>Riyadh</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff2"><sup>2</sup><institution>Computer Engineering Department, College of Engineering, Hadhramout University</institution>, <addr-line>Al Mukalla</addr-line>, <country>Yemen</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1591406/overview">Hongying Liu</ext-link>, Tianjin University, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2195581/overview">Shi Xue Dai</ext-link>, Guangdong Provincial People&#x00027;s Hospital, China</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2942596/overview">Vijayarajan V.</ext-link>, VIT University, India</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Saad Al-Ahmadi <email>salahmadi&#x00040;ksu.edu.sa</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>11</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1646176</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>10</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Alotaibi, Al-Hadhrami and Al-Ahmadi.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Alotaibi, Al-Hadhrami and Al-Ahmadi</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Individuals with visual disabilities possess impairments that affect their ability to perceive visual information, ranging from partial to complete vision loss. Visual disabilities affect about 2.2 billion people globally. In this paper, we introduce a new multi-level Visual Questioning Answering (VQA) framework for visually disabled people that leverages the strengths of various VQA models of the multi-level components to enhance system performance. The model relies on a bi-level architecture that employs two distinct layers. In the first level, the model classifies the question type. This classification guides the visual question to the appropriate component model in the second level. This bi-level architecture incorporates a switch function that enables the system to select the optimal VQA model for each specific question, hence enhancing overall accuracy. The experimental findings indicate that the multi-level VQA technique is significantly effective. The bi-level VQA model enhances the overall accuracy over the state-of-the-art from 87.41% to 88.41%. This finding suggests the use of multiple levels with different models can boost the VQA systems&#x00027; performance. This research presents a promising direction for developing advanced, multi-level VQA systems. Future work may explore optimizing and experimenting with various model levels to enhance performance further.</p></abstract>
<kwd-group>
<kwd>disability-aware VQA</kwd>
<kwd>ELECTRA</kwd>
<kwd>Med-VQA</kwd>
<kwd>medical visual question answering</kwd>
<kwd>multi-level VQA</kwd>
<kwd>question answering</kwd>
<kwd>SWIN</kwd>
<kwd>vision-language models</kwd>
</kwd-group>
<counts>
<fig-count count="11"/>
<table-count count="9"/>
<equation-count count="14"/>
<ref-count count="96"/>
<page-count count="22"/>
<word-count count="12749"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Machine Learning and Artificial Intelligence</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Visual disabilities affect millions of people worldwide, posing a major global concern. These impairments severely restrict access to visual information and limit participation in many daily activities (<xref ref-type="bibr" rid="B32">Gurari et al., 2018</xref>). The World Health Organization (WHO) reports that more than 2.2 billion individuals worldwide experience some form of visual impairment or blindness, with many cases arising from preventable or treatable conditions such as diabetic retinopathy (DR) (<xref ref-type="bibr" rid="B88">Organization, 2019</xref>).</p>
<p>The standard practice for diagnosing and assessing DR involves ophthalmologists manually analyzing fundus images to determine disease severity. However, this method is time-intensive, error-prone, and highly subjective. The global shortage of ophthalmologists further exacerbates these challenges (<xref ref-type="bibr" rid="B4">Abr&#x000E0;moff et al., 2018</xref>).</p>
<p>These limitations hinder timely and accurate diagnosis, especially as the prevalence of DR continues to rise. Recent deep learning developments show promise in automating DR detection and grading, offering potential solutions to these challenges. However, these methods face practical limitations, including data scarcity, difficulty generalizing to real-world scenarios, and suboptimal performance in handling complex diagnostic questions (<xref ref-type="bibr" rid="B40">Jagan Mohan et al., 2022</xref>). Advancements in assistive technologies powered by artificial intelligence (AI) promise to transform lives, enhance independence, and elevate the quality of life for these individuals (<xref ref-type="bibr" rid="B22">de Freitas et al., 2022</xref>). Recent deep learning developments To bridge this gap, Visual Question Answering (VQA) has emerged as a promising development capable of extracting meaningful insights by answering user-defined questions based on image content (<xref ref-type="bibr" rid="B53">Lin et al., 2023</xref>). In the medical domain, Medical Visual Question Answering (Med-VQA) has recently become a potential solution (<xref ref-type="bibr" rid="B53">Lin et al., 2023</xref>). Med-VQA combines advancements from Computer Vision (CV) and Natural Language Processing (NLP) to answer complex medical questions using images like fundus images, CT scans, and X-rays (<xref ref-type="bibr" rid="B53">Lin et al., 2023</xref>; <xref ref-type="bibr" rid="B31">Gu et al., 2024</xref>).</p>
<p>Integrating text with image data, Med-VQA offers several advantages. It facilitates expedited and accurate diagnoses for physicians. It also alleviates their workload by delivering immediate responses to common inquiries and provides medical students with a valuable study resource.</p>
<p>Additionally, Med-VQA empowers patients by providing access to information regarding their ailments using straightforward question-and-answer interfaces (<xref ref-type="bibr" rid="B8">Al-Hadhrami et al., 2023</xref>). The Med-VQA area is nascent and has numerous constraints, notably the lack of high-quality labeled data (<xref ref-type="bibr" rid="B31">Gu et al., 2024</xref>). Presently accessible datasets such as VQA-RAD (<xref ref-type="bibr" rid="B96">Zhu et al., 2016</xref>), SLAKE (<xref ref-type="bibr" rid="B54">Liu et al., 2021</xref>), VQA-Med (<xref ref-type="bibr" rid="B2">Abacha et al., 2019</xref>, <xref ref-type="bibr" rid="B1">2020</xref>), and DME (<xref ref-type="bibr" rid="B78">Tascon-Morales et al., 2022</xref>) serve as foundational references. Nonetheless, numerous efforts are inadequate due to insufficient question diversity, limited data volumes, and, in certain instances, the lack of clinical validation, hindering the development of robust and generalizable models.</p>
<p>In the context of available Med-VQA datasets, this study tackles the challenge of the data limitation by fine-tuning models on comprehensive datasets that encompass various types and modalities of questions (<xref ref-type="bibr" rid="B93">Zhang et al., 2023</xref>). By utilizing all available data and focusing on specific question types or classes during the fine-tuning process, the proposed methodology mitigates issues related to model generalization and overfitting. The hierarchical model structure introduced in this research outperforms conventional methods by categorizing visual question types according to image, text, or combined image-text. This distinctive classification methodology that highlights the image and text modalities differs this work from the existing literature and introduces a new direction for improved VQA performance.</p>
<p>In the context of VQA, <xref ref-type="bibr" rid="B8">Al-Hadhrami et al. (2023)</xref> demonstrated that models fine-tuned using various hyperparameters perform best for different question types or response classes. This finding highlights the importance of models being designed for given question classes, the next fundamental step toward the enhancement of the effectiveness and efficiency of the models for VQA. Based on this finding, this work highlights the importance for models being flexible and adaptive for handling multiple question types dynamically, eventually enhancing the performance and delivering more accurate responses for real-world applications.</p>
<p>State-of-the-art (SOTA) methods often incorporate several advanced techniques. These include joint embedding (<xref ref-type="bibr" rid="B70">Ren et al., 2015</xref>; <xref ref-type="bibr" rid="B12">Antol et al., 2015</xref>; <xref ref-type="bibr" rid="B59">Malinowski et al., 2015</xref>), attention mechanisms (<xref ref-type="bibr" rid="B41">Jiang et al., 2015</xref>; <xref ref-type="bibr" rid="B16">Chen et al., 2015</xref>; <xref ref-type="bibr" rid="B38">Ilievski et al., 2016</xref>; <xref ref-type="bibr" rid="B11">Andreas et al., 2016b</xref>; <xref ref-type="bibr" rid="B75">Song et al., 2022</xref>), compositional reasoning (<xref ref-type="bibr" rid="B10">Andreas et al., 2016a</xref>,<xref ref-type="bibr" rid="B11">b</xref>; <xref ref-type="bibr" rid="B90">Xiong et al., 2016</xref>; <xref ref-type="bibr" rid="B47">Kumar et al., 2016</xref>; <xref ref-type="bibr" rid="B64">Noh and Han, 2016</xref>; <xref ref-type="bibr" rid="B29">Gao et al., 2019</xref>), and knowledge-enhanced approaches (<xref ref-type="bibr" rid="B87">Wang et al., 2015</xref>, <xref ref-type="bibr" rid="B86">2017</xref>; <xref ref-type="bibr" rid="B89">Wu et al., 2016</xref>; <xref ref-type="bibr" rid="B96">Zhu et al., 2016</xref>).</p>
<p>More recently, most models have attempted to employ attention mechanisms for mapping text and image features together (<xref ref-type="bibr" rid="B66">Peng et al., 2018</xref>; <xref ref-type="bibr" rid="B57">Lu et al., 2016</xref>; <xref ref-type="bibr" rid="B16">Chen et al., 2015</xref>; <xref ref-type="bibr" rid="B72">Shi et al., 2018</xref>). Moreover, pre-trained visual-language (V &#x0002B; L) models such as visualBERT (<xref ref-type="bibr" rid="B49">Li et al., 2019a</xref>), UNITER (<xref ref-type="bibr" rid="B18">Chen et al., 2020b</xref>), VilBERT (<xref ref-type="bibr" rid="B56">Lu et al., 2019</xref>), and CLIP (<xref ref-type="bibr" rid="B69">Radford et al., 2021</xref>) demonstrated their ability for increased performance. Additionally, researchers also began using image captioning to provide models with increased text context for understanding complex medical queries (<xref ref-type="bibr" rid="B19">Cong et al., 2022</xref>). However, despite these advancements, the available models fail to easily deal with the diversity of the type of questions encountered during real-life medical scenarios. This lack of adaptability limits their utility and points toward the need for novel approaches for the unique Med-VQA concerns. This paper addresses the limitations of existing Med-VQA approaches by introducing a new architecture that improves flexibility and accuracy in answering medical questions. Unlike conventional models, our approach hierarchically categorizes questions based on their dependence on image, text, or mixed modalities. This classification allows for the fine-tuning of models for each form of the question independently, eliminating the generalization and overfitting issues. Besides, by using the switch function for adaptive best-fitting model selection for each form of the question, the solution is extremely flexible and adjustable, providing much improved performance and accuracy.</p>
<p>The key contributions of this study are listed as follows:</p>
<list list-type="bullet">
<list-item><p>This study introduces a novel multi-level VQA framework designed to handle diverse medical question types by categorizing them hierarchically based on their reliance on image, text, or combined modalities.</p></list-item>
<list-item><p>The proposed VQA system employs a bi-level architecture where the first-level classifies the input question type. The second level utilizes specialized component models for each question type, improving accuracy by dynamically selecting the most suitable model using a switch function.</p></list-item>
<list-item><p>The bi-level model is constructed using components selected from the best-performing state-of-the-art models on diabetic retinopathy. This design demonstrates how the proposed approach can enhance their performance within a unified framework. Those models are the ELECTRA-SWIN and two GS-ELECTRA-SWIN models with different hyper-parameters.</p></list-item>
</list>
<p>The subsequent sections of the paper are structured as follows: Section 2 presents the existing literature and methodologies for Med-VQA, highlighting current limitations and research gaps. Section 3 details the proposed hierarchical Med-VQA framework, encompassing its architecture and implementation. Section 4 presents the experimental results and discusses the performance improvements achieved by the proposed method. Finally, Section 5 presents with principal findings, implications, and recommendations for future research endeavors.</p>
</sec>
<sec id="s2">
<title>2 Related works</title>
<p>VQA systems generally consist of four essential elements: vision featurization, text featurization, fusion models, and answer classification or generation. These elements collectively enable the effective processing of image-based queries.</p>
<sec>
<title>2.1 Vision Featurization</title>
<p>In the domain of VQA, vision featurization is a fundamental component of the multimodal architecture. Its primary role is to extract essential visual information from images. Representing an image as a numerical vector&#x02013;known as image featurization&#x02014;involves applying various techniques. These techniques include the scale-invariant feature transform (SIFT) (<xref ref-type="bibr" rid="B55">Lowe, 1999</xref>), simple RGB vectors, histogram of oriented gradients (HOG) (<xref ref-type="bibr" rid="B21">Dalal and Triggs, 2005</xref>), Haar transform (<xref ref-type="bibr" rid="B52">Lienhart and Maydt, 2002</xref>), and deep learning methodologies.</p>
<p>Deep learning approaches, particularly Convolutional Neural Networks (CNNs), play a pivotal role in visual feature extraction by leveraging neural networks to learn essential visual features. Deep learning can involve training models from scratch, which requires large datasets. Alternatively, transfer learning techniques yield strong performance even with limited data. Given the constraints of medical VQA datasets, researchers often resort to leveraging pre-trained models to enhance performance.</p>
<p>Widely used pre-trained models include AlexNet (<xref ref-type="bibr" rid="B46">Krizhevsky et al., 2017</xref>), VGGNet (<xref ref-type="bibr" rid="B73">Simonyan and Zisserman, 2015</xref>; <xref ref-type="bibr" rid="B92">Zhang et al., 2019</xref>; <xref ref-type="bibr" rid="B3">Abacha et al., 2018</xref>; <xref ref-type="bibr" rid="B80">Verma and Ramachandran, 2020a</xref>; <xref ref-type="bibr" rid="B15">Bounaama and Abderrahim, 2019</xref>), GoogLeNet (<xref ref-type="bibr" rid="B76">Szegedy et al., 2015</xref>), ResNet (<xref ref-type="bibr" rid="B34">He et al., 2016</xref>; <xref ref-type="bibr" rid="B28">Fukui et al., 2016</xref>; <xref ref-type="bibr" rid="B43">Kim et al., 2017</xref>; <xref ref-type="bibr" rid="B14">Ben-Younes et al., 2017</xref>; <xref ref-type="bibr" rid="B37">Huang et al., 2023</xref>; <xref ref-type="bibr" rid="B78">Tascon-Morales et al., 2022</xref>, <xref ref-type="bibr" rid="B79">2023</xref>; <xref ref-type="bibr" rid="B33">Haridas et al., 2022</xref>), and DenseNet-121 (<xref ref-type="bibr" rid="B45">Kovaleva et al., 2020</xref>). These architectures have shown strong effectiveness in extracting image features.</p>
<p>In addition, ensemble models&#x02014;combinations of multiple neural networks&#x02014;have gained traction in vision feature extraction within VQA systems. By aggregating outputs, ensembles can outperform individual models. This potential has motivated researchers to explore their utility in enhancing vision feature extraction (<xref ref-type="bibr" rid="B51">Liao et al., 2020</xref>; <xref ref-type="bibr" rid="B26">Do et al., 2021</xref>; <xref ref-type="bibr" rid="B30">Gong et al., 2021</xref>; <xref ref-type="bibr" rid="B84">Wang et al., 2022a</xref>,<xref ref-type="bibr" rid="B85">b</xref>).</p>
</sec>
<sec>
<title>2.2 Text featurization in visual question answering</title>
<p>Text featurization, Like vision featurization, is crucial in converting questions into numeric vectors and facilitating mathematical computations in VQA systems. The selection of an appropriate text embedding method often involves an iterative process (<xref ref-type="bibr" rid="B61">Manmadhan and Kovoor, 2020</xref>). Various text embedding techniques employed in SOTA models significantly influence the multi-modal nature of VQA systems.</p>
<p>Among the prevalent text embedding methods used in question modeling, Long Short-Term Memory (LSTM) (<xref ref-type="bibr" rid="B36">He et al., 2020b</xref>; <xref ref-type="bibr" rid="B45">Kovaleva et al., 2020</xref>; <xref ref-type="bibr" rid="B35">He et al., 2020a</xref>; <xref ref-type="bibr" rid="B78">Tascon-Morales et al., 2022</xref>, <xref ref-type="bibr" rid="B79">2023</xref>; <xref ref-type="bibr" rid="B84">Wang et al., 2022a</xref>,<xref ref-type="bibr" rid="B85">b</xref>), Gated Recurrent Units (GRU) (<xref ref-type="bibr" rid="B36">He et al., 2020b</xref>,<xref ref-type="bibr" rid="B35">a</xref>), Recurrent Neural Networks (RNNs) (<xref ref-type="bibr" rid="B9">Allaouzi et al., 2018</xref>; <xref ref-type="bibr" rid="B3">Abacha et al., 2018</xref>; <xref ref-type="bibr" rid="B95">Zhou et al., 2018b</xref>; <xref ref-type="bibr" rid="B77">Talafha and Al-Ayyoub, 2018</xref>), Faster-RNN (<xref ref-type="bibr" rid="B36">He et al., 2020b</xref>,<xref ref-type="bibr" rid="B35">a</xref>), and the encoder-decoder approach (<xref ref-type="bibr" rid="B82">Vu et al., 2020</xref>; <xref ref-type="bibr" rid="B28">Fukui et al., 2016</xref>; <xref ref-type="bibr" rid="B43">Kim et al., 2017</xref>; <xref ref-type="bibr" rid="B14">Ben-Younes et al., 2017</xref>; <xref ref-type="bibr" rid="B81">Verma and Ramachandran, 2020b</xref>; <xref ref-type="bibr" rid="B44">Kiros et al., 2015</xref>) are widely utilized.</p>
<p>Moreover, pre-trained models like Generalized Auto-regressive Pre-training for Language Understanding (XLNet) (<xref ref-type="bibr" rid="B91">Yang et al., 2019</xref>) and Bidirectional Encoder Representations from Transformers (BERT) (<xref ref-type="bibr" rid="B25">Devlin et al., 2019</xref>; <xref ref-type="bibr" rid="B81">Verma and Ramachandran, 2020b</xref>; <xref ref-type="bibr" rid="B37">Huang et al., 2023</xref>; <xref ref-type="bibr" rid="B33">Haridas et al., 2022</xref>) have gained prominence in text featurization within VQA frameworks. Notably, specific models opt to bypass explicit text featurization, treating the problem as an image classification task (<xref ref-type="bibr" rid="B30">Gong et al., 2021</xref>; <xref ref-type="bibr" rid="B27">Eslami et al., 2021</xref>; <xref ref-type="bibr" rid="B71">Schilling et al., 2021</xref>). This enhanced description of text featurization in VQA systems emphasizes the diverse range of methods and pre-trained models used to transform textual queries into numerical representations, thereby enhancing the model&#x00027;s overall performance and multimodal capabilities.</p>
</sec>
<sec>
<title>2.3 Fusion in visual question answering systems</title>
<p>The fusion step in VQA systems involves the integration of independently extracted text and image features. This fusion process serves as a pivotal stage in VQA pipelines. <xref ref-type="bibr" rid="B61">Manmadhan and Kovoor (2020)</xref> have categorized fusion into three main categories: baseline fusion models, joint attention models, and end-to-end neural network models. Baseline fusion models encompass a variety of fusion techniques, such as element-wise addition (<xref ref-type="bibr" rid="B12">Antol et al., 2015</xref>), element-wise multiplication, and concatenation (<xref ref-type="bibr" rid="B94">Zhou et al., 2018a</xref>). They also include combinations of these methods (<xref ref-type="bibr" rid="B60">Malinowski et al., 2017</xref>) and hybrid approaches involving polynomial functions.</p>
<p>End-to-end neural network models are instrumental in seamlessly merging image and text features. Noteworthy methods include neural module networks (NMNs) (<xref ref-type="bibr" rid="B11">Andreas et al., 2016b</xref>), multi-modal approaches like MCB (<xref ref-type="bibr" rid="B28">Fukui et al., 2016</xref>), and dynamic parameter prediction networks (DPPNs) (<xref ref-type="bibr" rid="B65">Noh et al., 2016</xref>). Other approaches include multi-modal residual networks (MRNs) (<xref ref-type="bibr" rid="B42">Kim et al., 2016</xref>), cross-modal multistep fusion (CMF) networks (<xref ref-type="bibr" rid="B62">Mingrui et al., 2018</xref>), and basic MCB models enhanced with deep attention neural tensor network (DA-NTN) modules (<xref ref-type="bibr" rid="B13">Bai et al., 2018</xref>). Additional methods employ MLPs (<xref ref-type="bibr" rid="B63">Narasimhan and Schwing, 2018</xref>) and encoder-decoder techniques (<xref ref-type="bibr" rid="B17">Chen et al., 2020a</xref>; <xref ref-type="bibr" rid="B50">Li et al., 2019b</xref>).</p>
<p>Joint attention models include the word-to-region attention network (WRAN) (<xref ref-type="bibr" rid="B66">Peng et al., 2018</xref>), co-attention mechanisms (<xref ref-type="bibr" rid="B57">Lu et al., 2016</xref>), question-guided attention maps (QAM) (<xref ref-type="bibr" rid="B16">Chen et al., 2015</xref>), and question type-guided attention (QTA) (<xref ref-type="bibr" rid="B72">Shi et al., 2018</xref>). These approaches are designed to capture nuanced semantic relationships between text and image attentions (<xref ref-type="bibr" rid="B61">Manmadhan and Kovoor, 2020</xref>).</p>
<p>In addition to traditional neural network methods like LSTM and encoder-decoder architectures, Verma and Ramachandran (<xref ref-type="bibr" rid="B81">Verma and Ramachandran, 2020b</xref>) have introduced a multi-model approach incorporating encoder-decoder, LSTM, and GloVe embeddings. Moreover, integrating vision-language pre-trained models, as seen in <xref ref-type="bibr" rid="B33">Haridas et al. (2022)</xref>, further enriches the fusion process within VQA systems. Recent improvements in VQA show that most methods integrate vision and text processing to enhance accuracy. Vision featurization now relies on CNNs for detailed feature extraction, while text featurization employs models such as LSTMs and BERT for efficient question encoding. Advanced fusion techniques, especially attention-based networks, refine image-text alignment and push VQA toward higher levels of cross-modal understanding and performance. Recent work by <xref ref-type="bibr" rid="B79">Tascon-Morales et al. (2023)</xref> benchmarked transformer-based VQA models on datasets like VQA-RAD and PathVQA. This revealed problems with dataset bias and generalization. Other models, such as ViLT, VisualBERT, and GLoRIA, perform well due to vision-language pretraining and attention-based fusion. Unlike these flat architectures, our approach introduces a bi-level structure with question-type routing to improve specialization and robustness.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Proposed method</title>
<p>A multi-level VQA system is a VQA that has multiple levels, each with several VQAs to handle particular questions or answers. This section highlights the methodology for the multi-level VQA system, encompassing problem specification, an outline of the proposed approach, a description of its elements, and subsequent model training strategies.</p>
<p>This section delineates the methodology for the multi-level VQA system, including problem specification, an overview of the proposed approach, a description of its components, and subsequent model training procedures.</p>
<sec>
<title>3.1 The proposed method overview</title>
<p>From the VQA models presented by <xref ref-type="bibr" rid="B8">Al-Hadhrami et al. (2023)</xref>, we found that the models with various hyper-parameters outperform each other in different question types or particular answers. Therefore, designing models focusing on appropriate question types or classes has become increasingly crucial to enhancing the performance and effectiveness of VQA models. The flexibility of this approach allows for the customization of models to fit specific question types, leading to improved performance and more accurate answers. The existing Med-VQA datasets, several methods exist to handle the data and fine-tune the models on these datasets. For example, the dataset with multi-modality images and the dataset with different question types. Since limited data is one gap in the Med-VQA, splitting the data into sub-data can affect the model generalization and lead to overfitting. Therefore, using all data to fine-tune the model by focusing only on particular question types or classes could help to overcome those issues. In SOTA, a hierarchical model is proposed by splitting the data based on the image modality using image modality recognition or question type, such as open and closed, based on text only. Visual question types can be classified based on image, text, or both. The first two question type classifications are used in the literature, while we proposed utilizing the last method. In this paper, the multi-level VQA model is a hierarchical model composed of multiple levels of VQAs, each addressing specific aspects of the question. The predicted answer could be gained from the different model levels or the last one. For instance, one level may handle image-related inquiries, the next level focuses on question types, and the final level addresses the primary visual question. The investigation of the multi-level VQA model aims to improve overall performance. This study employs a Bi-level VQA model. <xref ref-type="fig" rid="F1">Figure 1</xref> shows the overall multi-level VQA method structure. Each level contains two or more VQA except the first-level, which includes one or more VQA. In this study, the first-level only has a single VQA. A single or multiple switch function separates the levels from each other. <xref ref-type="table" rid="T10">Algorithm 1</xref> shows the procedure of the multi-level VQA framework.</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>Overall structure of the multi-level VQA framework. The system is composed of n levels, each containing multiple models. The input image and question are first processed at first-level, and the switch function routes them to the appropriate model in the next level. The final answer is produced by one of the models, depending on the routing decisions across levels.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1646176-g0001.tif">
<alt-text>Schematic diagram illustrating the overall structure of the proposed multi-level Visual Question Answering (VQA) framework. The process starts at Level 1, which identifies the type of input question and routes it through a switch function to the appropriate model at the next level. Each subsequent level contains multiple VQA modules that further refine the analysis, and the final answers are generated after progressive routing across all levels.</alt-text>
</graphic>
</fig>
<table-wrap position="float" id="T10">
<label>Algorithm 1</label>
<caption><p>Multilevel VQA system.</p></caption>
<table frame="hsides" rules="groups">
<tbody>
<tr>
<td align="left" valign="top"><monospace><bold>Require:</bold> Image <italic>I</italic>, Question <italic>Q</italic>, Set of Levels <italic>L</italic> &#x0003D; {<italic>L</italic><sub>1</sub>, <italic>L</italic><sub>2</sub>, &#x02026;, <italic>L</italic><sub><italic>n</italic></sub>}</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace><bold>Ensure:</bold> Answer <italic>A</italic></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>1: Initialize <italic>current</italic>_<italic>level</italic>&#x02190;1</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>2: <italic>i</italic>&#x02190;1 {<italic>i</italic> is the level number}</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>3: <italic>j</italic>&#x02190;1 {<italic>j</italic> is the model number}</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>4: <italic>a</italic><sub><italic>ij</italic></sub>&#x02190;<italic>model</italic>(<italic>I, Q</italic>) {<italic>a</italic><sub><italic>ij</italic></sub> is the answer of the model <italic>j</italic> in the level <italic>i</italic>}</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>5: <bold>while</bold> <italic>current</italic>_<italic>level</italic> &#x02264; <italic>n</italic> <bold>do</bold></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>6: &#x02003;<italic>models</italic>&#x02190;<italic>L</italic><sub><italic>Next</italic>_<italic>level</italic></sub> {Retrieve models in current level}</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>7: &#x02003;<italic>selected</italic>_<italic>model</italic><sub><italic>ij</italic></sub>&#x02190;SwitchFunction(<italic>I, Q, models, a</italic><sub><italic>ij</italic></sub>) {Select the most suitable model}</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>8: &#x02003;<italic>a</italic><sub><italic>ij</italic></sub>&#x02190;<italic>selected</italic>_<italic>model</italic>(<italic>I, Q</italic>) {Answer from selected model}</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>9: &#x02003;{Intermediate answer is used only by switch function and not passed to next level}</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>10: &#x02003;<bold>if</bold> <italic>a</italic><sub><italic>ij</italic></sub> is final answer <bold>then</bold></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>11: &#x02003;&#x02003;<italic>A</italic>&#x02190;<italic>a</italic><sub><italic>ij</italic></sub></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>12: &#x02003;&#x02003;<bold>return</bold> <italic>A</italic></monospace> </td>
</tr>
<tr>
<td align="left" valign="top"><monospace>13: &#x02003;<bold>end if</bold></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>14: &#x02003;<italic>current</italic>_<italic>level</italic>&#x02190;<italic>current</italic>_<italic>level</italic>&#x0002B;1</monospace> </td>
</tr>
<tr>
<td align="left" valign="top"><monospace>15: <bold>end while</bold></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>16: <italic>A</italic>&#x02190;<italic>a</italic><sub><italic>ij</italic></sub></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>17: <bold>return</bold> A</monospace></td>
</tr>
</tbody>
</table>
</table-wrap>
<sec>
<title>3.1.1 Problem specification and formulation</title>
<p>Med-VQA attempts to accurately predict the correct answers based on a combination of medical images and text questions. The task requires generating a brief and accurate textual answer from the input pair comprising a medical image <italic>I</italic> and a question <italic>Q</italic>. This process is mathematically expressed as in <xref ref-type="disp-formula" rid="E1">Equation 1</xref>:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mo>=</mml:mo><mml:mi>V</mml:mi><mml:mi>Q</mml:mi><mml:mi>A</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>I</mml:mi><mml:mo>,</mml:mo><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>I</italic> is the input image, <italic>Q</italic> is the question, <italic>A</italic> is the predicted answer, and &#x00398; represents the model parameters.</p>
 <p>In multi-level VQA (<italic>MVQA</italic>), both <italic>I</italic> and <italic>Q</italic> are processed by several models <italic>M</italic><sub><italic>k</italic></sub>, where <italic>k</italic> &#x0003C; &#x0003D; <italic>n</italic>, <italic>n</italic> is the levels number. Each level <italic>i</italic> has <italic>j</italic> models, where <italic>j</italic>&#x0003E;0. Therefore, <italic>M</italic><sub><italic>ij</italic></sub> denotes to the model <italic>j</italic> in the level <italic>i</italic>. Those levels are separated by switch function <italic>S</italic><sub><italic>i</italic></sub>, where <italic>i</italic> is the level that precedes it. So, the intermediate answers at each level given by in <xref ref-type="disp-formula" rid="E2">Equation (2)</xref>:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>I</mml:mi><mml:mo>,</mml:mo><mml:mi>Q</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>a</italic><sub><italic>ij</italic></sub> is the answer of the models <italic>j</italic> in the level <italic>i</italic>.</p>
<p>and the decision to proceed to the next level is given by in <xref ref-type="disp-formula" rid="E3">Equation (3)</xref>:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo> <mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mn>0</mml:mn></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mtext>if&#x000A0;no&#x000A0;extra&#x000A0;level&#x000A0;and&#x000A0;final&#x000A0;answer&#x000A0;is&#x000A0;detected</mml:mtext></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mn>1</mml:mn></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mtext>if&#x000A0;routing&#x000A0;to&#x000A0;the&#x000A0;next&#x000A0;</mml:mtext><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mi>j</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow> </mml:mrow></mml:mrow></mml:math></disp-formula>
<p>The final answer is given by in <xref ref-type="disp-formula" rid="E4">Equation (4)</xref>:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>I</mml:mi><mml:mo>,</mml:mo><mml:mi>Q</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>D</italic><sub><italic>i</italic>&#x02212;1</sub> &#x0003D; 0.</p>
</sec>
<sec>
<title>3.1.2 Bi-Level VQA</title>
<p>Our proposed model is designed to enhance accuracy and efficiency by leveraging a hierarchical structure consisting of two distinct levels. The first-level serves as a classification system, which identifies the type of input question and produces specific information to guide subsequent processing. A switch function uses this information to route the visual question to the suitable VQA model in the second level to predict the answer. The first-level employs the GS-ELECTRA-SWIN VQA model as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, known for its efficiency in question types classification, as discussed by <xref ref-type="bibr" rid="B8">Al-Hadhrami et al. (2023)</xref>.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>Overall structure of the first-level in the multi-level VQA framework. The model predicts the question type but does not provide the final answer.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1646176-g0002.tif">
<alt-text>Flowchart showing a process involving image analysis. On the left, a retinal image is analyzed with the question, &#x0201C;Is there a hard exudate in the image?&#x0201D; This feeds into a central system labeled &#x0201C;GS-ELECTRA-SWIN VQA,&#x0201D; which outputs to multiple question types, labeled sequentially from Question type 1 to Question type n.</alt-text>
</graphic>
</fig>
<p>The second level can include differently designed models, where each model fits well for one or more question types, or the same model but fine-tuned with different hyper-parameters to be suitable for such a question type. The second level used the ELECTRA-SWIN and GS-ELECTRA-SWIN VQA models, as presented in <xref ref-type="fig" rid="F3">Figure 3</xref>. The explanation of each transformer is given below.</p>
<list list-type="bullet">
<list-item><p><bold>ELECTRA-SWIN</bold> The proposed VQA model combines ELECTRA and Swin Transformers to extract text and visual features. The ELECTRA model is implemented to extract informative text features from the input question, while the Swin Transformer captures salient visual features from the corresponding image. The extracted text and visual features are subsequently combined and normalized to ensure they are on a similar scale. Finally, the normalized, concatenated features are passed to a MLP network, which classifies the answer based on the integrated information from both modalities. The general structure of the ELECTRA-SWIN model is shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. <xref ref-type="table" rid="T11">Algorithms 2</xref> and <xref ref-type="table" rid="T12">3</xref> show the ELECTRA and SWIN features extraction respectively.
<list list-type="simple">
<list-item><p>The essential advantage of this architecture is its ability to leverage the robust feature extraction capabilities of ELECTRA and Swin Transformer, which have demonstrated SOTA performance on various NLP and computer-vision tasks. By fusing the text and visual features and passing them through an MLP, the model can effectively analyze the input image and the question to determine the accurate answer. This method offers an adaptable and scalable way to handle various possible answer choices, making it well-suited for diverse VQA scenarios.</p></list-item>
</list></p></list-item>
<list-item><p><bold>GS-ELECTRA-SWIN</bold> The ELECTRA-SWIN model is produced from the optimal selection criterion (<xref ref-type="bibr" rid="B8">Al-Hadhrami et al., 2023</xref>), which selects the model exhibiting the highest validation accuracy throughout the training. It can be mathematically written as follow in <xref ref-type="disp-formula" rid="E5">Equation (5)</xref>:</p></list-item>
</list>
<disp-formula id="E5"><label>(5)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext class="textrm" mathvariant="normal">argmax</mml:mtext></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mtext class="textrm" mathvariant="normal">ValAcc</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<list list-type="simple">
<list-item><p>The GS-ELECTRA-SWIN model integrates the greedy soup technique with the ELECTRA-SWIN model. The final model is chosen according to the models generated during the training phase, significantly impacting the average of all notable validation accuracies of fine-tuned models. The mathematical formulation of the model is as:</p></list-item>
<list-item><p>Let <italic>M</italic> &#x0003D; {<italic>m</italic><sub>1</sub>, <italic>m</italic><sub>2</sub>, &#x02026;, <italic>m</italic><sub><italic>n</italic></sub>} and &#x003B8; &#x0003D; {&#x003B8;<sub>1</sub>, &#x003B8;<sub>2</sub>, &#x02026;, &#x003B8;<sub><italic>n</italic></sub>} represent the number of models and their corresponding parameters, accordingly. Additionally, consider &#x003B8;&#x02212;<italic>SoupIngredients</italic> &#x0003D; {&#x003B8;<sub>1</sub>, &#x003B8;<sub>2</sub>, &#x02026;, &#x003B8;<sub><italic>k</italic></sub>} and <italic>M</italic><sub><italic>k</italic></sub> &#x0003D; {<italic>m</italic><sub><italic>k</italic>1</sub>, <italic>m</italic><sub><italic>k</italic>2</sub>, &#x02026;, <italic>m</italic><sub><italic>kk</italic></sub>} as the set of selected parameters or the soup ingredients of the models under evaluation. At each validation computation step <italic>i</italic>, model <italic>m</italic><sub><italic>i</italic></sub> is included if its validation accuracy satisfies the condition <italic>valAcc</italic>(<italic>m</italic><sub><italic>i</italic></sub>&#x0222A;<italic>m</italic><sub><italic>kk</italic></sub>)&#x0003E;min(<italic>m</italic><sub><italic>kk</italic></sub>). The models in <italic>M</italic><sub><italic>k</italic></sub> are arranged in decreasing order based on their <italic>valAcc</italic> scores. Among the models in <italic>M</italic><sub><italic>k</italic></sub> and their corresponding &#x003B8;&#x02212;<italic>SoupIngredients</italic> parameters, a model is selected for fusion if, for each step from <italic>i</italic> to <italic>k</italic>, the validation accuracy <italic>valAcc</italic>(<italic>sg</italic><sub><italic>i</italic>&#x02212;1</sub>&#x0222A;{&#x003B8;<sub><italic>i</italic></sub>}) exceeds <italic>valAcc</italic>(&#x003B8;&#x02212;<italic>SoupIngredients</italic><sub><italic>i</italic>&#x02212;1</sub>). Let &#x003B8;&#x02212;<italic>SoupIngredients</italic><sub><italic>i</italic></sub> denote the set of <italic>j</italic> selected models. The final model parameters &#x003B8;&#x02032; are computed as the average of the parameters of the chosen &#x003B8;&#x02212;<italic>SoupIngredients</italic> models, calculated as in <xref ref-type="disp-formula" rid="E6">Equation (6)</xref>:</p></list-item>
</list>
<disp-formula id="E6"><label>(6)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<list list-type="simple">
<list-item><p>The presented model depends on the top three leading models (<italic>k</italic> &#x0003D; 3), with validation evaluated at midway and at the end of each epoch. <xref ref-type="table" rid="T14">Algorithm 5</xref> details the greedy soup algorithm for fusing three models with varying hyperparameters. <xref ref-type="fig" rid="F5">Figure 5</xref> illustrates the overarching greedy soup framework for merging three models.</p></list-item>
</list>
<list list-type="bullet">
<list-item><p><bold>Switch function</bold> The switch function is tasked with determining whether to switch the input question and image to the appropriate model in the subsequent stage or to provide the final answer to the visual question. <xref ref-type="disp-formula" rid="E3">Equations 3</xref>, <xref ref-type="disp-formula" rid="E4">4</xref> outline the mathematical process involved in making this decision. <xref ref-type="table" rid="T13">Algorithm 4</xref> shows the switch function procedure.</p></list-item>
</list>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>The Overall second-level structure of the multi-level VQA technique. Second-level structure: the first-level predicts the question type; the switch routes the question to the specialized model for that type, which directly outputs the final answer.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1646176-g0003.tif">
<alt-text>Flowchart depicting a decision-making process for detecting hard exudates in a retinal image. It includes two levels: Level 1 with a GS-ELECTRA-SWIN function, followed by a switch function. Level 2 has three functions: ELECTRA-SWIN and two GS-ELECTRA-SWIN functions. Each function leads to a set of answers. The process starts with the question, &#x0201C;Is there a hard exudate in the image?&#x0201D; and ends with sets of answers.</alt-text>
</graphic>
</fig>
<fig position="float" id="F4">
<label>Figure 4</label>
<caption><p>Overall structure of the ELECTRA-SWIN model. The input question is encoded using the ELECTRA discriminator, while the image is processed through the Swin Transformer. The resulting textual and visual feature representations are normalized, concatenated, and passed through an MLP to generate the final answer.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1646176-g0004.tif">
<alt-text>Flowchart illustrating a model combining text and image processing. An Electra Transformer processes input text, producing a text vector. SWIN Transformer handles image patches, generating an image vector. Both vectors are concatenated and normalized before passing through an MLP, producing answers such as numbers or &#x0201C;yes&#x0201D;/&#x0201D;no&#x0201D;. Image includes an eye scan for reference.</alt-text>
</graphic>
</fig>
<table-wrap position="float" id="T11">
<label>Algorithm 2</label>
<caption><p>Feature extraction from ELECTRA-Base transformer.</p></caption>
<table frame="hsides" rules="groups">
<tbody>
<tr>
<td align="left" valign="top"><monospace><bold>Require:</bold> Input text <italic>X</italic>, pre-trained ELECTRA-Base model <italic>M</italic></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace><bold>Ensure:</bold> Output feature vector <italic>F</italic>&#x02208;&#x0211D;<sup><italic>D</italic></sup></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>1: Tokenize input text and add special tokens for classification: <italic>T</italic> &#x0003D; [<italic>CLS</italic>]&#x0002B;<italic>X</italic>&#x0002B;[<italic>SEP</italic>]</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>2: Convert tokens to their corresponding token IDs: <italic>I</italic> &#x0003D; tokenizer(<italic>T</italic>)</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>3: Pass input token IDs through the pre-trained ELECTRA-Base model to derive the final hidden state: <italic>H</italic> &#x0003D; <italic>M</italic>(<italic>I</italic>)</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>4: Extract the final hidden-state of the special [CLS] token as the output feature vector: <italic>F</italic> &#x0003D; <italic>H</italic><sub>1, :</sub>, where <italic>H</italic><sub>1, :</sub> denotes the first row of the hidden state matrix <italic>H</italic></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>5: <bold>Return</bold> Output feature vector <italic>F</italic></monospace></td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap position="float" id="T12">
<label>Algorithm 3</label>
<caption><p>Feature extraction from SWIN-Base transformer.</p></caption>
<table frame="hsides" rules="groups">
<tbody>
<tr>
<td align="left" valign="top"><monospace><bold>Require:</bold> Input image <italic>X</italic>&#x02208;&#x0211D;<sup><italic>H</italic>&#x000D7;<italic>W</italic>&#x000D7;<italic>C</italic></sup>, pre-trained SWIN-Base model <italic>M</italic></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace><bold>Ensure:</bold> Output feature vector <italic>F</italic>&#x02208;&#x0211D;<sup><italic>D</italic></sup></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>1: Normalize input image: <inline-formula><mml:math id="M7"><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>X</mml:mi><mml:mo>-</mml:mo><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow></mml:mfrac></mml:math></inline-formula></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>2: Pad input image to a multiple of the patch size: <italic>X</italic>&#x02033; &#x0003D; ZeroPad(<italic>X</italic>&#x02032;, <italic>P</italic>)</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>3: Split input image into non-overlapping patches of size <italic>P</italic>: <inline-formula><mml:math id="M8"><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x02033;</mml:mo></mml:mrow></mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula>, where <italic>p</italic><sub><italic>i</italic></sub> denotes the coordinates of the <italic>i</italic>-th patch</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>4: Embed each patch using a learnable embedding layer: <italic>E</italic><sub><italic>i</italic></sub> &#x0003D; <italic>W</italic><sub><italic>emb</italic></sub>(<italic>X</italic><sub><italic>i</italic></sub>), where <italic>W</italic><sub><italic>emb</italic></sub> is a learnable weight matrix</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>5: Add positional embeddings to each patch embedding: <italic>E</italic><sub><italic>i</italic></sub> &#x0003D; <italic>E</italic><sub><italic>i</italic></sub>&#x0002B;<italic>P</italic><sub><italic>i</italic></sub>, where <italic>P</italic><sub><italic>i</italic></sub> is a learnable positional embedding</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>6: Pass input patches through the pre-trained SWIN-Base model to obtain the final hidden state: <italic>H</italic> &#x0003D; <italic>M</italic>(<italic>E</italic>)</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>7: Apply a global average pooling function to the hidden state to obtain the output feature vector: <inline-formula><mml:math id="M9"><mml:mi>F</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, where <italic>N</italic> is the total number of patches</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>8: <bold>Return</bold> Output feature vector <italic>F</italic></monospace></td>
</tr>
</tbody>
</table>
</table-wrap>
<fig position="float" id="F5">
<label>Figure 5</label>
<caption><p>Overall structure of the GS-ELECTRA-SWIN model. Three independently trained ELECTRA-SWIN models, each with different training epochs, are combined using the greedy-soup technique to produce the final ensemble model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1646176-g0005.tif">
<alt-text>Diagram of a model architecture featuring three primary sections: Model 1, Model 2, and Greedy Soups, leading to a Final Model. Each model processes input through ELECTRA Transformers and SWIN Transformer Blocks, focusing on text and image vectors. Models are compared for accuracy in Greedy Soups, iteratively refining parameters to optimize performance. The final model combines these components for a robust VQA (Visual Question Answering) system.</alt-text>
</graphic>
</fig>
<table-wrap position="float" id="T13">
<label>Algorithm 4</label>
<caption><p>Bi-level VQA with switch function.</p></caption>
<table frame="hsides" rules="groups">
<tbody>
<tr>
<td align="left" valign="top"><monospace><bold>Require:</bold> Image <italic>I</italic>, Question <italic>Q</italic></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace><bold>Ensure:</bold> Final Answer <italic>A</italic></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>1: <italic>a</italic><sub>1</sub>&#x02190;GS-ELECTRA-SWIN(<italic>I, Q</italic>) {Level 1 prediction}</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>2: <italic>decision</italic>&#x02190;SwitchFunction(<italic>a</italic><sub>1</sub>) {Determine next action}</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>3: <bold>if</bold> <italic>decision</italic> &#x0003D; = <monospace>final</monospace> <bold>then</bold></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>4: &#x02003;<italic>A</italic>&#x02190;<italic>a</italic><sub>1</sub></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>5: &#x02003;<bold>return</bold> <italic>A</italic></monospace> </td>
</tr>
<tr>
<td align="left" valign="top"><monospace>6: <bold>else if</bold> <italic>decision</italic> &#x0003D; = <monospace>route_to_model1</monospace> <bold>then</bold></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>7: &#x02003;<italic>A</italic>&#x02190;ELECTRA-SWIN(<italic>I, Q</italic>) {Level 2 - Model 1}</monospace> </td>
</tr>
<tr>
<td align="left" valign="top"><monospace>8: <bold>else if</bold> <italic>decision</italic> &#x0003D; = <monospace>route_to_model2</monospace> <bold>then</bold></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>9: &#x02003;<italic>A</italic>&#x02190;GS-ELECTRA-SWIN(<italic>I, Q</italic>) {Level 2 - Model 2}</monospace> </td>
</tr>
<tr>
<td align="left" valign="top"><monospace>10: <bold>else</bold></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>11: &#x02003;<italic>A</italic>&#x02190;GS-ELECTRA-SWIN(<italic>I, Q</italic>) {Level 2 - Model 3. Model 3 has different hyperparameters from Model.}</monospace> </td>
</tr>
<tr>
<td align="left" valign="top"><monospace>12: <bold>end if</bold></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>13: <bold>return</bold> <italic>A</italic></monospace></td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec>
<title>3.2 Training using greedy soup technique</title>
<p>The proposed multi-level VQA model employs pre-trained models for both textual and visual feature extraction. These models are fine-tuned using the designed Med-VQA dataset to adapt to the specific requirements of medical question-answering. To improve the generalization and performance of the model, the greedy soup method is applied for fine-tuning parameters. This technique integrates several fine-tuned models by fusing their parameters, thus providing a general and efficient configuration.</p>
<p>During the training process, the model undergoes multiple rounds of fine-tuning, and validation accuracy is calculated at each stage. The final parameters are derived by averaging the weights of the best <italic>k</italic> models, selected according to their validation performance. This fusion technique significantly improves the generalization capability of the model, minimizing overfitting and enhancing performance. For this research, the final model is generated using the top three fine-tuned configurations, with different hyperparameters, integrated through the greedy soup technique. This process is illustrated in the pseudocode (<xref ref-type="fig" rid="F6">Figure 6</xref>), <xref ref-type="table" rid="T14">Algorithm 5</xref> and depicted in the flowchart in <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
<fig position="float" id="F6">
<label>Figure 6</label>
<caption><p>The Pseudo code of the greedy-soup ensemble technique, where the final model weights are generated based on the three most significant model weights regarding the validation accuracy.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1646176-g0006.tif">
<alt-text>Python code snippet demonstrates a model averaging process. It sorts models by validation accuracy, initializes with the best model, and performs greedy averaging. Each new model is temporarily added, evaluated, and retained if accuracy improves or remains constant. The process iterates through all models, and the final averaged model is returned.</alt-text>
</graphic>
</fig>
<table-wrap position="float" id="T14">
<label>Algorithm 5</label>
<caption><p>Greedy soup for model ensembling</p></caption>
<table frame="hsides" rules="groups">
<tbody>
<tr>
<td align="left" valign="top"><monospace><bold>Require:</bold> <inline-formula><mml:math id="M10"><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>M</mml:mi></mml:mstyle></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>: List of model checkpoints</monospace>,<break/> &#x02003;<inline-formula><mml:math id="M11"><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">val</mml:mtext></mml:mstyle></mml:mrow></mml:msub></mml:math></inline-formula>: Validation dataset,<break/> &#x02003;<inline-formula><mml:math id="M12"><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">Acc</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003B8;</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">val</mml:mtext></mml:mstyle></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>: <monospace>Accuracy function</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace><bold>Ensure:</bold> &#x003B8;<sub>soup</sub>: Final averaged model</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>1: Sort <inline-formula><mml:math id="M13"><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>M</mml:mi></mml:mstyle></mml:mrow></mml:math></inline-formula> by descending accuracy:</monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>2: &#x02003;<inline-formula><mml:math id="M14"><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">Acc</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">val</mml:mtext></mml:mstyle></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02265;</mml:mo><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">Acc</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">val</mml:mtext></mml:mstyle></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02265;</mml:mo><mml:mo>&#x022EF;</mml:mo><mml:mo>&#x02265;</mml:mo><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">Acc</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">val</mml:mtext></mml:mstyle></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>3: Initialize soup set: <inline-formula><mml:math id="M15"><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>S</mml:mi></mml:mstyle></mml:mrow><mml:mo>&#x02190;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>4: Initialize soup model: &#x003B8;<sub>soup</sub>&#x02190;&#x003B8;<sub>1</sub></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>5: <bold>for</bold> <italic>i</italic> &#x0003D; 2 to <italic>N</italic> <bold>do</bold></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>6: &#x02003;Compute temporary average</monospace>:<break/> &#x02003;&#x02003;&#x02003;&#x02003;&#x02003;&#x02003;&#x02003;&#x02003; <inline-formula><mml:math id="M16"><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02190;</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>S</mml:mi></mml:mstyle></mml:mrow><mml:mo>|</mml:mo><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>S</mml:mi></mml:mstyle></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:mi>&#x003B8;</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>7: <bold>if</bold> <inline-formula><mml:math id="M17"><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">Acc</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">val</mml:mtext></mml:mstyle></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02265;</mml:mo><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">Acc</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">soup</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">val</mml:mtext></mml:mstyle></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> <bold>then</bold></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>8: &#x02003;&#x02003;<inline-formula><mml:math id="M18"><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>S</mml:mi></mml:mstyle></mml:mrow><mml:mo>&#x02190;</mml:mo><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>S</mml:mi></mml:mstyle></mml:mrow><mml:mo>&#x0222A;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>9: &#x02003;&#x02003;<inline-formula><mml:math id="M19"><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">soup</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>10: &#x02003;<bold>end if</bold></monospace> </td>
</tr>
<tr>
<td align="left" valign="top"><monospace>11: <bold>end for</bold></monospace></td>
</tr>
<tr>
<td align="left" valign="top"><monospace>12: <bold>return</bold> &#x003B8;<sub>soup</sub></monospace></td>
</tr>
</tbody>
</table>
</table-wrap>
 <p>Let &#x00398;<sub>1</sub>, &#x00398;<sub>2</sub>, and &#x00398;<sub>3</sub> denote the parameters of the visual, textual, and MLP components, respectively, while <italic>k</italic> denotes the number of models used for parameter fusion. The parameter <inline-formula><mml:math id="M20"><mml:msubsup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> refers to the parameters of the <italic>j</italic><sup>th</sup> model in the <italic>i</italic><sup>th</sup> component. The combined parameter &#x00398; for the final model is calculated as follows in <xref ref-type="disp-formula" rid="E7">Equation 7</xref>:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M21"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathcolor="#0000ff"><mml:msup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle mathcolor="#0000ff"><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mstyle><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mstyle mathcolor="#0000ff"><mml:msup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle mathcolor="#0000ff"><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mstyle><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mstyle mathcolor="#0000ff"><mml:msup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle mathcolor="#0000ff"><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mstyle><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x02003;&#x02003;</mml:mtext><mml:mo>&#x022EE;</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mstyle mathcolor="#0000ff"><mml:msup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle mathcolor="#0000ff"><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The final fused parameters <italic>Theta</italic> are computed using the following <xref ref-type="disp-formula" rid="E8">Equation 8</xref>:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M22"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mo>&#x00398;</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:munderover></mml:mstyle><mml:msubsup><mml:mrow><mml:mo>&#x00398;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The combination of pre-trained backbones results in multiple concatenated configurations, which are normalized and processed through an MLP for final classification. These configurations are fused using the greedy soup technique to enhance performance and robustness, as demonstrated in <xref ref-type="bibr" rid="B8">Al-Hadhrami et al. (2023)</xref>. <xref ref-type="fig" rid="F5">Figure 5</xref> illustrates the overall structure of the model integrated using the greedy soup method. Other configurations follow the same structure, replacing the pre-trained models used for feature extraction. This training strategy ensures that the final model effectively handles diverse question types, leveraging hierarchical VQA architecture and robust parameter fusion to deliver accurate and reliable predictions in Med-VQA tasks.</p>
</sec>
<sec>
<title>3.3 Accessibility considerations and system framework</title>
<p>This work proposes a multi-level Med-VQA framework aimed at enhancing accessibility for visually impaired users. The primary focus of the current study is on developing and validating the underlying machine learning models and architectural design, rather than delivering a fully integrated end-user system. The proposed framework provides a modular and extensible structure that demonstrates how various components&#x02013;such as question classification, visual feature extraction, and answer generation&#x02014;can be combined effectively. This modularity enables potential integration with accessible platforms in the future, including mobile and web applications that support assistive technologies like screen readers, Braille displays, and voice input/output systems. Accessibility considerations in the framework are informed by established standards, including the Web Content Accessibility Guidelines (WCAG) (<xref ref-type="bibr" rid="B83">W3C, 2023</xref>) and ISO 9241 (<xref ref-type="bibr" rid="B39">International Organization for Standardization, 2008</xref>), which provide internationally recognized guidelines for accessible system design. While the framework lays the technical groundwork, actual integration into real-world accessible interfaces and user-facing applications remains future work. Prior research underscores the importance of tailored interface solutions for visually impaired users. For example, <xref ref-type="bibr" rid="B7">Alhadhrami et al. (2015)</xref> showed that adaptive interfaces coupled with multimodal feedback significantly improve spatial awareness and usability. Similarly, recent studies highlight the benefits of embedding VQA capabilities into intelligent assistive devices such as wearable smart glasses and voice-controlled platforms to enhance user autonomy (<xref ref-type="bibr" rid="B6">Ainary, 2025</xref>).</p>
<p>To ensure practical accessibility impact, future efforts will include participatory evaluations with visually impaired individuals and clinical professionals. These studies will assess task efficiency, user satisfaction, and interaction quality through metrics such as time-to-answer, error rates, and voice/haptic response accuracy. Additionally, a prototype user interface featuring multimodal interaction (voice and haptics) is planned to explore usability and contextual adaptation further.</p>
<p>In summary, the presented framework serves as a foundational architecture that outlines how Med-VQA components can be systematically integrated to support accessibility. The subsequent stages of research will focus on system integration, user-centered design, and rigorous validation to translate this framework into effective assistive technologies for visually impaired users. <xref ref-type="fig" rid="F7">Figure 7</xref> illustrates the DR VQR system framework architecture, which allows visually impaired users to create personal accounts to store their questions and answers. Additionally, the system can be accessed both online and offline.</p>
<fig position="float" id="F7">
<label>Figure 7</label>
<caption><p>Framework architecture of the DR-VQR system. The user uploads a retinal image through the website or the iOS/Android application, formulates a related question, and submits the request. The query is processed by the bi-level VQA model, which generates an answer. The result is then displayed to the user via the mobile application or web interface.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1646176-g0007.tif">
<alt-text>Flowchart depicting a user interacting with a system interface. The interface includes a computer and smartphone, leading to a bi-level model with two levels. Level 1 includes &#x0201C;GS-ELECTRA-SWIN,&#x0201D; while Level 2 has &#x0201C;ELECTRA-SWIN&#x0201D; repetitions. A switch function connects both levels. The process involves analyzing a retinal image with the question, &#x0201C;Is there a hard exudate in the image?&#x0201D; The model processes this to provide an answer back to the user.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="s4">
<title>4 Evaluation protocol</title>
<sec>
<title>4.1 Experimental environment configuration</title>
<p>The models undergo training on a premium Google Colab utilizing NVIDIA A100-SXM4-40 GB (Nvidia Corporation, Santa Clara, CA, USA) with 80 GB RAM or an NVIDIA Tesla T4 with15GB and 25 GB or 51 GB RAM. The optimization function utilizes AdamW with a learning rate of 1.0 &#x000D7; 10<sup>&#x02212;3</sup> and weight decay of 0.9. A fixed random seed (seed = 42) was configured to ensure deterministic behavior and reproducibility of the results. Consequently, the outputs remain consistent across runs, resulting in zero variance in the reported scores. While traditional statistical significance tests rely on variability across multiple runs, in our setup, reproducibility implies that even a small performance gain (e.g., 0.1%) is meaningful and reliable under the same evaluation conditions. More details about model configuration are listed in Table A1 in the <xref ref-type="sec" rid="A1">Appendix</xref>.</p>
</sec>
<sec>
<title>4.2 Assessment criteria</title>
<p>Model performance is evaluated based on the calculation of metrics: precision, model accuracy, F1 score, recall (<xref ref-type="bibr" rid="B68">Powers, 2011</xref>), macro-average recall, macro-average precision, weighted average precision, macro-average F1 score, weighted average F1 score, and weighted average recall (<xref ref-type="bibr" rid="B48">learn developers, 2024</xref>). The performance metrics utilized to evaluate the model and compare the findings with other state-of-the-art models are presented below. The equation representing each metric is given below.</p>
<list list-type="bullet">
<list-item><p><bold>Accuracy:</bold> This is determined using the formula shown below <xref ref-type="disp-formula" rid="E9">Equation 9</xref>:</p></list-item>
</list>
<disp-formula id="E9"><label>(9)</label><mml:math id="M23"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">Accuracy</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mtext class="textrm" mathvariant="normal">TN &#x0002B; TP</mml:mtext></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">TN &#x0002B; TP &#x0002B; FN &#x0002B; FP</mml:mtext></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<list list-type="simple">
<list-item><p>True positives (TP) refer to actual positive instances that are correctly predicted by the model as positive. True Negatives (TN) represent the negative instances accurately classified as negative. False positives (FP) occur when negative instances are incorrectly predicted as positive. Lastly, false negatives (FN) denote positive instances that the model mistakenly classifies as negative.</p></list-item>
</list>
<list list-type="bullet">
<list-item><p><bold>Precision:</bold> measures the ratio of correctly predicted true positive instances relative to the total predicted positive instances. This metric is defined as in <xref ref-type="disp-formula" rid="E10">Equation 10</xref>:</p></list-item>
</list>
<disp-formula id="E10"><label>(10)</label><mml:math id="M24"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">Precision</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<list list-type="bullet">
<list-item><p><bold>Recall sensitivity:</bold> quantifies the proportion of correctly predicted positive instances relative to the total actual positive instances. This metric is measured by <xref ref-type="disp-formula" rid="E11">Equation 11</xref>:</p></list-item>
</list>
<disp-formula id="E11"><label>(11)</label><mml:math id="M25"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">Recall</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<list list-type="bullet">
<list-item><p><bold>F1-score:</bold> The F1-score assesses a model&#x00027;s accuracy in detecting positive instances by determining the harmonic mean of precision and recall. It is computed as in <xref ref-type="disp-formula" rid="E12">Equation 12</xref>:</p></list-item>
</list>
<disp-formula id="E12"><label>(12)</label><mml:math id="M26"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mfrac><mml:mrow><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<list list-type="bullet">
<list-item><p><bold>Average macro accuracy:</bold> The macro average accuracy assesses the model&#x00027;s performance by calculating the accuracy of each class independently and after that averaging these accuracies. The macro accuracy average formula is given in: <xref ref-type="disp-formula" rid="E13">(13)</xref>:</p></list-item>
</list>
<disp-formula id="E13"><label>(13)</label><mml:math id="M27"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">Macro Accuracy Average</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>C</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>C</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<list list-type="simple">
<list-item><p>where <italic>C</italic> denotes the total number of classes, <italic>TP</italic><sub><italic>c</italic></sub> is the number of true positives for class <italic>c</italic>, and <italic>FP</italic><sub><italic>c</italic></sub> is the number of false positives for class <italic>c</italic> .</p></list-item>
</list>
<list list-type="bullet">
<list-item><p><bold>Weighted average accuracy:</bold> calculates the average accuracy for individual classes, considering the class frequencies in the dataset to assign weights. The weighted average accuracy equation is given by <xref ref-type="disp-formula" rid="E14">Equation 14</xref>:</p></list-item>
</list>
<disp-formula id="E14"><label>(14)</label><mml:math id="M28"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">Weighted average accuracy</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mtext class="textrm" mathvariant="normal">TP</mml:mtext></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">n</mml:mtext></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where <italic>n</italic> is the total number of samples in the dataset and <italic>n</italic><sub><italic>c</italic></sub> is the number of samples belonging to class <italic>c</italic>. The <italic>TP</italic><sub><italic>c</italic></sub> are as defined above.</p>
</sec>
<sec>
<title>4.3 Dataset</title>
<p>In this study, the Diabetic Macular Edema (DME) (<xref ref-type="bibr" rid="B78">Tascon-Morales et al., 2022</xref>) is used, which was automatically generated from the Indian Diabetic Retinopathy Image Dataset (IDRiD) (<xref ref-type="bibr" rid="B67">Porwal et al., 2018</xref>) and the e-Ophta dataset (<xref ref-type="bibr" rid="B24">Decenciere et al., 2013</xref>). It comprises 13,470 question-answer pairs and 679 images, divided into 433 images and 9,779 question-answer pairs for training, 134 images and 2380 pairs for validation, and 112 images and 1,311 pairs for testing.</p>
<p>The dataset includes questions regarding hard exudates, optic discs, and the grading of exudates. The dataset includes a question asking whether a hard exudate is present in the image or a specific region of the image. If a hard exudate is present, the answer is labeled as &#x0201C;Yes&#x0201D;; otherwise, it is labeled as &#x0201C;No&#x0201D;. The grading system classifies hard exudates on the retina as follows: grade 0 indicates no presence of hard exudates, grade 1 signifies hard exudates located in the peripheral retina, and grade 2 denotes the presence of hard macular exudates. Additionally, the dataset provides original images along with masks that highlight specific regions of the images, which must be utilized for pre-training. <xref ref-type="table" rid="T1">Table 1</xref> presents the distribution of classes in the training, validation, and testing datasets.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Number of instances per answer for each part of the DME dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Set</bold></th>
<th valign="top" align="center"><bold>Yes</bold></th>
<th valign="top" align="center"><bold>No</bold></th>
<th valign="top" align="center"><bold>0</bold></th>
<th valign="top" align="center"><bold>1</bold></th>
<th valign="top" align="center"><bold>2</bold></th>
<th valign="top" align="center"><bold>Total</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Train</td>
<td valign="top" align="center">4,713</td>
<td valign="top" align="center">4,639</td>
<td valign="top" align="center">166</td>
<td valign="top" align="center">41</td>
<td valign="top" align="center">220</td>
<td valign="top" align="center">9,779</td>
</tr>
<tr>
<td valign="top" align="left">Val</td>
<td valign="top" align="center">1,151</td>
<td valign="top" align="center">1,123</td>
<td valign="top" align="center">39</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">59</td>
<td valign="top" align="center">2,380</td>
</tr>
<tr>
<td valign="top" align="left">Test</td>
<td valign="top" align="center">530</td>
<td valign="top" align="center">650</td>
<td valign="top" align="center">49</td>
<td valign="top" align="center">15</td>
<td valign="top" align="center">67</td>
<td valign="top" align="center">1,311</td>
</tr>
<tr>
<td valign="top" align="left">Total</td>
<td valign="top" align="center">6,394</td>
<td valign="top" align="center">6,412</td>
<td valign="top" align="center">254</td>
<td valign="top" align="center">64</td>
<td valign="top" align="center">346</td>
<td valign="top" align="center">13,470</td>
</tr></tbody>
</table>
</table-wrap>
<p>The DME dataset consists of four distinct types of questions, each with different levels of complexity:</p>
<p><bold>Whole:</bold> e.g., &#x0201C;Are there hard exudates in this image?&#x0201D;&#x02014;requires a binary decision at the image level.</p>
<p><bold>Region:</bold> e.g., &#x0201C;Are there hard exudates in the region?&#x0201D;&#x02014;focuses on a predefined mask in the image and typically requires less complex reasoning since the region is already localized.</p>
<p><bold>Fovea:</bold> e.g., &#x0201C;Are there hard exudates in the fovea?&#x0201D;&#x02014;requires detection of exudates and precise spatial reasoning to determine whether they fall within the foveal region.</p>
<p><bold>Grade:</bold> e.g., &#x0201C;What is the diabetic macular edema grade for this image?&#x0201D;&#x02014;involves multi-class classification based on exudate presence and location.</p>
<p>While Region-type questions are the most frequent, Fovea and Grade questions are the most complex. They first require the system to detect the presence of hard exudates and then localize them accurately relative to the foveal region. This question complexity demands multi-step spatial understanding and is more aligned with clinical decision-making processes. The differences in question complexity provide a valuable framework for evaluating the robustness and reasoning capabilities of VQA models. The distribution of question types across each part of the dataset is shown in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Number of instances per question type for each part of the DME dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Question type</bold></th>
<th valign="top" align="center"><bold>Train</bold></th>
<th valign="top" align="center"><bold>Validation</bold></th>
<th valign="top" align="center"><bold>Test</bold></th>
<th valign="top" align="center"><bold>Total</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Grade</td>
<td valign="top" align="center">427</td>
<td valign="top" align="center">106</td>
<td valign="top" align="center">131</td>
<td valign="top" align="center">664</td>
</tr>
<tr>
<td valign="top" align="left">Macula (Fovea)</td>
<td valign="top" align="center">427</td>
<td valign="top" align="center">106</td>
<td valign="top" align="center">131</td>
<td valign="top" align="center">664</td>
</tr>
<tr>
<td valign="top" align="left">Whole</td>
<td valign="top" align="center">427</td>
<td valign="top" align="center">106</td>
<td valign="top" align="center">131</td>
<td valign="top" align="center">664</td>
</tr>
<tr>
<td valign="top" align="left">Region</td>
<td valign="top" align="center">8,498</td>
<td valign="top" align="center">2,062</td>
<td valign="top" align="center">918</td>
<td valign="top" align="center">11,478</td>
</tr>
<tr>
<td valign="top" align="left">Total</td>
<td valign="top" align="center">9,779</td>
<td valign="top" align="center">2,380</td>
<td valign="top" align="center">1,311</td>
<td valign="top" align="center">13,470</td>
</tr></tbody>
</table>
</table-wrap>
<p>The dataset includes retinal images captured under varied conditions, such as differences in illumination, patient eye positions, and inherent noise. This variability reflects realistic clinical scenarios and adds robustness to the evaluation of the proposed VQA framework. <xref ref-type="fig" rid="F8">Figure 8</xref> shows samples on dataset images.</p>
<fig position="float" id="F8">
<label>Figure 8</label>
<caption><p>Examples of dataset images captured under varying conditions, including differences in illumination, clarity, noise levels, size, and object positioning.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1646176-g0008.tif">
<alt-text>Six retinal images showing variations in appearance. The images display different aspects of the eye, with visible blood vessels and optic discs. Some images highlight discoloration and lesions, indicating potential abnormalities.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<title>4.4 Test significance and impact of seed setting</title>
<p>Randomness in machine learning experiments, such as weight initialization and data shuffling, can lead to variability in model performance. We used a fixed random seed during training and evaluation to mitigate this. Setting a random seed enhances the reproducibility of experiments and ensures that the reported results are stable and not artifacts of random initialization.</p>
<p>To evaluate the impact of the random seed on initial weight settings, we conducted five independent experiments using the base model. All experiments shared the same architecture and training configuration, differing only in the random seed used for weight initialization. The selected seeds were chosen randomly: 10, 23, 42, 70, and 100. <xref ref-type="table" rid="T3">Table 3</xref> reports the Accuracy obtained for each seed. The accuracies ranged from 85.89% to 87.41%, with a mean accuracy of <bold>86.32%</bold> and a standard deviation of <bold>0.62</bold>. We compared these results against the state-of-the-art (SOTA) results reported by (<xref ref-type="bibr" rid="B78">Tascon-Morales et al., 2022</xref>, <xref ref-type="bibr" rid="B79">2023</xref>), which achieved 83.00% and 83.69% accuracy, respectively.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Accuracy results for different random seeds compared to the SOTA baseline of <xref ref-type="bibr" rid="B78">Tascon-Morales et al. (2022</xref>, <xref ref-type="bibr" rid="B79">2023)</xref>.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Seed</bold></th>
<th valign="top" align="center"><bold>Accuracy (%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">10</td>
<td valign="top" align="center">86.04</td>
</tr>
<tr>
<td valign="top" align="left">23</td>
<td valign="top" align="center">86.12</td>
</tr>
<tr>
<td valign="top" align="left">42</td>
<td valign="top" align="center"><bold>87.41</bold></td>
</tr>
<tr>
<td valign="top" align="left">70</td>
<td valign="top" align="center">86.12</td>
</tr>
<tr>
<td valign="top" align="left">100</td>
<td valign="top" align="center">85.89</td>
</tr>
<tr>
<td valign="top" align="left"><bold>Mean</bold></td>
<td valign="top" align="center">86.32 &#x000B1; 0.62</td>
</tr>
<tr>
<td valign="top" align="left"><italic><bold>p</bold></italic><bold>-value</bold></td>
<td valign="top" align="center">0.0007</td>
</tr>
<tr>
<td valign="top" align="left"><xref ref-type="bibr" rid="B78">Tascon-Morales et al. (2022)</xref></td>
<td valign="top" align="center">83.00</td>
</tr>
<tr>
<td valign="top" align="left"><xref ref-type="bibr" rid="B79">Tascon-Morales et al. (2023)</xref></td>
<td valign="top" align="center">83.59 &#x000B1; 0.69</td>
</tr></tbody>
</table>
</table-wrap>
<p>Among the tested seeds, seed 42 achieved the highest Accuracy (87.41%), and we adopted this setting for all subsequent experiments, including the first-level classifier and the other components in our framework.</p>
<p>To determine whether the improvements over <xref ref-type="bibr" rid="B79">Tascon-Morales et al. (2023)</xref> are statistically significant, we performed a paired two-tailed t-test using the accuracies of our five seed experiments and the baseline of 83.69% (<xref ref-type="bibr" rid="B79">Tascon-Morales et al., 2023</xref>). The t-test yielded a t-statistic of 9.49 and a <italic>p</italic>-value of 0.0007, indicating that the performance improvement is statistically significant at the 0.01 level.</p>
<p>These results justify our choice of seed 42 for all subsequent experiments, as it consistently provided the best initialization and final Accuracy. Furthermore, the statistical test confirms that our method achieves a significant improvement over previous works.</p>
</sec>
<sec>
<title>4.5 Result and analysis</title>
<p>Our proposed model employs a two-level system. The initial level comprises a VQA model that inputs an image and a question as input and outputs the question type. We fine-tuned the model using the DME dataset, which contains four question types: grade, whole, region, and fovea or macula.</p>
<p>During this stage, we fine-tuned the GS-SWIN-ELECTRA model with a batch size of 32 and learning rate of 1.0 &#x000D7; 10<sup>&#x02212;4</sup>. Instead of answers, we replaced the classes with the question types. Remarkably, the model quickly converged within the first epoch, allowing us to train it just once. The model exhibited remarkable performance, achieving 99.85% for all performance metrics.</p>
<p>These results arise from several characteristics of the dataset. Firstly, the number of questions is relatively limited. Moreover, question types such as grade and fovea are directly reflected in the question text itself. In contrast, questions related to regions and wholes do not have distinct textual characteristics for classification. Instead, the classification between these two question types relies on the image provided, distinguishing between a whole image and a specific region based on the applied mask. <xref ref-type="table" rid="T4">Table 4</xref> presents the model&#x00027;s performance, while <xref ref-type="fig" rid="F9">Figure 9</xref> illustrates the model&#x00027;s confusion matrix.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>The result of the first-level model.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Answer</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>F1-Score</bold></th>
<th valign="top" align="center"><bold>Instances no</bold>.</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Fovea</td>
<td valign="top" align="center">1.0000</td>
<td valign="top" align="center">0.9924</td>
<td valign="top" align="center">0.9962</td>
<td valign="top" align="center">131</td>
</tr>
<tr>
<td valign="top" align="left">Grade</td>
<td valign="top" align="center">1.0000</td>
<td valign="top" align="center">0.9924</td>
<td valign="top" align="center">0.9962</td>
<td valign="top" align="center">131</td>
</tr>
<tr>
<td valign="top" align="left">Region</td>
<td valign="top" align="center">0.9989</td>
<td valign="top" align="center">1.0000</td>
<td valign="top" align="center">0.9995</td>
<td valign="top" align="center">918</td>
</tr>
<tr>
<td valign="top" align="left">Whole</td>
<td valign="top" align="center">0.9924</td>
<td valign="top" align="center">1.0000</td>
<td valign="top" align="center">0.9962</td>
<td valign="top" align="center">131</td>
</tr>
<tr>
<td valign="top" align="left"><bold>Accuracy</bold></td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center"><bold>0.9985</bold></td>
<td valign="top" align="center"><bold>1,311</bold></td>
</tr>
<tr>
<td valign="top" align="left"><bold>Macro Avg</bold></td>
<td valign="top" align="center">0.9978</td>
<td valign="top" align="center">0.9962</td>
<td valign="top" align="center">0.9970</td>
<td valign="top" align="center">1,311</td>
</tr>
<tr>
<td valign="top" align="left"><bold>Weighted Avg</bold></td>
<td valign="top" align="center">0.9985</td>
<td valign="top" align="center">0.9985</td>
<td valign="top" align="center">0.9985</td>
<td valign="top" align="center">1,311</td>
</tr></tbody>
</table>
</table-wrap>
<fig position="float" id="F9">
<label>Figure 9</label>
<caption><p>Onfusion matrix of the question-type classification model, where labels 0, 1, 2, and 3 correspond to whole, grade, fovea, and region, respectively. The results show that the model correctly classifies most question types, with the primary misclassification occurring when region questions are predicted as grade.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1646176-g0009.tif">
<alt-text>Confusion matrix titled &#x0201C;Question Type Confusion Matrix,&#x0201D; with true labels on the Y-axis and predicted labels on the X-axis, ranging from zero to three. The diagonal shows high accuracy, with most values concentrated at 130 for labels zero and one, 918 for label two, and 131 for label three. Misclassifications are minimal, with few off-diagonal entries. A color gradient from blue to red indicates value magnitude, with red representing higher values.</alt-text>
</graphic>
</fig>
<p>Achieving high performance in the lower levels is critical in our proposed multi-level framework, as these levels route visual questions to the appropriate upper levels. In our case, the first-level achieved 99.85% accuracy, which we attribute to the abovementioned reasons. However, this may not generalize to all problem domains. However, this high accuracy may not generalize across different problem domains. This sensitivity to the first-level performance highlights a potential limitation in our approach: the overall system&#x00027;s effectiveness depends on the performance of the initial routing decisions.</p>
<p>We also recognize a limitation in our statistical methods because we used a fixed random seed for all experiments. This approach guarantees reproducibility, but it removes natural variation and hinders the accurate estimation of variance or significance. Therefore, the reported improvements, like the 1% gain over SOTA baselines, should be viewed with caution. In future work, we intend to include repeated runs with different seeds and report confidence intervals to better evaluate performance stability and significance.</p>
<p>Moreover, our evaluation focused specifically on diabetic retinopathy in the context of disability, using the only publicly available dataset in this domain. We did not validate the framework on other datasets. In future work, we will expand the dataset to include more diverse DR cases and explore cross-lingual generalization by applying the method to data in additional languages.</p>
<p>Our current evaluation is limited to the DME-VQA dataset, which may constrain the generalizability of our findings. While this dataset is the only publicly available benchmark for diabetic retinopathy visual question answering that focuses on accessibility, future work will tackle this issue by doing cross-dataset evaluations with resources like extending the DME-VQA dataset or new public ones. Additionally, to measure real-world impact, we plan to include usability testing with visually impaired users.</p>
<p>After generating a prediction at the first-level, the model passes it through a switch function, which routes the visual question to the appropriate model at the second level. The second level comprises three models: SWIN-ELECTRA (with a batch size of 32 and a learning rate of 1.0 &#x000D7; 10<sup>&#x02212;4</sup>), GS-SWIN-ELECTRA (with a batch size of 32 and a learning rate of 1.0 &#x000D7; 10<sup>&#x02212;4</sup>), and GS-SWIN-ELECTRA (with a batch size of 16 and a learning rate of 1.0 &#x000D7; 10<sup>&#x02212;4</sup>). These models are specifically designed to handle different question types: grade, whole, fovea, and region, respectively. <xref ref-type="table" rid="T5">Table 5</xref> provides insights into the performance of the bi-level model across various evaluation metrics. To further visualize the prediction outcomes, <xref ref-type="fig" rid="F10">Figure 10</xref> displays the confusion matrix, illustrating the predictions for each answer.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>The result of the bi-level model.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Answer</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>F1-Score</bold></th>
<th valign="top" align="center"><bold>Instances no</bold>.</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">0</td>
<td valign="top" align="center">1.0000</td>
<td valign="top" align="center">0.7755</td>
<td valign="top" align="center">0.8736</td>
<td valign="top" align="center">49</td>
</tr>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="center">0.4444</td>
<td valign="top" align="center">0.8000</td>
<td valign="top" align="center">0.5714</td>
<td valign="top" align="center">15</td>
</tr>
<tr>
<td valign="top" align="left">2</td>
<td valign="top" align="center">0.9242</td>
<td valign="top" align="center">0.9104</td>
<td valign="top" align="center">0.9173</td>
<td valign="top" align="center">67</td>
</tr>
<tr>
<td valign="top" align="left">No</td>
<td valign="top" align="center">0.8798</td>
<td valign="top" align="center">0.9231</td>
<td valign="top" align="center">0.9009</td>
<td valign="top" align="center">650</td>
</tr>
<tr>
<td valign="top" align="left">Yes</td>
<td valign="top" align="center">0.8996</td>
<td valign="top" align="center">0.8453</td>
<td valign="top" align="center">0.8716</td>
<td valign="top" align="center">530</td>
</tr>
<tr>
<td valign="top" align="left"><bold>Accuracy</bold></td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center"><bold>0.8841</bold></td>
<td valign="top" align="center"><bold>1,311</bold></td>
</tr>
<tr>
<td valign="top" align="left"><bold>Macro Avg</bold></td>
<td valign="top" align="center">0.8296</td>
<td valign="top" align="center">0.8509</td>
<td valign="top" align="center">0.8270</td>
<td valign="top" align="center">1,311</td>
</tr>
<tr>
<td valign="top" align="left"><bold>Weighted Avg</bold></td>
<td valign="top" align="center">0.8896</td>
<td valign="top" align="center">0.8841</td>
<td valign="top" align="center">0.8851</td>
<td valign="top" align="center">1,311</td>
</tr></tbody>
</table>
</table-wrap>
<fig position="float" id="F10">
<label>Figure 10</label>
<caption><p>Confusion matrices for the SWIN-ELECTRA and GS-SWIN-ELECTRA models, where 1.0 &#x000D7; 10<sup>&#x02212;4</sup> denotes the learning rate, and 32 and 16 denote the batch sizes. In the answer labels, &#x0201C;3&#x0201D; corresponds to No and &#x0201C;4&#x0201D; corresponds to Yes. The results clearly show that the bi-level model performs equal to or better than its strongest component across most answers, with the exception of answer &#x0201C;0&#x0201D;, where the GS-SWIN-ELECTRA model achieves superior performance, correctly identifying 41 cases.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1646176-g0010.tif">
<alt-text>Four confusion matrix plots comparing different models: SWIN-ELECTRA-32, GS-SWIN-ELECTRA-32, GS-SWIN-ELECTRA-16, and Bi-Level. Each matrix visualizes true versus predicted labels from zero to four, with color gradients indicating the count of predictions, from blue (low) to red (high). SWIN-ELECTRA-32 shows 530 true positives at label three. GS-SWIN-ELECTRA-32 has 557 at the same label. GS-SWIN-ELECTRA-16 shows 597, and Bi-Level displays 600. Other counts vary per model.</alt-text>
</graphic>
</fig>
<p>To evaluate the effectiveness of our model, we analyze its performance to that of its individual components. This evaluation was conducted using several performance metrics, including F1 score, recall, and precision for each answer, as well as model accuracy, macro average precision, macro average recall, macro average F1 score, weighted average precision, weighted average recall, and weighted average F1-score. <xref ref-type="table" rid="T6">Tables 6</xref>, <xref ref-type="table" rid="T7">7</xref> present a comprehensive comparison of the proposed model and its individual components across these evaluation metrics. Furthermore, <xref ref-type="fig" rid="F10">Figure 10</xref> illustrates the confusion matrices, highlighting how each model distributed its predicted answers.</p>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>The result comparison per each answer for Bi-level model and its model components, where Model-1 is the SWIN-ELECTRA model with a 32 batch size and 1.0 &#x000D7; 10<sup>&#x02212;4</sup> learning rate, Model-2 is the GS-SWIN-ELECTRA model with 32 batch size and 1.0 &#x000D7; 10<sup>&#x02212;4</sup> learning rate, Model-3 is the GS-SWIN-ELECTRA model with 16 batch size and 1.0 &#x000D7; 10<sup>&#x02212;4</sup> learning rate, and Model-4 is the Bi-level model with 32 batch size and 1.0 &#x000D7; 10<sup>&#x02212;4</sup> learning rate</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Metric</bold></th>
<th valign="top" align="center"><bold>Answer</bold></th>
<th valign="top" align="center"><bold>Model 1</bold></th>
<th valign="top" align="center"><bold>Model 2</bold></th>
<th valign="top" align="center"><bold>Model 3</bold></th>
<th valign="top" align="center"><bold>Model 4</bold></th>
<th valign="top" align="center"><bold>Samples&#x00023;</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="5">Precision</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center"><bold>1.000</bold></td>
<td valign="top" align="center">0.9512</td>
<td valign="top" align="center">0.8913</td>
<td valign="top" align="center"><bold>1.0000</bold></td>
<td valign="top" align="center">49</td>
</tr>
 <tr>
<td valign="top" align="center">1</td>
<td valign="top" align="center"><bold>0.4444</bold></td>
<td valign="top" align="center">0.4091</td>
<td valign="top" align="center">0.4000</td>
<td valign="top" align="center"><bold>0.4444</bold></td>
<td valign="top" align="center">15</td>
</tr>
 <tr>
<td valign="top" align="center">2</td>
<td valign="top" align="center"><bold>0.9242</bold></td>
<td valign="top" align="center">0.8971</td>
<td valign="top" align="center">0.9077</td>
<td valign="top" align="center"><bold>0.9242</bold></td>
<td valign="top" align="center">67</td>
</tr>
 <tr>
<td valign="top" align="center">no</td>
<td valign="top" align="center">0.9029</td>
<td valign="top" align="center"><bold>0.9057</bold></td>
<td valign="top" align="center">0.8703</td>
<td valign="top" align="center">0.8798</td>
<td valign="top" align="center">650</td>
</tr>
 <tr>
<td valign="top" align="center">yes</td>
<td valign="top" align="center">0.7976</td>
<td valign="top" align="center">0.8354</td>
<td valign="top" align="center">0.8927</td>
<td valign="top" align="center"><bold>0.8996</bold></td>
<td valign="top" align="center">530</td>
</tr>
<tr>
<td valign="top" align="left" rowspan="5">Recall</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">0.7755</td>
<td valign="top" align="center">0.7959</td>
<td valign="top" align="center"><bold>0.8367</bold></td>
<td valign="top" align="center">0.7755</td>
<td valign="top" align="center">49</td>
</tr>
 <tr>
<td valign="top" align="center">1</td>
<td valign="top" align="center"><bold>0.8000</bold></td>
<td valign="top" align="center">0.6000</td>
<td valign="top" align="center">0.5333</td>
<td valign="top" align="center"><bold>0.8000</bold></td>
<td valign="top" align="center">15</td>
</tr>
 <tr>
<td valign="top" align="center">2</td>
<td valign="top" align="center"><bold>0.9104</bold></td>
<td valign="top" align="center"><bold>0.9104</bold></td>
<td valign="top" align="center">0.8806</td>
<td valign="top" align="center"><bold>0.9104</bold></td>
<td valign="top" align="center">67</td>
</tr>
 <tr>
<td valign="top" align="center">no</td>
<td valign="top" align="center">0.8154</td>
<td valign="top" align="center">0.8569</td>
<td valign="top" align="center">0.9185</td>
<td valign="top" align="center">0.9231</td>
<td valign="top" align="center">650</td>
</tr>
 <tr>
<td valign="top" align="center">yes</td>
<td valign="top" align="center"><bold>0.8925</bold></td>
<td valign="top" align="center">0.8906</td>
<td valign="top" align="center">0.8321</td>
<td valign="top" align="center">0.8453</td>
<td valign="top" align="center">530</td>
</tr>
<tr>
<td valign="top" align="left" rowspan="5">F1-score</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center"><bold>0.8736</bold></td>
<td valign="top" align="center">0.8667</td>
<td valign="top" align="center">0.8632</td>
<td valign="top" align="center"><bold>0.8736</bold></td>
<td valign="top" align="center">49</td>
</tr>
 <tr>
<td valign="top" align="center">1</td>
<td valign="top" align="center"><bold>0.5714</bold></td>
<td valign="top" align="center">0.4865</td>
<td valign="top" align="center">0.4571</td>
<td valign="top" align="center"><bold>0.5714</bold></td>
<td valign="top" align="center">15</td>
</tr>
 <tr>
<td valign="top" align="center">2</td>
<td valign="top" align="center"><bold>0.9173</bold></td>
<td valign="top" align="center">0.9037</td>
<td valign="top" align="center">0.8939</td>
<td valign="top" align="center"><bold>0.9173</bold></td>
<td valign="top" align="center">67</td>
</tr>
 <tr>
<td valign="top" align="center">no</td>
<td valign="top" align="center">0.8569</td>
<td valign="top" align="center">0.8806</td>
<td valign="top" align="center">0.8937</td>
<td valign="top" align="center"><bold>0.9009</bold></td>
<td valign="top" align="center">650</td>
</tr>
<tr>
<td valign="top" align="center">yes</td>
<td valign="top" align="center">0.8424</td>
<td valign="top" align="center">0.8621</td>
<td valign="top" align="center">0.8613</td>
<td valign="top" align="center"><bold>0.8716</bold></td>
<td valign="top" align="center">530</td>
</tr></tbody>
</table>
</table-wrap>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>The result comparison of Bi-level model and its model components.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Metric</bold></th>
<th valign="top" align="center"><bold>Model 1 <xref ref-type="bibr" rid="B8">Al-Hadhrami et al. (2023)</xref></bold></th>
<th valign="top" align="center"><bold>Model 2 <xref ref-type="bibr" rid="B8">Al-Hadhrami et al. (2023)</xref></bold></th>
<th valign="top" align="center"><bold>Model 3 <xref ref-type="bibr" rid="B8">Al-Hadhrami et al. (2023)</xref></bold></th>
<th valign="top" align="center"><bold>Model 4</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Macro avg Precision</td>
<td valign="top" align="center">0.8138</td>
<td valign="top" align="center">0.7997</td>
<td valign="top" align="center">0.7924</td>
<td valign="top" align="center">0.8296 &#x000B1; 0.0189 <italic>p</italic>-value = 0.0017</td>
</tr>
<tr>
<td valign="top" align="left">Macro avg Recall</td>
<td valign="top" align="center">0.8388</td>
<td valign="top" align="center">0.8108</td>
<td valign="top" align="center">0.8002</td>
<td valign="top" align="center">0.8509 &#x000B1; 0.0251 <italic>p</italic>-value = 0.0014</td>
</tr>
<tr>
<td valign="top" align="left">Macro avg F1-score</td>
<td valign="top" align="center">0.8123</td>
<td valign="top" align="center">0.7999</td>
<td valign="top" align="center">0.7939</td>
<td valign="top" align="center">0.8270 &#x000B1; 0.0234 <italic>p</italic>-value = 0.0033</td>
</tr>
<tr>
<td valign="top" align="left">Weighted avg Precision</td>
<td valign="top" align="center">0.8598</td>
<td valign="top" align="center">0.8729</td>
<td valign="top" align="center">0.8767</td>
<td valign="top" align="center">0.8896 &#x000B1; 0.0078 <italic>p</italic>-value = 0.0015</td>
</tr>
<tr>
<td valign="top" align="left">Weighted avg Recall</td>
<td valign="top" align="center">0.8497</td>
<td valign="top" align="center">0.8680</td>
<td valign="top" align="center">0.8741</td>
<td valign="top" align="center">0.8841 &#x000B1; 0.0059 <italic>p</italic>-value = 0.0015</td>
</tr>
<tr>
<td valign="top" align="left">Weighted avg F1-score</td>
<td valign="top" align="center">0.8515</td>
<td valign="top" align="center">0.8693</td>
<td valign="top" align="center">0.8745</td>
<td valign="top" align="center">0.8851 &#x000B1; 0.0067 <italic>p</italic>-value = 0.0017</td>
</tr>
<tr>
<td valign="top" align="left">Accuracy</td>
<td valign="top" align="center">0.8497</td>
<td valign="top" align="center">0.8680</td>
<td valign="top" align="center">0.8741</td>
<td valign="top" align="center">0.8841 &#x000B1; 0.0059 <italic>p</italic>-value = 0.0015</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Model 1 is the SWIN-ELECTRA model with 1.0 &#x000D7; 10<sup>&#x02212;4</sup> learning rate and 32 batch size (<xref ref-type="bibr" rid="B8">Al-Hadhrami et al., 2023</xref>). Model 2 is GS-SWIN-ELECTRA model with 1.0 &#x000D7; 10<sup>&#x02212;4</sup> learning rate and 32 batch size (<xref ref-type="bibr" rid="B8">Al-Hadhrami et al., 2023</xref>). Model 3 is the GS-SWIN-ELECTRA model with 1.0 &#x000D7; 10<sup>&#x02212;4</sup> learning rate and 16 batch size (<xref ref-type="bibr" rid="B8">Al-Hadhrami et al., 2023</xref>). Model 4 is the Bi-level model with 1.0 &#x000D7; 10<sup>&#x02212;4</sup> learning rate and 32 batch size.</p>
</table-wrap-foot>
</table-wrap>
<p>The bi-level model consistently achieves higher accuracy for each question type compared to its component models. We selected the component models based on their superior performance in those specific question types. This strategy allowed the bilevel model to achieve the highest performance among the component models and improve its overall accuracy. For each question type, our proposed bi-level VQA model consistently achieves the highest accuracy compared to its individual component models, demonstrating its effectiveness and contributing to the best overall performance across the dataset. In <xref ref-type="table" rid="T8">Table 8</xref>, we present the performance comparison between the bi-level model and its component models, providing an insightful overview of their respective performances.</p>
<table-wrap position="float" id="T8">
<label>Table 8</label>
<caption><p>The result comparison of Bi-level model and its model components based on question types, where Model-1 is the SWIN-ELECTRA model with 1.0 &#x000D7; 10<sup>&#x02212;4</sup> learning rate and 32 batch size, Model 2 is GS-SWIN-ELECTRA model with 1.0 &#x000D7; 10<sup>&#x02212;4</sup> learning rate and 32 batch size, Model-3 is the GS-SWIN-ELECTRA model with 1.0 &#x000D7; 10<sup>&#x02212;4</sup> learning rate and 16 batch size, Model 4 is the Bi-level model with 1.0 &#x000D7; 10<sup>&#x02212;4</sup> learning rate and 32 batch size.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center"><bold>Overall</bold></th>
<th valign="top" align="center"><bold>Grade</bold></th>
<th valign="top" align="center"><bold>Whole</bold></th>
<th valign="top" align="center"><bold>Macula</bold></th>
<th valign="top" align="center"><bold>Region</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SOTA 2022 <xref ref-type="bibr" rid="B78">Tascon-Morales et al. (2022)</xref></td>
<td valign="top" align="center">83.49</td>
<td valign="top" align="center">80.69</td>
<td valign="top" align="center">84.96</td>
<td valign="top" align="center">87.18</td>
<td valign="top" align="center">83.16</td>
</tr>
<tr>
<td valign="top" align="left">SOTA 2023 <xref ref-type="bibr" rid="B79">Tascon-Morales et al. (2023)</xref></td>
<td valign="top" align="center">83.59 &#x000B1; 0.69</td>
<td valign="top" align="center">80.15 &#x000B1; 0.95</td>
<td valign="top" align="center">86.22 &#x000B1; 1.67</td>
<td valign="top" align="center">88.18 &#x000B1; 1.07</td>
<td valign="top" align="center">82.62 &#x000B1; 1.02</td>
</tr>
<tr>
<td valign="top" align="left">Model 1</td>
<td valign="top" align="center">84.97</td>
<td valign="top" align="center"><bold>84.73</bold></td>
<td valign="top" align="center">90.84</td>
<td valign="top" align="center">85.29</td>
<td valign="top" align="center">83.22</td>
</tr>
<tr>
<td valign="top" align="left">Model 2</td>
<td valign="top" align="center">86.80</td>
<td valign="top" align="center">83.21</td>
<td valign="top" align="center"><bold>92.37</bold></td>
<td valign="top" align="center"><bold>90.84</bold></td>
<td valign="top" align="center">85.95</td>
</tr>
<tr>
<td valign="top" align="left">Model 3</td>
<td valign="top" align="center">87.41</td>
<td valign="top" align="center">82.44</td>
<td valign="top" align="center">88.55</td>
<td valign="top" align="center">87.02</td>
<td valign="top" align="center"><bold>88.02</bold></td>
</tr>
<tr>
<td valign="top" align="left">Model 4 (proposed bi-level)</td>
<td valign="top" align="center"><bold>88.41&#x000B1;</bold> <bold>0.0059</bold></td>
<td valign="top" align="center"><bold>84.73</bold> <bold>&#x000B1;</bold> <bold>0.0125</bold></td>
<td valign="top" align="center"><bold>92.37</bold> <bold>&#x000B1;</bold> <bold>0.0185</bold></td>
<td valign="top" align="center"><bold>90.84</bold> <bold>&#x000B1;</bold> <bold>0.0211</bold></td>
<td valign="top" align="center"><bold>88.02</bold> <bold>&#x000B1;</bold> <bold>0.0133</bold></td>
</tr>
<tr>
<td/>
<td valign="top" align="center"><italic><bold>p</bold></italic><bold>-value= 0.0015</bold></td>
<td valign="top" align="center"><italic><bold>p</bold></italic><bold>-value:</bold> <bold>1.66 &#x000D7; 10<sup>&#x02212;8</sup></bold></td>
<td valign="top" align="center"><italic><bold>p</bold></italic><bold>-value:</bold> <bold>5.11 &#x000D7; 10<sup>&#x02212;8</sup></bold></td>
<td valign="top" align="center"><italic><bold>p</bold></italic><bold>-value:</bold> <bold>7.80 &#x000D7; 10<sup>&#x02212;8</sup></bold></td>
<td valign="top" align="center"><italic><bold>p</bold></italic><bold>-value:</bold> <bold>1.39 &#x000D7; 10<sup>&#x02212;8</sup></bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>In addition, the model is compared with SOTA 2022 (<xref ref-type="bibr" rid="B78">Tascon-Morales et al., 2022</xref>) and SOTA 2023 (<xref ref-type="bibr" rid="B79">Tascon-Morales et al., 2023</xref>).</p>
</table-wrap-foot>
</table-wrap>
<p>Our framework we introduced in this work is designed to be modular and adaptable, enabling its application beyond the DME-VQA dataset.</p>
<p>Its generalizability stems from its core design, which emphasizes structured understanding and decomposition of the problem domain. The framework can be adapted to various medical imaging tasks or other vision-language problems by analyzing the dataset and identifying distinct question types or visual characteristics. The bi-level architecture offers flexible integration of specialized models for distinct sub-tasks, enabling its extension to new datasets with varying class or diagnostic objectives distributions. Furthermore, this decomposition strategy improves interpretability and reduces the learning complexity, especially in scenarios with limited annotated data. The framework enables more efficient learning by transforming a complex VQA task into smaller, more focused subtasks, potentially improving performance and generalization even when data is scarce.</p>
<p>Furthermore, we evaluated the model on a dataset that incorporates real-world variability, including noise, inconsistent illumination, and diverse imaging angles. This diversity contributes to the robustness and reliability of the proposed system in clinically realistic settings.</p>
<p><xref ref-type="fig" rid="F11">Figure 11</xref> Grad-CAM shows visualizations of our bi-level VQA model, which has an impressive accuracy of 88.41%, effectively highlight the model&#x00027;s ability to focus on critical regions in retinal images for diabetic retinopathy classification. In the correctly predicted cases, as shown in <xref ref-type="fig" rid="F11">Figures 11a</xref>, <xref ref-type="fig" rid="F11">b</xref>, the heatmaps show strong attention to essential features such as hard axudates, aligning with the ground truth labels. This finding demonstrates the model&#x00027;s capacity to identify and leverage key visual cues, further validating its robustness in making accurate predictions. These visualizations confirm that the model is consistently able to attend to relevant areas of the image, supporting its high performance.</p>
<fig position="float" id="F11">
<label>Figure 11</label>
<caption><p>Grad-CAM visualizations of the bi-level VQA model (accuracy: 88.41%), highlighting its ability to attend to critical retinal regions for diabetic retinopathy classification. In correctly predicted cases <bold>(a, b)</bold>, the model focuses on key features such as hard exudates, aligning with ground truth labels. In misclassified cases <bold>(c&#x02013;e)</bold>, attention is diverted to irrelevant regions or image noise, occasionally leading to errors. Notably, in <bold>(e)</bold>, the model attends correctly to the lesion but interprets it as &#x0201C;No&#x0201D;. These visualizations demonstrate both the robustness of the model and areas needing refinement to improve attention consistency.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1646176-g0011.tif">
<alt-text>Five panels labeled (a) to (e), each containing three images: an original retinal image, a Grad-CAM heatmap highlighting areas of focus, and an overlay of both. Panels (a) to (e) compare predicted vs. true classifications related to eye conditions, with discrepancies noted in some predictions.</alt-text>
</graphic>
</fig>
<p>On the other hand, the incorrect predictions appear to stem from the model focusing on irrelevant regions as shown in <xref ref-type="fig" rid="F11">Figure 11c</xref>. In the case where the model predicts &#x0201C;yes&#x0201D; incorrectly, the heatmap shows attention to parts of the image that are not relevant to the key features of diabetic retinopathy, suggesting that the model might be misinterpreting image details. In the last image (d), the error could be attributed to image noise, which may have caused the model to focus on non-essential features, leading to an incorrect classification. In (e), the attention is correctly focused on the hard exudates, but it is interpreted as &#x0201C;no&#x0201D;. This case requires further analysis and study, which we will address in future work. These observations highlight areas for future work to refine the model&#x00027;s attention mechanism, improving its ability to focus on the most relevant features and reducing the impact of noise.</p>
</sec>
<sec>
<title>4.6 Ethical considerations for medical VQA</title>
<p>Ethical concerns in medical VQA include user privacy, deployment implications, and potential biases. Privacy safeguards are critical, as these systems handle sensitive patient data, requiring compliance with frameworks like HIPAA and informed consent protocols to protect autonomy and dignity (<xref ref-type="bibr" rid="B58">Majumder and Guerrini, 2016</xref>; <xref ref-type="bibr" rid="B5">Adeniyi et al., 2024</xref>; <xref ref-type="bibr" rid="B23">De Lusignan et al., 2015</xref>). Deployment strategies must prioritize equitable access, addressing cost barriers to ensure the widespread availability of these technologies (<xref ref-type="bibr" rid="B5">Adeniyi et al., 2024</xref>).</p>
<p>Algorithmic bias is another significant challenge, as it can lead to inequitable outcomes across diverse populations. Biases often arise from non-representative datasets or flawed model development processes, exacerbating healthcare disparities. Mitigation strategies include using diverse datasets, statistical debiasing methods, and rigorous validation through clinical trials (<xref ref-type="bibr" rid="B74">Smith et al., 2023</xref>; <xref ref-type="bibr" rid="B20">Cross et al., 2024</xref>). Addressing these ethical considerations ensures medical VQA systems contribute positively to society while minimizing risks.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="s5">
<title>5 Conclusion</title>
<p>Visual disabilities affect the ability of individuals to perceive and interpret visual information, highlighting the need for advanced solutions to solve these challenges. This paper introduces a multi-level VQA technique that leverages multiple VQA models for enhancing the VQA performance. We propose a bi-level, designed to enhance VQA performance. The bi-level model consists of two levels. The type of question is classified in the first-level, and the visual question is answered in the second level. The model employs a switch function to forward the visual question to the proper component model according to its question type. Through this multi-level VQA model, we demonstrate the efficacy of incorporating different levels and component models to enhance the accuracy of VQA systems. We believe this approach represents a step forward in making visual information more accessible to individuals with visual impairments.</p>
<p>Looking ahead, future work will focus on optimizing the number and structure of the levels to maximize performance. Exploring additional hierarchical levels may improve accuracy by enabling more fine-grained routing of visual questions. Moreover, we aim to conduct usability and accessibility evaluations involving users with visual impairments to validate the system&#x00027;s practical impact. Lastly, we plan to extend the framework to support multilingual datasets and evaluate its generalizability across diverse linguistic and demographic populations as new diabetic retinopathy VQA datasets become available. In addition, we aim to implement the full system and measure the user satisfication and system usibility.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: ZENODO repository at <ext-link ext-link-type="uri" xlink:href="https://zenodo.org/records/6784358">https://zenodo.org/records/6784358</ext-link>.</p>
</sec>
<sec sec-type="ethics-statement" id="s7">
<title>Ethics statement</title>
<p>Ethical approval was not required for this study since it used a publicly available dataset of medical images. No human participants were directly involved, and all data were fully anonymized and collected in accordance with relevant ethical guidelines and institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>SA-A: Project administration, Resources, Supervision, Validation, Writing &#x02013; review &#x00026; editing. SA-H: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Resources, Software, Validation, Visualization, Writing &#x02013; original draft. SA: Resources, Validation, Visualization, Writing &#x02013; original draft.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research and/or publication of this article.</p>
</sec>
<ack><p>The authors would like to acknowledge the Department of Computer Science, College of Computer and Information Sciences, King Saud University, for its support and collaboration.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that Gen AI was used in the creation of this manuscript. ChatGPT 3.5, ChatGPT 4.0, and Grammarly were used to enhance and proofread the manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abacha</surname> <given-names>A.</given-names></name> <name><surname>Datla</surname> <given-names>V.</given-names></name> <name><surname>Hasan</surname> <given-names>S.</given-names></name> <name><surname>Demner-Fushman</surname> <given-names>D.</given-names></name> <name><surname>M&#x000FC;ller</surname> <given-names>H.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Overview of the vqa-med task at imageclef 2020: Visual question answering and generation in the medical domain,&#x0201D;</article-title> in <source>Proceedings of the CLEF 2020&#x02013;Conference and Labs of the Evaluation Forum</source>, 1-9.</citation>
</ref>
<ref id="B2">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Abacha</surname> <given-names>A.</given-names></name> <name><surname>Hasan</surname> <given-names>S.</given-names></name> <name><surname>Datla</surname> <given-names>V.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Demner-Fushman</surname> <given-names>D.</given-names></name> <name><surname>M&#x000FC;ller</surname> <given-names>H.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;VQA-Med: Overview of the medical visual question answering task at imageclef 2019,&#x0201D;</article-title> in <source>CEUR Workshop Proceedings</source> (<publisher-loc>London</publisher-loc>: <publisher-name>CEUR-WS Team</publisher-name>).</citation>
</ref>
<ref id="B3">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Abacha</surname> <given-names>A. B.</given-names></name> <name><surname>Gayen</surname> <given-names>S.</given-names></name> <name><surname>Lau</surname> <given-names>J. J.</given-names></name> <name><surname>Rajaraman</surname> <given-names>S.</given-names></name> <name><surname>Demner-Fushman</surname> <given-names>D.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;NLM at imageclef 2018 visual question answering in the medical domain,&#x0201D;</article-title> in <source>Technical Report</source> (<publisher-loc>Aachen</publisher-loc>: <publisher-name>CEUR-WS Team</publisher-name>).</citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abr&#x000E0;moff</surname> <given-names>M. D.</given-names></name> <name><surname>Lavin</surname> <given-names>P. T.</given-names></name> <name><surname>Birch</surname> <given-names>M.</given-names></name> <name><surname>Shah</surname> <given-names>J. N.</given-names></name> <name><surname>Folk</surname> <given-names>J. C.</given-names></name></person-group> (<year>2018</year>). <article-title>Pivotal trial of an autonomous ai-based diagnostic system for detection of diabetic retinopathy in primary care offices</article-title>. <source>Nat. Digit. Med</source>. <volume>1</volume>, <fpage>1</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1038/s41746-018-0040-6</pub-id><pub-id pub-id-type="pmid">31304320</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Adeniyi</surname> <given-names>A. O.</given-names></name> <name><surname>Arowoogun</surname> <given-names>J. O.</given-names></name> <name><surname>Okolo</surname> <given-names>C. A.</given-names></name> <name><surname>Chidi</surname> <given-names>R.</given-names></name> <name><surname>Babawarun</surname> <given-names>O.</given-names></name></person-group> (<year>2024</year>). <article-title>Ethical considerations in healthcare it: a review of data privacy and patient consent issues</article-title>. <source>World J. Adv. Res. Rev</source>. <volume>21</volume>, <fpage>1660</fpage>&#x02013;<lpage>1668</lpage>. <pub-id pub-id-type="doi">10.30574/wjarr.2024.21.2.0593</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ainary</surname> <given-names>B.</given-names></name></person-group> (<year>2025</year>). <article-title>Audo-sight: enabling ambient interaction for blind and visually impaired individuals</article-title>. <source>arXiv</source> [preprint] arXiv:2505.00153. <pub-id pub-id-type="doi">10.48550/arXiv.2505.00153</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Alhadhrami</surname> <given-names>S.</given-names></name> <name><surname>Alnafessah</surname> <given-names>A.</given-names></name> <name><surname>Al-Ammar</surname> <given-names>M.</given-names></name> <name><surname>Alarifi</surname> <given-names>A.</given-names></name> <name><surname>Al-Khalifa</surname> <given-names>H.</given-names></name> <name><surname>Alsaleh</surname> <given-names>M.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;UWB indoor tracking system for visually impaired people,&#x0201D;</article-title> in <source>Proceedings of the 13th International Conference on Advances in Mobile Computing and Multimedia</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>54</fpage>&#x02013;<lpage>62</lpage>. <pub-id pub-id-type="doi">10.1145/2837126.2837141</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Al-Hadhrami</surname> <given-names>S.</given-names></name> <name><surname>Menai</surname> <given-names>M. E. B.</given-names></name> <name><surname>Al-Ahmadi</surname> <given-names>S.</given-names></name> <name><surname>Alnafessah</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>An effective med-vqa method using a transformer with weights fusion of multiple fine-tuned models</article-title>. <source>Appl. Sci</source>. <volume>13</volume>:<fpage>9735</fpage>. <pub-id pub-id-type="doi">10.3390/app13179735</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Allaouzi</surname> <given-names>I.</given-names></name> <name><surname>Benamrou</surname> <given-names>B.</given-names></name> <name><surname>Benamrou</surname> <given-names>M.</given-names></name> <name><surname>Ahmed</surname> <given-names>M. B.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Deep neural networks and decision tree classifier for visual question answering in the medical domain,&#x0201D;</article-title> in <source>Technical Report</source> (<publisher-loc>Aachen</publisher-loc>: <publisher-name>CEUR-WS Team</publisher-name>).</citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Andreas</surname> <given-names>J.</given-names></name> <name><surname>Rohrbach</surname> <given-names>M.</given-names></name> <name><surname>Darrell</surname> <given-names>T.</given-names></name> <name><surname>Klein</surname> <given-names>D.</given-names></name></person-group> (<year>2016a</year>). <article-title>Learning to compose neural networks for question answering</article-title>. <source>arXiv</source> [preprint] arXiv:1601.01705. <pub-id pub-id-type="doi">10.18653/v1/N16-1181</pub-id></citation>
</ref>
<ref id="B11">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Andreas</surname> <given-names>J.</given-names></name> <name><surname>Rohrbach</surname> <given-names>M.</given-names></name> <name><surname>Darrell</surname> <given-names>T.</given-names></name> <name><surname>Klein</surname> <given-names>D.</given-names></name></person-group> (<year>2016b</year>). <article-title>&#x0201C;Neural module networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>39</fpage>&#x02013;<lpage>48</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2016.12</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Antol</surname> <given-names>S.</given-names></name> <name><surname>Agrawal</surname> <given-names>A.</given-names></name> <name><surname>Lu</surname> <given-names>J.</given-names></name> <name><surname>Mitchell</surname> <given-names>M.</given-names></name> <name><surname>Batra</surname> <given-names>D.</given-names></name> <name><surname>Zitnick</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>&#x0201C;VQA: Visual question answering,&#x0201D;</article-title> in <source>Proceedings of the IEEE International Conference on Computer Vision</source> (<publisher-loc>Big Island, HI</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>2425</fpage>&#x02013;<lpage>2433</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV.2015.279</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bai</surname> <given-names>Y.</given-names></name> <name><surname>Fu</surname> <given-names>J.</given-names></name> <name><surname>Zhao</surname> <given-names>T.</given-names></name> <name><surname>Mei</surname> <given-names>T.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Deep attention neural tensor network for visual question answering,&#x0201D;</article-title> in <source>Proceedings of the European Conference on Computer Vision (ECCV)</source>, <fpage>20</fpage>&#x02013;<lpage>35</lpage>.</citation>
</ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ben-Younes</surname> <given-names>H.</given-names></name> <name><surname>Cadene</surname> <given-names>R.</given-names></name> <name><surname>Cord</surname> <given-names>M.</given-names></name> <name><surname>Thome</surname> <given-names>N.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Mutan: Multimodal tucker fusion for visual question answering,&#x0201D;</article-title> in <source>Proceedings of the IEEE International Conference on Computer Vision</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>2612</fpage>&#x02013;<lpage>2620</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Bounaama</surname> <given-names>R.</given-names></name> <name><surname>Abderrahim</surname> <given-names>M. E. A.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Tlemcen university at imageclef 2019 visual question answering task,&#x0201D;</article-title> in <source>Proceedings of the CLEF (Working Notes)</source> (<publisher-loc>Lugano</publisher-loc>: <publisher-name>CEUR-WS.org</publisher-name>).</citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>K.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Chen</surname> <given-names>L.</given-names></name> <name><surname>Gao</surname> <given-names>H.</given-names></name> <name><surname>Xu</surname> <given-names>W.</given-names></name> <name><surname>Nevatia</surname> <given-names>R.</given-names></name></person-group> (<year>2015</year>). <article-title>Abc-cnn: An attention based convolutional neural network for visual question answering</article-title>. <source>arXiv</source> [preprint] arXiv:1511.05960v2. <pub-id pub-id-type="doi">10.48550/arXiv.1511.05960</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>L.</given-names></name> <name><surname>Yan</surname> <given-names>X.</given-names></name> <name><surname>Xiao</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Pu</surname> <given-names>S.</given-names></name> <name><surname>Zhuang</surname> <given-names>Y.</given-names></name></person-group> (<year>2020a</year>). <article-title>&#x0201C;Counterfactual samples synthesizing for robust visual question answering,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-loc>Seattle, WA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>10800</fpage>&#x02013;<lpage>10809</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.-C.</given-names></name> <name><surname>Li</surname> <given-names>L.</given-names></name> <name><surname>Yu</surname> <given-names>L.</given-names></name> <name><surname>El Kholy</surname> <given-names>A.</given-names></name> <name><surname>Ahmed</surname> <given-names>F.</given-names></name> <name><surname>Gan</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2020b</year>). <article-title>&#x0201C;Uniter: Universal image-text representation learning,&#x0201D;</article-title> in <source>Computer Vision&#x02013;ECCV 2020, 16th European Conference</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>23</fpage>&#x02013;<lpage>28</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Cong</surname> <given-names>F.</given-names></name> <name><surname>Xu</surname> <given-names>S.</given-names></name> <name><surname>Guo</surname> <given-names>L.</given-names></name> <name><surname>Tian</surname> <given-names>Y.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Caption-aware medical vqa via semantic focusing and progressive cross-modality comprehension,&#x0201D;</article-title> in <source>Proceedings of the 30th ACM International Conference on Multimedia</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>3569</fpage>&#x02013;<lpage>3577</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cross</surname> <given-names>J. L.</given-names></name> <name><surname>Choma</surname> <given-names>M. A.</given-names></name> <name><surname>Onofrey</surname> <given-names>J. A.</given-names></name></person-group> (<year>2024</year>). <article-title>Bias in medical AI: implications for clinical decision-making</article-title>. <source>PLOS Digital Health</source> <volume>3</volume>:<fpage>e0000651</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pdig.0000651</pub-id><pub-id pub-id-type="pmid">39509461</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Dalal</surname> <given-names>N.</given-names></name> <name><surname>Triggs</surname> <given-names>B</given-names></name></person-group>. (<year>2005</year>). <article-title>&#x0201C;Histograms of oriented gradients for human detection,&#x0201D;</article-title> in <source>2005 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR&#x00027;05)</source> (<publisher-loc>San Diego, CA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>886</fpage>&#x02013;<lpage>893</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>de Freitas</surname> <given-names>M. P.</given-names></name> <name><surname>Piai</surname> <given-names>V. A.</given-names></name> <name><surname>Farias</surname> <given-names>R. H.</given-names></name> <name><surname>Fernandes</surname> <given-names>A. M.</given-names></name> <name><surname>de Moraes Rossetto</surname> <given-names>A. G.</given-names></name> <name><surname>Leithardt</surname> <given-names>V. R. Q.</given-names></name></person-group> (<year>2022</year>). <article-title>Artificial intelligence of things applied to assistive technology: a systematic literature review</article-title>. <source>Sensors</source> <volume>22</volume>:<fpage>8531</fpage>. <pub-id pub-id-type="doi">10.3390/s22218531</pub-id><pub-id pub-id-type="pmid">36366227</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>De Lusignan</surname> <given-names>S.</given-names></name> <name><surname>Liyanage</surname> <given-names>H.</given-names></name> <name><surname>Di Iorio</surname> <given-names>C. T.</given-names></name> <name><surname>Chan</surname> <given-names>T.</given-names></name> <name><surname>Liaw</surname> <given-names>S.-T.</given-names></name></person-group> (<year>2015</year>). <article-title>Using routinely collected health data for surveillance, quality improvement and research: Framework and key questions to assess ethics and privacy and enable data access</article-title>. <source>BMJ Health Care Inform</source>. <volume>22</volume>:<fpage>845</fpage>. <pub-id pub-id-type="doi">10.14236/jhi.v22i4.845</pub-id><pub-id pub-id-type="pmid">26855276</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Decenciere</surname> <given-names>E.</given-names></name> <name><surname>Cazuguel</surname> <given-names>G.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Thibault</surname> <given-names>G.</given-names></name> <name><surname>Klein</surname> <given-names>J.-C.</given-names></name> <name><surname>Meyer</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Teleophta: Machine learning and image processing methods for teleophthalmology</article-title>. <source>IRBM</source>, <volume>34</volume>, <fpage>196</fpage>&#x02013;<lpage>203</lpage>. <pub-id pub-id-type="doi">10.1016/j.irbm.2013.01.010</pub-id></citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Devlin</surname> <given-names>J.</given-names></name> <name><surname>Chang</surname> <given-names>M.-W.</given-names></name> <name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Toutanova</surname> <given-names>K.</given-names></name></person-group> (<year>2019</year>). <article-title>Bert: Pre-training of deep bidirectional transformers for language understanding</article-title>. <source>arXiv</source> [preprint] arXiv:1810.04805. <pub-id pub-id-type="doi">10.48550/arXiv.1810.04805</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Do</surname> <given-names>T.</given-names></name> <name><surname>Nguyen</surname> <given-names>B.</given-names></name> <name><surname>Tjiputra</surname> <given-names>E.</given-names></name> <name><surname>Tran</surname> <given-names>M.</given-names></name> <name><surname>Tran</surname> <given-names>Q.</given-names></name> <name><surname>Nguyen</surname> <given-names>A.</given-names></name></person-group> (<year>2021</year>). <article-title>Multiple meta-model quantifying for medical visual question answering</article-title>. <source>arXiv</source> [preprint] arXiv:2105.08913. <pub-id pub-id-type="doi">10.1007/978-3-030-87240-3_7</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Eslami</surname> <given-names>S.</given-names></name> <name><surname>de Melo</surname> <given-names>G.</given-names></name> <name><surname>Meinel</surname> <given-names>C.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Teams at vqa-med 2021: BBN-orchestra for long-tailed medical visual question answering,&#x0201D;</article-title> in <source>Working Notes of CLEF, volume 201</source> (<publisher-loc>Aachen</publisher-loc>: <publisher-name>CEUR-WS Team</publisher-name>).</citation>
</ref>
<ref id="B28">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Fukui</surname> <given-names>A.</given-names></name> <name><surname>Park</surname> <given-names>D. H.</given-names></name> <name><surname>Yang</surname> <given-names>D.</given-names></name> <name><surname>Rohrbach</surname> <given-names>A.</given-names></name> <name><surname>Darrell</surname> <given-names>T.</given-names></name> <name><surname>Rohrbach</surname> <given-names>M.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Multimodal compact bilinear pooling for visual question answering and visual grounding,&#x0201D;</article-title> in <source>Proceedings of the 2016 Conference on Empirical Methods in Natural Language Processing</source> (<publisher-loc>Stroudsburg, PA</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>457</fpage>&#x02013;<lpage>468</lpage>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>L.</given-names></name> <name><surname>Zeng</surname> <given-names>P.</given-names></name> <name><surname>Song</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Mei</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Structured two-stream attention network for video question answering</article-title>. <source>Proc. AAAI Conf. Artif. Intellig</source>. <volume>33</volume>, <fpage>6391</fpage>&#x02013;<lpage>6398</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v33i01.33016391</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Gong</surname> <given-names>H.</given-names></name> <name><surname>Huang</surname> <given-names>R.</given-names></name> <name><surname>Chen</surname> <given-names>G.</given-names></name> <name><surname>Li</surname> <given-names>G.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;SYSU-HCP at VQA-Med 2021: a data-centric model with efficient training methodology for medical visual question answering,&#x0201D;</article-title> in <source>Working Notes of CLEF, volume 201</source> (<publisher-loc>Aachen</publisher-loc>: <publisher-name>CEUR-WS Team</publisher-name>).</citation>
</ref>
<ref id="B31">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Gu</surname> <given-names>T.</given-names></name> <name><surname>Yang</surname> <given-names>K.</given-names></name> <name><surname>Liu</surname> <given-names>D.</given-names></name> <name><surname>Cai</surname> <given-names>W.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;LaPA: Latent prompt assist model for medical visual question answering,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Workshops</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>4971</fpage>&#x02013;<lpage>4980</lpage>.</citation>
</ref>
<ref id="B32">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Gurari</surname> <given-names>D.</given-names></name> <name><surname>Li</surname> <given-names>Q.</given-names></name> <name><surname>Stangl</surname> <given-names>A. J.</given-names></name> <name><surname>Guo</surname> <given-names>A.</given-names></name> <name><surname>Lin</surname> <given-names>C.</given-names></name> <name><surname>Grauman</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>&#x0201C;Vizwiz grand challenge: Answering visual questions from blind people,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>3608</fpage>&#x02013;<lpage>3617</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Haridas</surname> <given-names>H. T.</given-names></name> <name><surname>Fouda</surname> <given-names>M. M.</given-names></name> <name><surname>Fadlullah</surname> <given-names>Z. M.</given-names></name> <name><surname>Mahmoud</surname> <given-names>M.</given-names></name> <name><surname>ElHalawany</surname> <given-names>B. M.</given-names></name> <name><surname>Guizani</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;MED-GPVS: A deep learning-based joint biomedical image classification and visual question answering system for precision e-health,&#x0201D;</article-title> in <source>Proceedings of the ICC 2022&#x02013;IEEE International Conference on Communications</source> (<publisher-loc>Seoul</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>3838</fpage>&#x02013;<lpage>3843</lpage>.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Deep residual learning for image recognition,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>, <fpage>770</fpage>&#x02013;<lpage>778</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Mou</surname> <given-names>L.</given-names></name> <name><surname>Xing</surname> <given-names>E.</given-names></name> <name><surname>Xie</surname> <given-names>P.</given-names></name></person-group> (<year>2020a</year>). <article-title>Challenge-pathology visual question answering grand challenge</article-title>. <source>Grand Challenge</source>. <pub-id pub-id-type="doi">10.36227/techrxiv.13127537</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Mou</surname> <given-names>L.</given-names></name> <name><surname>Xing</surname> <given-names>E.</given-names></name> <name><surname>Xie</surname> <given-names>P.</given-names></name></person-group> (<year>2020b</year>). <article-title>Pathvqa: 30.000&#x0002B; questions for medical visual question answering</article-title>. <source>arXiv</source> [preprint] arXiv:2003.10286. <pub-id pub-id-type="doi">10.36227/techrxiv.13127537.v1</pub-id></citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>J.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>Z.</given-names></name> <name><surname>Gong</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>F. L.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Medical knowledge-based network for patient-oriented visual question answering</article-title>. <source>Inform. Proc. Managem</source>. <volume>60</volume>:<fpage>103241</fpage>. <pub-id pub-id-type="doi">10.1016/j.ipm.2022.103241</pub-id></citation>
</ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ilievski</surname> <given-names>I.</given-names></name> <name><surname>Yan</surname> <given-names>S.</given-names></name> <name><surname>Feng</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>A focused dynamic attention model for visual question answering</article-title>. <source>arXiv</source> [preprint] arXiv:1604.01485. <pub-id pub-id-type="doi">10.48550/arXiv.1604.01485</pub-id></citation>
</ref>
<ref id="B39">
<citation citation-type="book"><person-group person-group-type="author"><collab>International Organization for Standardization</collab></person-group> (<year>2008</year>). <article-title>&#x0201C;ISO 9241-171: Ergonomics of human-system interaction-guidance on software accessibility,&#x0201D;</article-title> in <source>Technical Report</source> (<publisher-loc>Geneva, Switzerland</publisher-loc>: <publisher-name>ISO Standard</publisher-name>).</citation>
</ref>
<ref id="B40">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Jagan Mohan</surname> <given-names>N.</given-names></name> <name><surname>Murugan</surname> <given-names>R.</given-names></name> <name><surname>Goel</surname> <given-names>T.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Deep learning for diabetic retinopathy detection: Challenges and opportunities,&#x0201D;</article-title> in <source>Next Generation Healthcare Informatics</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>213</fpage>&#x02013;<lpage>232</lpage>.</citation>
</ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>A.</given-names></name> <name><surname>Wang</surname> <given-names>F.</given-names></name> <name><surname>Porikli</surname> <given-names>F.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name></person-group> (<year>2015</year>). <article-title>Compositional memory for visual question answering</article-title>. <source>arXiv</source> [preprint] arXiv:1511.05676. <pub-id pub-id-type="doi">10.48550/arXiv.1511.05676</pub-id></citation>
</ref>
<ref id="B42">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>J.-H.</given-names></name> <name><surname>Lee</surname> <given-names>S.-W.</given-names></name> <name><surname>Kwak</surname> <given-names>D.</given-names></name> <name><surname>Heo</surname> <given-names>M.-O.</given-names></name> <name><surname>Kim</surname> <given-names>J.</given-names></name> <name><surname>Ha</surname> <given-names>J.-W.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>&#x0201C;Multimodal residual learning for visual QA,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source> (<publisher-loc>Red Hook, NY</publisher-loc>: <publisher-name>Curran Associates, Inc.</publisher-name>), <fpage>361</fpage>&#x02013;<lpage>369</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>J.-H.</given-names></name> <name><surname>On</surname> <given-names>K.-W.</given-names></name> <name><surname>Lim</surname> <given-names>W.</given-names></name> <name><surname>Kim</surname> <given-names>J.</given-names></name> <name><surname>Ha</surname> <given-names>J.-W.</given-names></name> <name><surname>Zhang</surname> <given-names>B.-T.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Hadamard product for low-rank bilinear pooling,&#x0201D;</article-title> in <source>Proceedings of the 5th International Conference on Learning Representations (ICLR</source> 2017).</citation>
</ref>
<ref id="B44">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kiros</surname> <given-names>R.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Salakhutdinov</surname> <given-names>R.</given-names></name> <name><surname>Zemel</surname> <given-names>R. S.</given-names></name> <name><surname>Torralba</surname> <given-names>A.</given-names></name> <name><surname>Urtasun</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>&#x0201C;Skip-thought vectors,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems, vol. 28</source> (<publisher-loc>Red Hook, NY</publisher-loc>: <publisher-name>Curran Associates, Inc.</publisher-name>), <fpage>3294</fpage>&#x02013;<lpage>3302</lpage>.</citation>
</ref>
<ref id="B45">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kovaleva</surname> <given-names>O.</given-names></name> <name><surname>Shivade</surname> <given-names>C.</given-names></name> <name><surname>Kashyap</surname> <given-names>S.</given-names></name> <name><surname>Kanjaria</surname> <given-names>K.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Ballah</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>&#x0201C;Towards visual dialog for radiology,&#x0201D;</article-title> in <source>Proceedings of the 19th SIGBioMed Workshop on Biomedical Language Processing</source> (<publisher-loc>Stroudsburg, PA</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>60</fpage>&#x02013;<lpage>69</lpage>.</citation>
</ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krizhevsky</surname> <given-names>A.</given-names></name> <name><surname>Sutskever</surname> <given-names>I.</given-names></name> <name><surname>Hinton</surname> <given-names>G. E.</given-names></name></person-group> (<year>2017</year>). <article-title>ImageNet classification with deep convolutional neural networks</article-title>. <source>Commun. ACM</source> <volume>60</volume>, <fpage>84</fpage>&#x02013;<lpage>90</lpage>. <pub-id pub-id-type="doi">10.1145/3065386</pub-id></citation>
</ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kumar</surname> <given-names>A.</given-names></name> <name><surname>Irsoy</surname> <given-names>O.</given-names></name> <name><surname>Ondruska</surname> <given-names>P.</given-names></name> <name><surname>Iyyer</surname> <given-names>M.</given-names></name> <name><surname>Bradbury</surname> <given-names>J.</given-names></name> <name><surname>Gulrajani</surname> <given-names>I.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>&#x0201C;Ask me anything: Dynamic memory networks for natural language processing,&#x0201D;</article-title> in <source>Proceedings of the International Conference on Machine Learning</source>, <fpage>1378</fpage>&#x02013;<lpage>1387</lpage>.</citation>
</ref>
<ref id="B48">
<citation citation-type="web"><person-group person-group-type="author"><collab>Learn Developers</collab></person-group> (<year>2024</year>). Sklearn.Metrics.Precision_Score-Scikit-Learn Documentation. Available online at: <ext-link ext-link-type="uri" xlink:href="https://scikit-learn.org/stable/modules/generated/sklearn.metrics.precision_score.html">https://scikit-learn.org/stable/modules/generated/sklearn.metrics.precision_score.html</ext-link> (Accessed April 04, 2025).</citation>
</ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>L.</given-names></name> <name><surname>Yatskar</surname> <given-names>M.</given-names></name> <name><surname>Yin</surname> <given-names>D.</given-names></name> <name><surname>Hsieh</surname> <given-names>C.</given-names></name> <name><surname>Chang</surname> <given-names>K.</given-names></name></person-group> (<year>2019a</year>). <article-title>Visualbert: A simple and performant baseline for vision and language</article-title>. <source>arXiv</source> [prerprint] arXiv:1908.03557. <pub-id pub-id-type="doi">10.48550/arXiv.1908.03557</pub-id></citation>
</ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>L. H.</given-names></name> <name><surname>Yatskar</surname> <given-names>M.</given-names></name> <name><surname>Yin</surname> <given-names>D.</given-names></name> <name><surname>Hsieh</surname> <given-names>C.-J.</given-names></name> <name><surname>Chang</surname> <given-names>K.-W.</given-names></name></person-group> (<year>2019b</year>). <article-title>VisualBERT: a simple and performant baseline for vision and language</article-title>. <source>arXiv</source> [preprint] arXiv:1908.03557. <pub-id pub-id-type="doi">10.48550/arXiv.1908.03557</pub-id></citation>
</ref>
<ref id="B51">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liao</surname> <given-names>Z.</given-names></name> <name><surname>Wu</surname> <given-names>Q.</given-names></name> <name><surname>Shen</surname> <given-names>C.</given-names></name> <name><surname>van den Hengel</surname> <given-names>A.</given-names></name> <name><surname>Verjans</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;AIML at VQA-Med 2020: Knowledge inference via a skeleton-based sentence mapping approach for medical domain visual question answering,&#x0201D;</article-title> in <source>Proceedings of the CLEF (Working Notes)</source> (<publisher-loc>Aachen</publisher-loc>: <publisher-name>CEUR-WS.org</publisher-name>).</citation>
</ref>
<ref id="B52">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lienhart</surname> <given-names>R.</given-names></name> <name><surname>Maydt</surname> <given-names>J</given-names></name></person-group>. (<year>2002</year>). <article-title>&#x0201C;An extended set of haar-like features for rapid object detection,&#x0201D;</article-title> in <source>Proceedings of the IEEE International Conference on Image Processing</source> (<publisher-loc>Rochester, NY</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>900</fpage>&#x02013;<lpage>903</lpage>.</citation>
</ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>D.</given-names></name> <name><surname>Tao</surname> <given-names>Q.</given-names></name> <name><surname>Shi</surname> <given-names>D.</given-names></name> <name><surname>Haffari</surname> <given-names>G.</given-names></name> <name><surname>Wu</surname> <given-names>Q.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Medical visual question answering: a survey</article-title>. <source>Artif. Intellig. Med</source>. <volume>143</volume>:<fpage>102611</fpage>. <pub-id pub-id-type="doi">10.1016/j.artmed.2023.102611</pub-id><pub-id pub-id-type="pmid">37673579</pub-id></citation></ref>
<ref id="B54">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>B.</given-names></name> <name><surname>Zhan</surname> <given-names>L.</given-names></name> <name><surname>Xu</surname> <given-names>L.</given-names></name> <name><surname>Ma</surname> <given-names>L.</given-names></name> <name><surname>Yang</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>X.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Slake: a semantically-labeled knowledge-enhanced dataset for medical visual question answering,&#x0201D;</article-title> in <source>Proceedings of the 2021 IEEE 18th International Symposium on Biomedical Imaging (ISBI)</source> (<publisher-loc>Nice</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1650</fpage>&#x02013;<lpage>1654</lpage>. <pub-id pub-id-type="doi">10.1109/ISBI48211.2021.9434010</pub-id></citation>
</ref>
<ref id="B55">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lowe</surname> <given-names>D. G.</given-names></name></person-group> (<year>1999</year>). <article-title>&#x0201C;Object recognition from local scale-invariant features,&#x0201D;</article-title> in <source>Proceedings of the Seventh IEEE International Conference on Computer Vision</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1150</fpage>&#x02013;<lpage>1157</lpage>.</citation>
</ref>
<ref id="B56">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lu</surname> <given-names>J.</given-names></name> <name><surname>Batra</surname> <given-names>D.</given-names></name> <name><surname>Parikh</surname> <given-names>D.</given-names></name> <name><surname>Lee</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems, vol. 32</source> (<publisher-loc>Red Hook, NY</publisher-loc>: <publisher-name>Curran Associates, Inc.</publisher-name>), <fpage>13</fpage>&#x02013;<lpage>23</lpage>.</citation>
</ref>
<ref id="B57">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lu</surname> <given-names>J.</given-names></name> <name><surname>Yang</surname> <given-names>J.</given-names></name> <name><surname>Batra</surname> <given-names>D.</given-names></name> <name><surname>Parikh</surname> <given-names>D.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Hierarchical question-image co-attention for visual question answering,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems, 29</source>, <fpage>289</fpage>&#x02013;<lpage>297</lpage>.</citation>
</ref>
<ref id="B58">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Majumder</surname> <given-names>M. A.</given-names></name> <name><surname>Guerrini</surname> <given-names>C. J.</given-names></name></person-group>. (<year>2016</year>). <article-title>Federal privacy protections: Ethical foundations, sources of confusion in clinical medicine, and controversies in biomedical research</article-title>. <source>AMA J. Ethics</source> <volume>18</volume>, <fpage>288</fpage>&#x02013;<lpage>298</lpage>. <pub-id pub-id-type="doi">10.1001/journalofethics.2016.18.3.pfor5-1603</pub-id><pub-id pub-id-type="pmid">27003001</pub-id></citation></ref>
<ref id="B59">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Malinowski</surname> <given-names>M.</given-names></name> <name><surname>Rohrbach</surname> <given-names>M.</given-names></name> <name><surname>Fritz</surname> <given-names>M.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Ask your neurons: a neural-based approach to answering questions about images,&#x0201D;</article-title> in <source>Proceedings of the IEEE International Conference on Computer Vision</source> (<publisher-loc>Santiago</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>9</lpage>.</citation>
</ref>
<ref id="B60">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Malinowski</surname> <given-names>M.</given-names></name> <name><surname>Rohrbach</surname> <given-names>M.</given-names></name> <name><surname>Fritz</surname> <given-names>M.</given-names></name></person-group> (<year>2017</year>). <article-title>Ask your neurons: a deep learning approach to visual question answering</article-title>. <source>Int. J. Comp. Vision</source> <volume>125</volume>, <fpage>110</fpage>&#x02013;<lpage>135</lpage>. <pub-id pub-id-type="doi">10.1007/s11263-017-1038-2</pub-id></citation>
</ref>
<ref id="B61">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Manmadhan</surname> <given-names>S.</given-names></name> <name><surname>Kovoor</surname> <given-names>B. C</given-names></name></person-group>. (<year>2020</year>). <article-title>Visual question answering: a state-of-the-art review</article-title>. <source>Artif. Intellig. Rev</source>. <volume>53</volume>, <fpage>5705</fpage>&#x02013;<lpage>5745</lpage>. <pub-id pub-id-type="doi">10.1007/s10462-020-09832-7</pub-id></citation>
</ref>
<ref id="B62">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mingrui</surname> <given-names>L.</given-names></name> <name><surname>Yanming</surname> <given-names>G.</given-names></name> <name><surname>Hui</surname> <given-names>W.</given-names></name> <name><surname>Xin</surname> <given-names>Z.</given-names></name></person-group> (<year>2018</year>). <article-title>Cross-modal multistep fusion network with co-attention for visual question answering</article-title>. <source>IEEE Access</source> <volume>6</volume>, <fpage>31516</fpage>&#x02013;<lpage>31524</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2018.2844789</pub-id></citation>
</ref>
<ref id="B63">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Narasimhan</surname> <given-names>M.</given-names></name> <name><surname>Schwing</surname> <given-names>A. G</given-names></name></person-group>. (<year>2018</year>). <article-title>&#x0201C;Straight to the facts: Learning knowledge base retrieval for factual visual question answering,&#x0201D;</article-title> in <source>Proceedings of the European Conference on Computer Vision (ECCV)</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>451</fpage>&#x02013;<lpage>468</lpage>.</citation>
</ref>
<ref id="B64">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Noh</surname> <given-names>H.</given-names></name> <name><surname>Han</surname> <given-names>B</given-names></name></person-group>. (<year>2016</year>). <article-title>Training recurrent answering units with joint loss minimization for vqa</article-title>. <source>arXiv</source> [preprint] arXiv:1606.03647. <pub-id pub-id-type="doi">10.48550/arXiv.1606.03647</pub-id></citation>
</ref>
<ref id="B65">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Noh</surname> <given-names>H.</given-names></name> <name><surname>Seo</surname> <given-names>P. H.</given-names></name> <name><surname>Han</surname> <given-names>B.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Image question answering using convolutional neural network with dynamic parameter prediction,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B66">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peng</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>F.</given-names></name> <name><surname>Rosen</surname> <given-names>M.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Umass at ImageCLEF medical visual question answering (med-vqa) 2018 task,&#x0201D;</article-title> in <source>Proceedings of the CEUR Workshop</source>.</citation>
</ref>
<ref id="B67">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Porwal</surname> <given-names>P.</given-names></name> <name><surname>Pachade</surname> <given-names>S.</given-names></name> <name><surname>Kamble</surname> <given-names>R.</given-names></name> <name><surname>Kokare</surname> <given-names>M.</given-names></name> <name><surname>Deshmukh</surname> <given-names>G.</given-names></name> <name><surname>Sahasrabuddhe</surname> <given-names>V.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Indian diabetic retinopathy image dataset (IDRID): A database for diabetic retinopathy screening research</article-title>. <source>Data</source> <volume>3</volume>:<fpage>25</fpage>. <pub-id pub-id-type="doi">10.3390/data3030025</pub-id></citation>
</ref>
<ref id="B68">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Powers</surname> <given-names>D. M. W.</given-names></name></person-group> (<year>2011</year>). <article-title>Evaluation: From precision, recall and f-measure to roc, informedness, markedness and correlation</article-title>. <source>J. Mach. Learn. Technol</source>. <volume>2</volume>, <fpage>37</fpage>&#x02013;<lpage>63</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2010.16061</pub-id></citation>
</ref>
<ref id="B69">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Radford</surname> <given-names>A.</given-names></name> <name><surname>Kim</surname> <given-names>J. W.</given-names></name> <name><surname>Hallacy</surname> <given-names>C.</given-names></name> <name><surname>Ramesh</surname> <given-names>A.</given-names></name> <name><surname>Goh</surname> <given-names>G.</given-names></name> <name><surname>Agarwal</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;Learning transferable visual models from natural language supervision,&#x0201D;</article-title> in <source>Proceedings of the International Conference on Machine Learning</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>8748</fpage>&#x02013;<lpage>8763</lpage>.</citation>
</ref>
<ref id="B70">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ren</surname> <given-names>M.</given-names></name> <name><surname>Kiros</surname> <given-names>R.</given-names></name> <name><surname>Zemel</surname> <given-names>R.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Image question answering: a visual semantic embedding model and a new dataset,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source> (<publisher-loc>Red Hook, NY</publisher-loc>: <publisher-name>Curran Associates, Inc.</publisher-name>), <fpage>5</fpage>.</citation>
</ref>
<ref id="B71">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Schilling</surname> <given-names>R.</given-names></name> <name><surname>Messina</surname> <given-names>P.</given-names></name> <name><surname>Parra</surname> <given-names>D.</given-names></name> <name><surname>Lobel</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Puc chile team at VQA-Med 2021: Approaching VQA as a classification task via fine-tuning a pretrained CNN,&#x0201D;</article-title> in <source>Working Notes of CLEF</source> (<publisher-loc>Aachen</publisher-loc>: <publisher-name>CEUR-WS Team</publisher-name>).</citation>
</ref>
<ref id="B72">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Shi</surname> <given-names>Y.</given-names></name> <name><surname>Furlanello</surname> <given-names>T.</given-names></name> <name><surname>Zha</surname> <given-names>S.</given-names></name> <name><surname>Anandkumar</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Question type guided attention in visual question answering,&#x0201D;</article-title> in <source>Proceedings of the ECCV 2018</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>151</fpage>&#x02013;<lpage>166</lpage>.</citation></ref>
<ref id="B73">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Simonyan</surname> <given-names>K.</given-names></name> <name><surname>Zisserman</surname> <given-names>A</given-names></name></person-group>. (<year>2015</year>). <article-title>Very deep convolutional networks for large-scale image recognition</article-title>. <source>arXiv</source> [preprint] arXiv:1409.1556v6. <pub-id pub-id-type="doi">10.48550/arXiv.1409.1556</pub-id></citation>
</ref>
<ref id="B74">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>J.</given-names></name> <name><surname>Holder</surname> <given-names>A.</given-names></name> <name><surname>Kamaleswaran</surname> <given-names>R.</given-names></name> <name><surname>Xie</surname> <given-names>Y.</given-names></name></person-group> (<year>2023</year>). <article-title>Detecting algorithmic bias in medical-ai models using trees</article-title>. <source>arXiv</source> [preprint] arXiv:2312.02959. <pub-id pub-id-type="doi">10.48550/arXiv.2312.02959</pub-id></citation>
</ref>
<ref id="B75">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Song</surname> <given-names>J.</given-names></name> <name><surname>Zeng</surname> <given-names>P.</given-names></name> <name><surname>Gao</surname> <given-names>L.</given-names></name> <name><surname>Shen</surname> <given-names>H.</given-names></name></person-group> (<year>2022</year>). <article-title>From pixels to objects: Cubic visual attention for visual question answering</article-title>. <source>arXiv</source> [preprint] arXiv:2206.01923. <pub-id pub-id-type="doi">10.48550/arXiv.2206.01923</pub-id></citation>
</ref>
<ref id="B76">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Szegedy</surname> <given-names>C.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Jia</surname> <given-names>Y.</given-names></name> <name><surname>Sermanet</surname> <given-names>P.</given-names></name> <name><surname>Reed</surname> <given-names>S.</given-names></name> <name><surname>Anguelov</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>&#x0201C;Going deeper with convolutions,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Boston, MA</publisher-loc>: <publisher-name>IEEE</publisher-name>)&#x0201E; <fpage>1</fpage>&#x02013;<lpage>9</lpage>.</citation>
</ref>
<ref id="B77">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Talafha</surname> <given-names>B.</given-names></name> <name><surname>Al-Ayyoub</surname> <given-names>M</given-names></name></person-group>. (<year>2018</year>). <article-title>&#x0201C;Just at VQA-Med: A VGG-Seq2Seq model,&#x0201D;</article-title> <source>Technical Report</source> (Aachen: CEUR-WS Team).</citation>
</ref>
<ref id="B78">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tascon-Morales</surname> <given-names>S.</given-names></name> <name><surname>M&#x000E1;rquez-Neila</surname> <given-names>P.</given-names></name> <name><surname>Sznitman</surname> <given-names>R.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Consistency-preserving visual question answering in medical imaging,&#x0201D;</article-title> in <source>Medical Image Computing and Computer Assisted Intervention - MICCAI 2022. MICCAI 2022. Lecture Notes in Computer Science, Vol. 13438</source>, eds. L. Wang, Q. Dou, P. T. Fletcher, S. Speidel, and S. Li (Cham: Springer). <pub-id pub-id-type="doi">10.1007/978-3-031-16452-1_37</pub-id></citation>
</ref>
<ref id="B79">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tascon-Morales</surname> <given-names>S.</given-names></name> <name><surname>M&#x000E1;rquez-Neila</surname> <given-names>P.</given-names></name> <name><surname>Sznitman</surname> <given-names>R.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Localized questions in medical visual question answering,&#x0201D;</article-title> in <source>Medical Image Computing and Computer Assisted Intervention - MICCAI 2023. MICCAI 2023. Lecture Notes in Computer Science, Vol. 14221</source>, eds H. Greenspan et al. (Cham: Springer). <pub-id pub-id-type="doi">10.1007/978-3-031-43895-0_34</pub-id></citation>
</ref>
<ref id="B80">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Verma</surname> <given-names>H.</given-names></name> <name><surname>Ramachandran</surname> <given-names>S</given-names></name></person-group>. (<year>2020a</year>). <article-title>&#x0201C;HARENDRAKV at VQA-Med 2020: Sequential vqa with attention for medical visual question answering,&#x0201D;</article-title> in <source>Proceedings of the CLEF (Working Notes)</source> (<publisher-loc>Aachen</publisher-loc>: <publisher-name>CEUR-WS.org</publisher-name>).</citation>
</ref>
<ref id="B81">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Verma</surname> <given-names>H. K.</given-names></name> <name><surname>Ramachandran</surname> <given-names>S</given-names></name></person-group>. (<year>2020b</year>). <article-title>&#x0201C;Harendrakv at vqa-med 2020: Sequential vqa with attention for medical visual question answering,&#x0201D;</article-title> in <source>Technical Report</source> (<publisher-loc>Aachen</publisher-loc>: <publisher-name>CEUR-WS Team</publisher-name>).</citation>
</ref>
<ref id="B82">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vu</surname> <given-names>M. H.</given-names></name> <name><surname>Lofstedt</surname> <given-names>T.</given-names></name> <name><surname>Nyholm</surname> <given-names>T.</given-names></name> <name><surname>Sznitman</surname> <given-names>R.</given-names></name></person-group> (<year>2020</year>). <article-title>A question-centric model for visual question answering in medical imaging</article-title>. <source>IEEE Trans. Medical Imag</source>. <volume>39</volume>, <fpage>2856</fpage>&#x02013;<lpage>2868</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2020.2978284</pub-id><pub-id pub-id-type="pmid">32149682</pub-id></citation></ref>
<ref id="B83">
<citation citation-type="web"><person-group person-group-type="author"><collab>W3C</collab></person-group> (<year>2023</year>). <source>Web Content Accessibility Guidelines (WCAG) 2.2</source>. World Wide Web Consortium (W3C) Recommendation. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.w3.org/TR/WCAG22/">https://www.w3.org/TR/WCAG22/</ext-link></citation>
</ref>
<ref id="B84">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Pan</surname> <given-names>H.</given-names></name> <name><surname>Zhang</surname> <given-names>K.</given-names></name> <name><surname>He</surname> <given-names>S.</given-names></name> <name><surname>Chen</surname> <given-names>C.</given-names></name></person-group> (<year>2022a</year>). <article-title>&#x0201C;M2fNet: multi-granularity feature fusion network for medical visual question answering,&#x0201D;</article-title> in <source>Proceedings of the PRICAI 2022: Trends in Artificial Intelligence, 19th Pacific Rim International Conference on Artificial Intelligence, PRICAI 2022</source> (<publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>141</fpage>&#x02013;<lpage>154</lpage>.</citation>
</ref>
<ref id="B85">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>M.</given-names></name> <name><surname>He</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>L.</given-names></name> <name><surname>Qing</surname> <given-names>L.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2022b</year>). <article-title>Medical visual question answering based on question-type reasoning and semantic space constraint</article-title>. <source>Artif. Intellig. Med</source>. <volume>131</volume>:<fpage>102346</fpage>. <pub-id pub-id-type="doi">10.1016/j.artmed.2022.102346</pub-id><pub-id pub-id-type="pmid">36100340</pub-id></citation></ref>
<ref id="B86">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>P.</given-names></name> <name><surname>Wu</surname> <given-names>Q.</given-names></name> <name><surname>Shen</surname> <given-names>C.</given-names></name> <name><surname>Dick</surname> <given-names>A.</given-names></name> <name><surname>Van Den Hengel</surname> <given-names>A.</given-names></name></person-group> (<year>2017</year>). <article-title>FVQA: Fact-based visual question answering</article-title>. <source>IEEE Trans. Pattern Analy. Mach. Intellig</source>. <volume>40</volume>, <fpage>2413</fpage>&#x02013;<lpage>2427</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2017.2754246</pub-id><pub-id pub-id-type="pmid">28945588</pub-id></citation></ref>
<ref id="B87">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>P.</given-names></name> <name><surname>Wu</surname> <given-names>Q.</given-names></name> <name><surname>Shen</surname> <given-names>C.</given-names></name> <name><surname>Hengel</surname> <given-names>A.</given-names></name> <name><surname>Dick</surname> <given-names>A.</given-names></name></person-group> (<year>2015</year>). <article-title>Explicit knowledge-based reasoning for visual question answering</article-title>. <source>arXiv</source> [preprint] arXiv:1511.02570. <pub-id pub-id-type="doi">10.48550/arXiv.1511.02570</pub-id></citation>
</ref>
<ref id="B88">
<citation citation-type="book"><person-group person-group-type="author"><collab>World Health Organization</collab></person-group> (<year>2019</year>). <source>World Report on Vision. Technical Report</source>. <publisher-loc>Geneva</publisher-loc>: <publisher-name>World Health Organization</publisher-name>.</citation>
</ref>
<ref id="B89">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>Q.</given-names></name> <name><surname>Wang</surname> <given-names>P.</given-names></name> <name><surname>Shen</surname> <given-names>C.</given-names></name> <name><surname>Dick</surname> <given-names>A.</given-names></name> <name><surname>Van Den Hengel</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Ask me anything: Free-form visual question answering based on knowledge from external sources,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>4622</fpage>&#x02013;<lpage>4630</lpage>.</citation>
</ref>
<ref id="B90">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiong</surname> <given-names>C.</given-names></name> <name><surname>Merity</surname> <given-names>S.</given-names></name> <name><surname>Socher</surname> <given-names>R.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Dynamic memory networks for visual and textual question answering,&#x0201D;</article-title> in <source>Proceedings of the International Conference on Machine Learning</source> [Cambridge MA: Proceedings of Machine Learning Research (PMLR)], <fpage>2397</fpage>&#x02013;<lpage>2406</lpage>.</citation>
</ref>
<ref id="B91">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>Z.</given-names></name> <name><surname>Dai</surname> <given-names>Z.</given-names></name> <name><surname>Yang</surname> <given-names>Y.</given-names></name> <name><surname>Carbonell</surname> <given-names>J.</given-names></name> <name><surname>Salakhutdinov</surname> <given-names>R.</given-names></name> <name><surname>Le</surname> <given-names>Q. V.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;XLNet: Generalized autoregressive pretraining for language understanding,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source> (<publisher-loc>Red Hook, NY</publisher-loc>: <publisher-name>Curran Associates, Inc.</publisher-name>), <fpage>5753</fpage>&#x02013;<lpage>5763</lpage>.</citation>
</ref>
<ref id="B92">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>D.</given-names></name> <name><surname>Cao</surname> <given-names>R.</given-names></name> <name><surname>Wu</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>Information fusion in visual question answering: a survey</article-title>. <source>Inform. Fusion</source> <volume>52</volume>, <fpage>268</fpage>&#x02013;<lpage>280</lpage>. <pub-id pub-id-type="doi">10.1016/j.inffus.2019.03.005</pub-id></citation>
</ref>
<ref id="B93">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Wu</surname> <given-names>C.</given-names></name> <name><surname>Zhao</surname> <given-names>Z.</given-names></name> <name><surname>Lin</surname> <given-names>W.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>PMC-VQA: Visual instruction tuning for medical visual question answering</article-title>. <source>arXiv</source> preprint arXiv:2305.10415. <pub-id pub-id-type="doi">10.48550/arXiv.2305.10415</pub-id></citation>
</ref>
<ref id="B94">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Y.</given-names></name> <name><surname>Jun</surname> <given-names>Y.</given-names></name> <name><surname>Chenchao</surname> <given-names>X.</given-names></name> <name><surname>Jianping</surname> <given-names>F.</given-names></name> <name><surname>Dacheng</surname> <given-names>T.</given-names></name></person-group> (<year>2018a</year>). <article-title>Beyond bilinear: Generalized multimodal factorized high-order pooling for visual question answering</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst</source>. <volume>29</volume>, <fpage>5947</fpage>&#x02013;<lpage>5959</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2018.2817340</pub-id><pub-id pub-id-type="pmid">29993847</pub-id></citation></ref>
<ref id="B95">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Y.</given-names></name> <name><surname>Kang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>F.</given-names></name></person-group> (<year>2018b</year>). <article-title>&#x0201C;Employing inception-resnet-v2 and bi-lstm for medical domain visual question answering,&#x0201D;</article-title> in <source>Technical Report</source> (<publisher-loc>Aachen</publisher-loc>: <publisher-name>CEUR-WS Team</publisher-name>).</citation>
</ref>
<ref id="B96">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Groth</surname> <given-names>O.</given-names></name> <name><surname>Bernstein</surname> <given-names>M.</given-names></name> <name><surname>Fei-Fei</surname> <given-names>L.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Visual7W: Grounded question answering in images,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>4995</fpage>&#x02013;<lpage>5004</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2016.540</pub-id></citation>
</ref>
</ref-list>
<app-group>
<app id="A1">
<title>Appendix</title>
<table-wrap position="float" id="T9">
<label>Table A1</label>
<caption><p>Model configuration summary.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Component</bold></th>
<th valign="top" align="left"><bold>Configuration details</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><bold>Model names</bold></td>
<td valign="top" align="left">fusion_mlp, hf_text, timm_image</td>
</tr>
<tr>
<td valign="top" align="left"><bold>hf_text</bold></td>
<td valign="top" align="left">Checkpoint: local://hf_text<break/> Pooling Mode: cls<break/> Max Text Length: 512<break/> Tokenizer: hf_auto<break/> Segment Num: 2<break/> Insert SEP: true<break/> Text Aug Detect Length: 10</td>
</tr>
<tr>
<td valign="top" align="left"><bold>timm_image</bold></td>
<td valign="top" align="left">Checkpoint: swin_base_patch4_window7_224<break/> Mix Choice: all_logits<break/> Transforms: resize_shorter_side, center_crop,<break/> trivial_augment<break/> Image Norm: imagenet<break/> Max Images per Column: 2</td>
</tr>
<tr>
<td valign="top" align="left"><bold>fusion_mlp</bold></td>
<td valign="top" align="left">Weight: 0.1<break/> Hidden Sizes: 128<break/> Activation: leaky_relu<break/> Drop Rate: 0.1<break/> Normalization: layer_norm</td>
</tr>
<tr>
<td valign="top" align="left"><bold>Optimization</bold></td>
<td valign="top" align="left">Optimizer: Adam W<break/> Learning Rate: 0.0001<break/> Weight Decay: 0.001<break/> LR Schedule: cosine_decay<break/> Max Epochs: 10<break/> Gradient Clipping: 1 (norm)<break/> Loss Function: auto<break/> Focal Loss &#x003B3;: 2.0</td>
</tr>
<tr>
<td valign="top" align="left"><bold>LoRA</bold></td>
<td valign="top" align="left">Modules: query, value, <sup><italic>q</italic></sup>, <sup><italic>v</italic></sup>, <sup><italic>k</italic></sup>, <sup><italic>o</italic></sup><break/> Rank (r): 8<break/> Alpha: 8</td>
</tr>
<tr>
<td valign="top" align="left"><bold>Environment</bold></td>
<td valign="top" align="left">GPUs: 1<break/> Batch Size: 16 (8 per GPU)<break/> Precision: 16-bit<break/> Workers: 2<break/> Strategy: auto_select_gpus</td>
</tr>
<tr>
<td valign="top" align="left"><bold>Backbone (electra)</bold></td>
<td valign="top" align="left">Architecture: ElectraForPreTraining<break/> Hidden Size: 768<break/> Layers: 12<break/> Heads: 12<break/> Dropout: 0.1<break/> Activation: gelu<break/> Vocab Size: 30522</td>
</tr>
</tbody>
</table>
</table-wrap>
</app>
</app-group>
</back>
</article>