<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Oncol.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Oncology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Oncol.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2234-943X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fonc.2025.1665690</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Predicting isocitrate dehydrogenase status in glioma using hierarchical attention-based deep 3D multiple instance learning</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Xie</surname><given-names>Qinqin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Sun</surname><given-names>Yongheng</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Liang</surname><given-names>Yuxia</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Shang</surname><given-names>Yu</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname><given-names>Haifeng</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname><given-names>Fan</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Wei</surname><given-names>Rong</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Chen</surname><given-names>Bin</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname><given-names>Ming</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/793326/overview"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Niu</surname><given-names>Chen</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>*</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3133398/overview"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Funding acquisition" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label><institution>PET-CT Center, The First Affiliated Hospital of Xi&#x2019;an Jiaotong University</institution>, <city>Xi&#x2019;an</city>,&#xa0;<country country="cn">China</country></aff>
<aff id="aff2"><label>2</label><institution>Department of Network Information, The First Affiliated Hospital of Xi&#x2019;an Jiaotong University</institution>, <city>Xi&#x2019;an</city>,&#xa0;<country country="cn">China</country></aff>
<aff id="aff3"><label>3</label><institution>School of Mathematics and Statistics, Xi&#x2019;an Jiaotong University</institution>, <city>Xi&#x2019;an</city>,&#xa0;<country country="cn">China</country></aff>
<aff id="aff4"><label>4</label><institution>Data Center, Hangzhou First People&#x2019;s Hospital</institution>, <city>Hangzhou</city>,&#xa0;<country country="cn">China</country></aff>
<author-notes>
<corresp id="c001"><label>*</label>Correspondence: Chen Niu, <email xlink:href="mailto:niuchen.xjtu@xjtu.edu.cn">niuchen.xjtu@xjtu.edu.cn</email></corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-12-18">
<day>18</day>
<month>12</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>15</volume>
<elocation-id>1665690</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>07</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>03</day>
<month>12</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>02</day>
<month>12</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Xie, Sun, Liang, Shang, Wang, Wang, Wei, Chen, Zhang and Niu.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Xie, Sun, Liang, Shang, Wang, Wang, Wei, Chen, Zhang and Niu</copyright-holder>
<license>
<ali:license_ref start_date="2025-12-18">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>According to the 2021 WHO classification of tumors of the central nervous system, isocitrate dehydrogenase (IDH) status serve an independent prognostic biomarker and is closely associated with tumor diagnosis and treatment response. At present, the determination of IDH status still relies on invasive surgical procedures.</p>
</sec>
<sec>
<title>Method</title>
<p>A total of 345 patients with pathologically confirmed gliomas diagnosed at the First Affiliated Hospital of Xi&#x2019;an Jiaotong University between October 2019 and October 2024 were retrospectively included, comprising 148 (42.9%) IDH-wild and 197 (57.1%) IDH-mutant. An additional 495 glioma patients were obtained from the public TCIA dataset. Patients were randomly split into training, validation, and test cohorts 6:2:2. A Hierarchical Attention-Based Multiple Instance Learning (HAB-MIL) framework was developed, integrating auxiliary positional encoding into feature maps to capture spatially specific information and generate refined 3D lesion representations. Model performance was evaluated using five-fold cross-validation, with receiver operating characteristic (ROC) curves, area under the curve (AUC), sensitivity, and specificity as assessment metrics.</p>
</sec>
<sec>
<title>Result</title>
<p>HAB-MIL achieved competitive performance, with AUCs of 0.917 and 0.892 on the glioma datasets from TCIA and the First Affiliated Hospital of Xi&#x2019;an Jiaotong University. Additionally, our work achieves results that are comparable to the state-of-the-art methods in TCIA dataset and demonstrates that multiple instance learning has great potential for IDH prediction.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>The proposed HAB-MIL achieved IDH classification based on conventional preoperative MRI images, eliminating the need for pixel-level annotations and significantly reducing the annotation burden for doctors.</p>
</sec>
</abstract>
<kwd-group>
<kwd>dynamic gated attention</kwd>
<kwd>glioma</kwd>
<kwd>IDH</kwd>
<kwd>location encoding</kwd>
<kwd>Multiple Instance Learning</kwd>
</kwd-group>
<funding-group>
<award-group id="gs1">
<funding-source id="sp1">
<institution-wrap>
<institution>Key Industry Innovation Chain of Shaanxi</institution>
<institution-id institution-id-type="doi" vocab="open-funder-registry" vocab-identifier="10.13039/open_funder_registry">10.13039/501100017591</institution-id>
</institution-wrap>
</funding-source>
<award-id rid="sp1">2024SF-ZDCYL-02-10</award-id>
</award-group>
<funding-statement>The author(s) declared that financial support was received for this work and/or its publication. This study was funded by the Shaanxi Provincial Key Industrial Innovation Chain Project (No. 2024SF-ZDCYL02-10),2024 Research Project on Clinical Applications of Medical Artificial Intelligence (No.YLXX24AIA021).</funding-statement>
</funding-group>
<counts>
<fig-count count="5"/>
<table-count count="7"/>
<equation-count count="16"/>
<ref-count count="44"/>
<page-count count="13"/>
<word-count count="7435"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Cancer Imaging and Image-directed Interventions</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Gliomas, the most prevalent malignant tumor in the central nervous system, account for approximately 80% of all intracranial tumors (<xref ref-type="bibr" rid="B1">1</xref>, <xref ref-type="bibr" rid="B2">2</xref>). According to the WHO 2021 classification of central nervous system tumors, isocitrate dehydrogenase (IDH) status is considered as a critical biomarker for the diagnosis and treatment of glioma and plays a vital role in prognosis and treatment strategies (<xref ref-type="bibr" rid="B3">3</xref>, <xref ref-type="bibr" rid="B4">4</xref>). Previous studies have shown that patients with IDH-wild generally have a poorer prognosis and exhibit less sensitivity to the therapeutic targeting of lDH mutations with vorasidenib compared with those with IDH-mutant (<xref ref-type="bibr" rid="B5">5</xref>&#x2013;<xref ref-type="bibr" rid="B7">7</xref>). At present, determination of IDH status primarily depends on surgical sampling followed by genetic sequencing. However, these approaches have several limitations: lesions that are deeply situated or located near eloquent brain regions may be difficult or impossible to sample (<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B9">9</xref>). Consequently, the accurate and noninvasive prediction of IDH status has become an urgent clinical need.</p>
<p>Recent advances in artificial intelligence, particularly in deep learning, have enabled noninvasive assessment of IDH status through MRI (<xref ref-type="bibr" rid="B10">10</xref>&#x2013;<xref ref-type="bibr" rid="B12">12</xref>). Deep learning models can automatically learn complex patterns directly from raw images, thereby minimizing the need for manual feature extraction and extensive domain expertise. In contrast to traditional tissue biopsies, it provides a safe, efficient, and repeatable way to preoperative evaluation and long-term follow-up (<xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B14">14</xref>). However, it largely depends on access to high-quality annotated datasets (<xref ref-type="bibr" rid="B15">15</xref>). Data labeling is labor-intensive and requires the expertise of highly trained specialists.</p>
<p>To overcome the challenges, weakly supervised learning approaches&#x2014;such as Multiple Instance Learning (MIL)&#x2014;have gained considerable attention in medical image analysis (<xref ref-type="bibr" rid="B16">16</xref>&#x2013;<xref ref-type="bibr" rid="B18">18</xref>). In the MIL framework, an image is regarded as a &#x201c;bag&#x201d; consisting of multiple &#x201c;instances,&#x201d; such as image patches or slices. For binary classification tasks, a bag is labeled as positive if at least one instance is positive. By relying on bag-level rather than instance-level labels, MIL is well suited for medical imaging lacking of fine-grained labeling (<xref ref-type="bibr" rid="B19">19</xref>).</p>
<p>Although MIL approaches, such as Attention-based Multiple Instance Learning (AB-MIL) and Clustering-constrained Attention Multiple Instance Learning (CLAM), have achieved remarkable progress in instance-level feature aggregation, they often overlook the spatial positional relationships among instances within the original image. In AB-MIL, the attention weights are derived from a fixed parameter matrix shared across all samples. This global static attention mechanism struggles to adapt to the diverse morphological variations of lesions and tends to overemphasize irrelevant background or noisy regions (<xref ref-type="bibr" rid="B20">20</xref>, <xref ref-type="bibr" rid="B21">21</xref>). Similarly, CLAM, despite introducing a clustering constraint to improve instance-level discriminability, still relies on a single static attention mechanism and therefore lacks the capability to dynamically adjust attention distributions in response to varying feature patterns across different samples (<xref ref-type="bibr" rid="B22">22</xref>, <xref ref-type="bibr" rid="B23">23</xref>).</p>
<p>To address these limitations, we propose a MIL network named Hierarchical Attention-Based Multiple Instance Learning (HAB-MIL). The model includes a Dynamic Gated Attention (DGA) module that integrates a learnable singular value decomposition (LSVD) framework. This framework adaptively decomposes and reweights the instance feature matrix, allowing dynamic modulation of the key components within the attention map. It enables the network to capture latent inter-instance dependencies and adjust the attention distribution in a data-adaptive manner.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Method</title>
<sec id="s2_1">
<label>2.1</label>
<title>Patient</title>
<p>This study included patients who were preoperatively diagnosed with glioma and underwent surgical treatment at the Department of Neurosurgery, the First Affiliated Hospital of Xi&#x2019;an Jiaotong University, between October 2019 and October 2024. A total of 506 patients with histopathologically confirmed glioma and available genetic testing were initially screened according to the 2021 WHO classification of tumors of the central nervous system, 5th edition. After applying the inclusion and exclusion criteria, 345 patients who met all requirements were ultimately enrolled in the study cohort, as shown in <xref ref-type="fig" rid="f1"><bold>Figure&#xa0;1</bold></xref>. The IDH status of all data was determined by DNA sequencing.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Flowchart of patient inclusion and exclusion.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1665690-g001.tif">
<alt-text content-type="machine-generated">Initially, 505 patients were identified, with 39 excluded due to other conditions. Of the remaining 466 patients, 81 were excluded for incomplete molecular testing, resulting in 385 patients with a confirmed IDH genotype. Finally, data from 344 patients met the inclusion criteria, after excluding 41 due to poor cooperation or motion artifacts.</alt-text>
</graphic></fig>
<p>Inclusion Criteria:</p>
<list list-type="order">
<list-item>
<p>Patients aged 18&#x2013;80 years with a single intracranial lesion confirmed as glioma by postoperative pathology, with complete molecular genetic testing results.</p></list-item>
<list-item>
<p>Mini-Mental State Examination (MMSE) score between 28 and 30, indicating no significant cognitive impairment and sufficient ability to cooperate during MRI scanning.</p></list-item>
<list-item>
<p>No MRI contraindications, such as metallic implants, cardiac pacemakers, or severe claustrophobia.</p></list-item>
<list-item>
<p>No prior history of craniotomy or radiotherapy.</p></list-item>
</list>
<p>Exclusion Criteria:</p>
<list list-type="order">
<list-item>
<p>Poor imaging quality or severe artifacts resulting in data unsuitable for analysis.</p></list-item>
<list-item>
<p>Presence of infectious, structural, or metabolic brain diseases that could interfere with image interpretation.</p></list-item>
<list-item>
<p>Incomplete or indeterminate pathological and molecular genetic testing results.</p></list-item>
</list>
<p>All patients underwent preoperative cranial MRI examinations and provided written informed consent prior to imaging. The study was approved by the Ethics Committee of the First Affiliated Hospital of Xi&#x2019;an Jiaotong University and was registered at <ext-link ext-link-type="uri" xlink:href="http://www.ClinicalTrials.gov">ClinicalTrials.gov</ext-link> (registration number: NCT05019196).</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Method</title>
<p>Given a complete histopathological image slice <inline-formula>
<mml:math display="inline" id="im1"><mml:mi>X</mml:mi></mml:math></inline-formula>, our goal is to predict the image label <inline-formula>
<mml:math display="inline" id="im2"><mml:mi>Y</mml:mi></mml:math></inline-formula> by analyzing the features extracted from discrimination patches <inline-formula>
<mml:math display="inline" id="im3"><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. To achieve this, we designed a two-stage framework for the classification of IDH status, as illustrated in <xref ref-type="fig" rid="f2"><bold>Figure&#xa0;2</bold></xref>. In the first stage, we encode the spatial features of the tumor. In the second stage, we introduce a novel HAB-MIL network, which comprises multiple layers of convolutional neural networks and an attention mechanism module. This module aggregates the features of the candidate instance in the <inline-formula>
<mml:math display="inline" id="im4"><mml:mi>Y</mml:mi></mml:math></inline-formula>image-level prediction by recalibrating the importance coefficients for each instance.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Overview of the hierarchical attention-based multiple instance learning for predicting isocitrate dehydrogenase status in glioma.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1665690-g002.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a machine learning pipeline involving collateral location encoding and dynamic gated attention. Brain MRI slices are transformed into features using convolutional and pooling layers. These features undergo location encoding with normalization and softmax operations. Instance features are processed with attention and transformation functions, distinguishing between IDH-Mutant and IDH-Wild types. Tanh and Sigmoid functions are applied, followed by Linear Singular Value Decomposition (LSVD), forming part of the dynamic gated attention mechanism. Element-wise and matrix multiplications are depicted, with corresponding mathematical equations.</alt-text>
</graphic></fig>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Deep instance generation</title>
<p>In MIL, the data is organized into &#x201c;bags&#x201d;, each containing multiple &#x201c;instances&#x201d;. The label assigned to a bag depends on the instances it contains. For a binary classification task, the bag is labeled positive if at least one instance within it is positive; otherwise, it is labeled negative. In the weakly supervised histopathology image classification problem, the dataset <inline-formula>
<mml:math display="inline" id="im5"><mml:mrow><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mi>N</mml:mi></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> consists of <inline-formula>
<mml:math display="inline" id="im6"><mml:mi>N</mml:mi></mml:math></inline-formula>MRI images, where each slide <inline-formula>
<mml:math display="inline" id="im7"><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>has label <inline-formula>
<mml:math display="inline" id="im8"><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, and the objective is to train a model to predict the labels of MRIs. This relationship is formally expressed as (<xref ref-type="disp-formula" rid="eq1">Equation 1</xref>):</p>
<disp-formula id="eq1"><label>(1)</label>
<mml:math display="block" id="M1"><mml:mrow><mml:mi>Y</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mstyle displaystyle="true"><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>M</mml:mi></mml:munderover><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:mstyle></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im9"><mml:mi>Y</mml:mi></mml:math></inline-formula> is the bag label, <inline-formula>
<mml:math display="inline" id="im10"><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is the label of instance, and is the number of instances in each bag. In our work, this assumption indicates that MRI is from a glioma patient if it involves at least one lesion. Based on the assumption, the empirical loss is formulated by (<xref ref-type="disp-formula" rid="eq2">Equation 2</xref>):</p>
<disp-formula id="eq2"><label>(2)</label>
<mml:math display="block" id="M2"><mml:mrow><mml:msub><mml:mover accent="true"><mml:mi>&#xca;</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mi>Q</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>f</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>m</mml:mi></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>m</mml:mi></mml:munderover><mml:mrow><mml:mi>l</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>f</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:mrow></mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im11"><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>f</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> represents a labeling function induced an MIL scoring function <inline-formula>
<mml:math display="inline" id="im12"><mml:mi>f</mml:mi></mml:math></inline-formula>, and <inline-formula>
<mml:math display="inline" id="im13"><mml:mrow><mml:mi>l</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mo>&#xb7;</mml:mo><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>can be any loss function. The MIL process can be mathematically represented as follows (<xref ref-type="disp-formula" rid="eq3">Equation 3</xref>):</p>
<disp-formula id="eq3"><label>(3)</label>
<mml:math display="block" id="M3"><mml:mrow><mml:mi>Y</mml:mi><mml:mo>=</mml:mo><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>h</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>N</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im14"><mml:mi>f</mml:mi></mml:math></inline-formula> is the feature extraction function, <inline-formula>
<mml:math display="inline" id="im15"><mml:mi>h</mml:mi></mml:math></inline-formula>is the instance-level aggregation operator, <inline-formula>
<mml:math display="inline" id="im16"><mml:mi>g</mml:mi></mml:math></inline-formula> is predicts the bag-level label. The model processes three MRI modalities: T1w, T2w and Flair, which represented as a 3D tensor <inline-formula>
<mml:math display="inline" id="im17"><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>C</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>W</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>D</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, where <inline-formula>
<mml:math display="inline" id="im18"><mml:mi>B</mml:mi></mml:math></inline-formula>is the batch size, <inline-formula>
<mml:math display="inline" id="im19"><mml:mi>C</mml:mi></mml:math></inline-formula>is the number of channels and <inline-formula>
<mml:math display="inline" id="im20"><mml:mrow><mml:mi>H</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>W</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>D</mml:mi></mml:mrow></mml:math></inline-formula>is the spatial resolution. The final layer of 3D fully CNN outputs a series of 3D feature maps <inline-formula>
<mml:math display="inline" id="im21"><mml:mrow><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>f</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> with the shape of <inline-formula>
<mml:math display="inline" id="im22"><mml:mrow><mml:msup><mml:mi>H</mml:mi><mml:mo>*</mml:mo></mml:msup><mml:mo>&#xd7;</mml:mo><mml:msup><mml:mi>W</mml:mi><mml:mo>*</mml:mo></mml:msup><mml:mo>&#xd7;</mml:mo><mml:msup><mml:mi>S</mml:mi><mml:mo>*</mml:mo></mml:msup><mml:mo>&#xd7;</mml:mo><mml:mi>D</mml:mi></mml:mrow></mml:math></inline-formula>, where <inline-formula>
<mml:math display="inline" id="im23"><mml:mrow><mml:msup><mml:mi>H</mml:mi><mml:mo>&#x2217;</mml:mo></mml:msup></mml:mrow></mml:math></inline-formula>, <inline-formula>
<mml:math display="inline" id="im24"><mml:mrow><mml:msup><mml:mi>W</mml:mi><mml:mo>&#x2217;</mml:mo></mml:msup></mml:mrow></mml:math></inline-formula>, <inline-formula>
<mml:math display="inline" id="im25"><mml:mrow><mml:msup><mml:mi>S</mml:mi><mml:mo>*</mml:mo></mml:msup></mml:mrow></mml:math></inline-formula> and <inline-formula>
<mml:math display="inline" id="im26"><mml:mi>D</mml:mi></mml:math></inline-formula>represent the high, width, spatial, and feature dimension of 3D feature maps, respectively. The feature extraction process consists of multiple 3D convolutional layers, batch normalization, ReLU activation, max pooling, and dropout layers. Finally, feature map is formulated as (<xref ref-type="disp-formula" rid="eq4">Equation 4</xref>):</p>
<disp-formula id="eq4"><label>(4)</label>
<mml:math display="block" id="M4"><mml:mrow><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>f</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>D</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mi>P</mml:mi><mml:mi>o</mml:mi><mml:mi>o</mml:mi><mml:mi>l</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x3c3;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>B</mml:mi><mml:mi>N</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo>&#x2217;</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mn>0</mml:mn></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im27"><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mn>0</mml:mn></mml:msub></mml:mrow></mml:math></inline-formula> is a 3D convolution kernels, with size <inline-formula>
<mml:math display="inline" id="im28"><mml:mrow><mml:mn>3</mml:mn><mml:mo>&#xd7;</mml:mo><mml:mn>3</mml:mn><mml:mo>&#xd7;</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:math></inline-formula>, and the output channels are 32. <inline-formula>
<mml:math display="inline" id="im29"><mml:mo>&#x2217;</mml:mo></mml:math></inline-formula>indicates the 3D convolution operation. Batch Normalization (BN) is used to stabilize training. <inline-formula>
<mml:math display="inline" id="im30"><mml:mi>&#x3c3;</mml:mi></mml:math></inline-formula>is the ReLU activation function.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Collateral location encoding</title>
<p>The spatial location of a tumor often provides critical diagnostic cues (<xref ref-type="bibr" rid="B24">24</xref>, <xref ref-type="bibr" rid="B25">25</xref>), since we proposed Collateral location encoding (CLE), as illustrated in <xref ref-type="fig" rid="f2"><bold>Figure&#xa0;2</bold></xref>. <inline-formula>
<mml:math display="inline" id="im31"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo>,</mml:mo><mml:mi>z</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>represents the voxel coordinates of the image. Since <inline-formula>
<mml:math display="inline" id="im32"><mml:mrow><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo>,</mml:mo><mml:mi>z</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>typically takes continuous values, employing a smooth nonlinear transformation like GELU allows for a more natural modeling of the relationships between coordinates, unlike ReLU, which introduces abrupt discontinuities. The location feature map <inline-formula>
<mml:math display="inline" id="im33"><mml:mrow><mml:mi>F</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>or each position of pixel <inline-formula>
<mml:math display="inline" id="im34"><mml:mrow><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo>,</mml:mo><mml:mi>z</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> is calculated as (<xref ref-type="disp-formula" rid="eq5">Equation 5</xref>):</p>
<disp-formula id="eq5"><label>(5)</label>
<mml:math display="block" id="M5"><mml:mrow><mml:mover accent="true"><mml:mi>F</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mrow><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>&#xb7;</mml:mo><mml:mi>G</mml:mi><mml:mi>E</mml:mi><mml:mi>L</mml:mi><mml:mi>U</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>&#xb7;</mml:mo><mml:mi>p</mml:mi><mml:mo>+</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow></mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im35"><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow></mml:math></inline-formula> are the learnable weights and biases of our neural positional encoding layers. The learned features <inline-formula>
<mml:math display="inline" id="im36"><mml:mi>F</mml:mi></mml:math></inline-formula> and location encoding <inline-formula>
<mml:math display="inline" id="im37"><mml:mover accent="true"><mml:mi>F</mml:mi><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula>from the previous feature extraction are first normalized to benefit the training and backpropagation process. Then features concatenated to be the input of this phase, which can be denoted as (<xref ref-type="disp-formula" rid="eq6">Equation 6</xref>):</p>
<disp-formula id="eq6"><label>(6)</label>
<mml:math display="block" id="M6"><mml:mrow><mml:msup><mml:mi>F</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>N</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi><mml:mo>(</mml:mo><mml:mi>F</mml:mi><mml:mo>)</mml:mo><mml:mrow><mml:mo>&#x2016;</mml:mo><mml:mrow><mml:mi>N</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi><mml:mo>(</mml:mo><mml:mover accent="true"><mml:mi>F</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im38"><mml:mrow><mml:mrow><mml:mo>&#x2016;</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>means the concatenation operation, <inline-formula>
<mml:math display="inline" id="im39"><mml:mrow><mml:mi>N</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi><mml:mo>(</mml:mo><mml:mi>F</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>, <inline-formula>
<mml:math display="inline" id="im40"><mml:mrow><mml:mi>N</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi><mml:mo>(</mml:mo><mml:mover accent="true"><mml:mi>F</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> refer to the normalization of <inline-formula>
<mml:math display="inline" id="im41"><mml:mi>F</mml:mi></mml:math></inline-formula>, <inline-formula>
<mml:math display="inline" id="im42"><mml:mover accent="true"><mml:mi>F</mml:mi><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula>, <inline-formula>
<mml:math display="inline" id="im43"><mml:mrow><mml:msup><mml:mi>H</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msup><mml:mi>H</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mo>&#x2016;</mml:mo><mml:mrow><mml:msup><mml:mover accent="true"><mml:mi>H</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msubsup><mml:mi>H</mml:mi><mml:mn>1</mml:mn><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo>&#x2016;</mml:mo><mml:mrow><mml:msubsup><mml:mover accent="true"><mml:mi>H</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mn>1</mml:mn><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msubsup><mml:mi>H</mml:mi><mml:mi>k</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo>&#x2016;</mml:mo><mml:mrow><mml:msubsup><mml:mover accent="true"><mml:mi>H</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mi>k</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula><inline-formula>
<mml:math display="inline" id="im44"><mml:mrow><mml:msubsup><mml:mi>H</mml:mi><mml:mi>k</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mover accent="true"><mml:mi>H</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mi>k</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mi>n</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula>. Then, using the self-attention mechanism, the importance of the features can be calculated by (<xref ref-type="disp-formula" rid="eq7">Equations 7</xref>, <xref ref-type="disp-formula" rid="eq8">8</xref>):</p>
<disp-formula id="eq7"><label>(7)</label>
<mml:math display="block" id="M7"><mml:mrow><mml:mi>A</mml:mi><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:msup><mml:mi>H</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>+</mml:mo><mml:mi>b</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math>
</disp-formula>
<disp-formula id="eq8"><label>(8)</label>
<mml:math display="block" id="M8"><mml:mrow><mml:mi>O</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mi>H</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>&#x2297;</mml:mo><mml:mi>A</mml:mi></mml:mrow></mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im45"><mml:mrow><mml:mi>W</mml:mi><mml:mo>,</mml:mo><mml:mi>b</mml:mi></mml:mrow></mml:math></inline-formula> denote the learnable weight and bias parameters. <inline-formula>
<mml:math display="inline" id="im46"><mml:msup><mml:mi>H</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup></mml:math></inline-formula>and <inline-formula>
<mml:math display="inline" id="im47"><mml:mi>O</mml:mi></mml:math></inline-formula> represent the input and output of the adaptive feature importance weighting mechanism based on the attention module. <inline-formula>
<mml:math display="inline" id="im48"><mml:mrow><mml:mi>S</mml:mi><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:math></inline-formula> operation is applied to normalize the attention scores and resulting in an attention weight matrix <inline-formula>
<mml:math display="inline" id="im49"><mml:mi>A</mml:mi></mml:math></inline-formula> that maps each weight to the interval <inline-formula>
<mml:math display="inline" id="im50"><mml:mrow><mml:mo>[</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula>and guarantees the sum of these mapped attention weights to be 1. <inline-formula>
<mml:math display="inline" id="im51"><mml:mo>&#x2297;</mml:mo></mml:math></inline-formula>denotes element-wise multiplication. Through this process, the input features can be weighted and assigned different importance, enabling the model to emphasize critical features and thereby enhance overall performance.</p>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Dynamic gated attention</title>
<p>Attention mechanisms are well known for dynamically highlighting important parts of input data by assigning different weights, allowing the model to focus on the most critical information for the task at hand (<xref ref-type="bibr" rid="B26">26</xref>&#x2013;<xref ref-type="bibr" rid="B28">28</xref>). To enhance the accuracy of learning attention weights, we propose Dynamic Gated Attention (DGA), which achieves superior performance. As illustrated in <xref ref-type="fig" rid="f2"><bold>Figure&#xa0;2</bold></xref>, the DGA module integrates two different processing methods.</p>
<p>The first component is an attention representation that uses Tanh as the activation function. With an output range of <inline-formula>
<mml:math display="inline" id="im52"><mml:mrow><mml:mo>[</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula>, Tanh provides both positive and negative activation signals, allowing the model to capture complex relationship features. In addition, a dropout function is incorporated to mitigate overfitting. The attention weights are defined as (<xref ref-type="disp-formula" rid="eq9">Equation 9</xref>):</p>
<disp-formula id="eq9"><label>(9)</label>
<mml:math display="block" id="M9"><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mi>x</mml:mi></mml:msup><mml:mo>&#x2212;</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>x</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mi>x</mml:mi></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>x</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mrow></mml:math>
</disp-formula>
<p>The second component utilizes a different attention representation. As illustrated in <xref ref-type="fig" rid="f3"><bold>Figure&#xa0;3</bold></xref>, we propose using Learnable Singular Value Decomposition (LSVD) to reduce the dimensionality of the feature map (<xref ref-type="bibr" rid="B29">29</xref>), followed by a Sigmoid activation function, which creates a &#x201c;switch-like&#x201d; effect, indicating whether the model should attend to or ignore a particular feature. The algorithm flowchart is shown in <xref ref-type="boxed-text" rid="algo1"><bold>Algorithm 1</bold></xref>.</p>
<boxed-text id="algo1" position="float">
<label>Algorithm 1</label>
<caption>
<title>Learnable Singular Value Decomposition.</title></caption>
<p>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1665690-g006.tif">
<alt-text content-type="machine-generated"></alt-text>
</graphic></p>
</boxed-text>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Illustration of the learnable singular value decomposition.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1665690-g003.tif">
<alt-text content-type="machine-generated">Diagram illustrating a neural network processing a feature matrix, resulting in a learnable softmax function with matrices and computations. It involves transformations represented visually by cubes and arrows showing the flow from input to output components, labeled as \(f_U(x)\), \(f_V(x)\), \(R(A)\), \(\alpha_U\), \(A'\), and \(\alpha_V^T\).</alt-text>
</graphic></fig>
<p>Fact1 The feature map <inline-formula>
<mml:math display="inline" id="im53"><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula> can be decomposed as <inline-formula>
<mml:math display="inline" id="im54"><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>=</mml:mo><mml:msub><mml:mi>&#x3b1;</mml:mi><mml:mi>U</mml:mi></mml:msub><mml:msup><mml:mi>&#x3a3;</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:msub><mml:mi>&#x3b1;</mml:mi><mml:mi>V</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, where <inline-formula>
<mml:math display="inline" id="im55"><mml:mrow><mml:msub><mml:mi>&#x3b1;</mml:mi><mml:mi>U</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>&#x211c;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> is an orthogonal matrix with the left singular vector of <inline-formula>
<mml:math display="inline" id="im56"><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula> as the column vector, where <inline-formula>
<mml:math display="inline" id="im57"><mml:mrow><mml:msubsup><mml:mi>&#x3b1;</mml:mi><mml:mi>V</mml:mi><mml:mi>T</mml:mi></mml:msubsup><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>&#x211c;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>A</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> is an orthogonal matrix formed by the right singular vectors of <inline-formula>
<mml:math display="inline" id="im58"><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula>. The matrix <inline-formula>
<mml:math display="inline" id="im59"><mml:mrow><mml:msup><mml:mo>&#x2211;</mml:mo><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mi>&#x211c;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#xd7;</mml:mo><mml:mi>&#x211c;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> is a diagonal matrix consisting of the singular values of <inline-formula>
<mml:math display="inline" id="im60"><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula>.</p>
<p>Proof of the Fact 1: According to the Singular Value Decomposition (SVD), any matrix <inline-formula>
<mml:math display="inline" id="im61"><mml:mrow><mml:mi>M</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> admits a factorization of the form <inline-formula>
<mml:math display="inline" id="im62"><mml:mrow><mml:mi>M</mml:mi><mml:mo>=</mml:mo><mml:mi>U</mml:mi><mml:mo>&#x2211;</mml:mo><mml:msup><mml:mi>V</mml:mi><mml:mi>T</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula>, where <inline-formula>
<mml:math display="inline" id="im63"><mml:mrow><mml:msub><mml:mi>&#x3c3;</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>&#x2265;</mml:mo><mml:msub><mml:mi>&#x3c3;</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>&#x2265;</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>&#x2265;</mml:mo><mml:msub><mml:mi>&#x3c3;</mml:mi><mml:mi>r</mml:mi></mml:msub><mml:mo>&gt;</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula> are the singular values and <inline-formula>
<mml:math display="inline" id="im64"><mml:mrow><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mi>&#x211c;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>M</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> is the rank of <inline-formula>
<mml:math display="inline" id="im65"><mml:mi>M</mml:mi></mml:math></inline-formula>. Hence, for <inline-formula>
<mml:math display="inline" id="im66"><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, the decomposition is as shown in (<xref ref-type="disp-formula" rid="eq10">Equation 10</xref>):</p>
<disp-formula id="eq10"><label>(10)</label>
<mml:math display="block" id="M10"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>&#x3bc;</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mo>&#x22ef;</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>&#x3bc;</mml:mi><mml:mi>m</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>&#x3c3;</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mo>&#x22ef;</mml:mo></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>&#x22ee;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x22f1;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x22ee;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x22ee;</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mo>&#x22ef;</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>&#x3c3;</mml:mi><mml:mi>r</mml:mi></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mo>&#x22ef;</mml:mo></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>v</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mo>&#x22ef;</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>v</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mi>T</mml:mi></mml:msubsup></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>&#x2248;</mml:mo><mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>&#x3bc;</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mo>&#x22ef;</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>&#x3bc;</mml:mi><mml:mi>r</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>&#x3c3;</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mo>&#x22ef;</mml:mo></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>&#x22ee;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x22f1;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x22ee;</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mo>&#x22ef;</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>&#x3c3;</mml:mi><mml:mi>r</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>v</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mo>&#x22ef;</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>v</mml:mi><mml:mi>r</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>r</mml:mi></mml:mrow><mml:mi>T</mml:mi></mml:msubsup></mml:mtd></mml:mtr></mml:mtable></mml:math>
</disp-formula>
<p>Here, <inline-formula>
<mml:math display="inline" id="im67"><mml:mrow><mml:msub><mml:mi>&#x3bc;</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>&#x3bd;</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> are the singular vectors, <inline-formula>
<mml:math display="inline" id="im68"><mml:mrow><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mi>&#x211c;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>is the number of nonzero singular values. For clarity, we denote the left singular matrix <inline-formula>
<mml:math display="inline" id="im69"><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>&#x3bc;</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:mo>&#x2026;</mml:mo></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>&#x3bc;</mml:mi><mml:mi>r</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> as <inline-formula>
<mml:math display="inline" id="im70"><mml:mrow><mml:msub><mml:mi>&#x3b1;</mml:mi><mml:mi>U</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, the diagonal matrix <inline-formula>
<mml:math display="inline" id="im71"><mml:mrow><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>a</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>&#x3c3;</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>&#x3c3;</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>&#x3c3;</mml:mi><mml:mi>r</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> as <inline-formula>
<mml:math display="inline" id="im72"><mml:msup><mml:mi>&#x3a3;</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup></mml:math></inline-formula>, and the right singular matrix <inline-formula>
<mml:math display="inline" id="im73"><mml:mrow><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>&#x3bd;</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:mo>&#x2026;</mml:mo></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>&#x3bd;</mml:mi><mml:mi>r</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow><mml:mi>T</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula> as <inline-formula>
<mml:math display="inline" id="im74"><mml:mrow><mml:msubsup><mml:mi>&#x3b1;</mml:mi><mml:mi>V</mml:mi><mml:mi>T</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula>, yielding: <inline-formula>
<mml:math display="inline" id="im75"><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>=</mml:mo><mml:msub><mml:mi>&#x3b1;</mml:mi><mml:mi>U</mml:mi></mml:msub><mml:msup><mml:mi>&#x3a3;</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:msubsup><mml:mi>&#x3b1;</mml:mi><mml:mi>V</mml:mi><mml:mi>T</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula>.</p>
<p>The remaining challenge is to determine <inline-formula>
<mml:math display="inline" id="im76"><mml:mrow><mml:msub><mml:mi>&#x3b1;</mml:mi><mml:mi>U</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msup><mml:mi>&#x3a3;</mml:mi><mml:mo>'</mml:mo></mml:msup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>&#x3b1;</mml:mi><mml:mi>V</mml:mi><mml:mi>T</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula>. A natural approach is to leverage neural networks to learn these latent representations. Specifically, assuming <inline-formula>
<mml:math display="inline" id="im77"><mml:msup><mml:mo>&#x2211;</mml:mo><mml:mo>&#x2032;</mml:mo></mml:msup></mml:math></inline-formula>is a trainable diagonal matrix defined as (<xref ref-type="disp-formula" rid="eq11">Equation 11</xref>):</p>
<disp-formula id="eq11"><label>(11)</label>
<mml:math display="block" id="M11"><mml:mrow><mml:msup><mml:mo>&#x2211;</mml:mo><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>=</mml:mo><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>a</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mi>r</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im78"><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mi>R</mml:mi></mml:mrow></mml:math></inline-formula> are trainable parameters in the neural network, which are initialized prior to training and subsequently optimized via backpropagation.</p>
<p>Then, we utilize two learnable layers of <inline-formula>
<mml:math display="inline" id="im79"><mml:mi>X</mml:mi></mml:math></inline-formula> to fit <inline-formula>
<mml:math display="inline" id="im80"><mml:mrow><mml:msub><mml:mi>&#x3b1;</mml:mi><mml:mi>U</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msubsup><mml:mi>&#x3b1;</mml:mi><mml:mi>V</mml:mi><mml:mi>T</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> as shown in <xref ref-type="disp-formula" rid="eq12">Equation 12</xref>, <xref ref-type="disp-formula" rid="eq13">13</xref>):</p>
<disp-formula id="eq12"><label>(12)</label>
<mml:math display="block" id="M12"><mml:mrow><mml:msub><mml:mi>&#x3b1;</mml:mi><mml:mi>U</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>U</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>X</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math>
</disp-formula>
<disp-formula id="eq13"><label>(13)</label>
<mml:math display="block" id="M13"><mml:mrow><mml:msubsup><mml:mi>&#x3b1;</mml:mi><mml:mi>V</mml:mi><mml:mi>T</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mo>(</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>V</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>X</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>Among these, <inline-formula>
<mml:math display="inline" id="im81"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>U</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mo>&#xb7;</mml:mo><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>V</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mo>&#xb7;</mml:mo><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> are projection layers to transform the <inline-formula>
<mml:math display="inline" id="im82"><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:math></inline-formula> matrix <inline-formula>
<mml:math display="inline" id="im83"><mml:mi>X</mml:mi></mml:math></inline-formula>into matrices of shape <inline-formula>
<mml:math display="inline" id="im84"><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:math></inline-formula>. In our implementation, we share the same set of two linear layers for <inline-formula>
<mml:math display="inline" id="im85"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>U</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mo>&#xb7;</mml:mo><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>V</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mo>&#xb7;</mml:mo><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> by employing a common set of linear transformation parameters <inline-formula>
<mml:math display="inline" id="im86"><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mi>a</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mi>d</mml:mi><mml:mo>&#xd7;</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mi>a</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#xd7;</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>. The transformation is defined as (<xref ref-type="disp-formula" rid="eq14">Equation 14</xref>):</p>
<disp-formula id="eq14"><label>(14)</label>
<mml:math display="block" id="M14"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>U</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>V</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>X</mml:mi><mml:msub><mml:mi>W</mml:mi><mml:mi>a</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mi>a</mml:mi></mml:msub></mml:mrow></mml:math>
</disp-formula>
<p>Formally, we denote <inline-formula>
<mml:math display="inline" id="im104"><mml:mrow><mml:mi>Y</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> by a bag of <inline-formula>
<mml:math display="inline" id="im105"><mml:mi>n</mml:mi></mml:math></inline-formula> deep instance, the attention- based MIL pooling is defined by as shown in (<xref ref-type="disp-formula" rid="eq15">Equations 15</xref>, <xref ref-type="disp-formula" rid="eq16">16</xref>):</p>
<disp-formula id="eq15"><label>(15)</label>
<mml:math display="block" id="M15"><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>L</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>S</mml:mi><mml:mi>V</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msub><mml:mo>&#xb7;</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math>
</disp-formula>
<disp-formula id="eq16"><label>(16)</label>
<mml:math display="block" id="M16"><mml:mrow><mml:mi>z</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:msub><mml:mi>y</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:mstyle></mml:mrow></mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im106"><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>S</mml:mi><mml:mi>V</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula>
<mml:math display="inline" id="im107"><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> are attention weights.</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Experiment</title>
<sec id="s3_1">
<label>3.1</label>
<title>Dataset</title>
<p>In this study, two distinct datasets were used for model development and evaluation. Both datasets extract the same features and do not specify whether the IDH-mutant are of the IDH1 or IDH2 type.</p>
<p>In-house data set: The first data set was acquired from the First Affiliated Hospital of Xi&#x2019;an Jiaotong University and comprised 789 MRI samples of 344 subjects (incorporating FLAIR, T1w, and T2w sequences, all occurrences of &#x201c;T1w&#x201d; refer to T1 without contrast injection, and &#x201c;T1c&#x201d; refers to T1 with contrast injection.), as summarized in <xref ref-type="table" rid="T1"><bold>Table&#xa0;1</bold></xref>. All patients underwent surgical resection followed by genetic sequencing to confirm IDH status. For patients diagnosed before 2021, we performed a centralized histopathological review. Two board-certified neuropathologists, who specialize in brain tumor pathology, re-examined the archival hematoxylin&#x2013;eosin and immunohistochemistry slides (including IDH1 R132H, ATRX, p53 and other markers when available) and reviewed the original molecular pathology reports. Based on these materials, tumors were reclassified according to the 2021 World Health Organization Classification of Tumors of the Central Nervous System (5th edition), using all available molecular markers. Importantly, no additional molecular testing (such as TERT, EGFR or chromosomal analyses) was performed because of the retrospective design and limited archival tissue in some case (<xref ref-type="bibr" rid="B30">30</xref>, <xref ref-type="bibr" rid="B31">31</xref>).</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Summary of datasets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Dataset</th>
<th valign="middle" colspan="2" align="center">TCIA</th>
<th valign="middle" colspan="2" align="center">Xi&#x2019;an</th>
</tr>
<tr>
<th valign="middle" align="center">IDH status</th>
<th valign="middle" align="center">Mutant</th>
<th valign="middle" align="center">Wild</th>
<th valign="middle" align="center">Mutant</th>
<th valign="middle" align="center">Wild</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Number</td>
<td valign="middle" align="center">100</td>
<td valign="middle" align="center">395</td>
<td valign="middle" align="center">147</td>
<td valign="middle" align="center">197</td>
</tr>
<tr>
<td valign="middle" align="center">Age</td>
<td valign="middle" align="center">38.80 &#xb1; 15.11</td>
<td valign="middle" align="center">61.54 &#xb1; 15.00</td>
<td valign="middle" align="center">47.58 &#xb1; 9.68</td>
<td valign="middle" align="center">52.74 &#xb1; 11.97</td>
</tr>
<tr>
<td valign="middle" align="center">Gender(M\F)</td>
<td valign="middle" align="center">47\153</td>
<td valign="middle" align="center">234\161</td>
<td valign="middle" align="center">90\57</td>
<td valign="middle" align="center">113\84</td>
</tr>
<tr>
<td valign="middle" align="center">Astrocytoma, IDH-mutant</td>
<td valign="middle" align="center">87(17.9%)</td>
<td valign="middle" align="center">\</td>
<td valign="middle" align="center">100(29.0%)</td>
<td valign="middle" align="center">\</td>
</tr>
<tr>
<td valign="middle" align="center">Astrocytoma, IDH-wildtype</td>
<td valign="middle" align="center">\</td>
<td valign="middle" align="center">24(4.7%)</td>
<td valign="middle" align="center">\</td>
<td valign="middle" align="center">17(4.9%)</td>
</tr>
<tr>
<td valign="middle" align="center">Astrocytoma, IDH-wildtype, NOS</td>
<td valign="middle" align="center">\</td>
<td valign="middle" align="center">\</td>
<td valign="middle" align="center">\</td>
<td valign="middle" align="center">53(15.4%)</td>
</tr>
<tr>
<td valign="middle" align="center">Oligodendroglioma</td>
<td valign="middle" align="center">13(2.5%)</td>
<td valign="middle" align="center">\</td>
<td valign="middle" align="center">47(13.6%)</td>
<td valign="middle" align="center">\</td>
</tr>
<tr>
<td valign="middle" align="center">Glioblastoma</td>
<td valign="middle" align="center">\</td>
<td valign="middle" align="center">371(74.6%)</td>
<td valign="middle" align="center">\</td>
<td valign="middle" align="center">99(28.7%)</td>
</tr>
<tr>
<td valign="middle" align="center">Diffuse midline glioma</td>
<td valign="middle" align="center">\</td>
<td valign="middle" align="center">\</td>
<td valign="middle" align="center">\</td>
<td valign="middle" align="center">28(8.1%)</td>
</tr>
<tr>
<td valign="middle" align="center">WHO grade 2</td>
<td valign="middle" align="center">46</td>
<td valign="middle" align="center">10</td>
<td valign="middle" align="center">111</td>
<td valign="middle" align="center">18</td>
</tr>
<tr>
<td valign="middle" align="center">WHO grade 3</td>
<td valign="middle" align="center">29</td>
<td valign="middle" align="center">14</td>
<td valign="middle" align="center">36</td>
<td valign="middle" align="center">6</td>
</tr>
<tr>
<td valign="middle" align="center">WHO grade 4</td>
<td valign="middle" align="center">25</td>
<td valign="middle" align="center">371</td>
<td valign="middle" align="center">\</td>
<td valign="middle" align="center">173</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Public data set: The second data set was obtained from The Cancer Imaging Archive (TCIA), which included 1485 MRI samples from 495 subjects (395 wild-type cases and 100 mutant-type cases). Each subject in the TCIA dataset provided Flair, T1w, and T2w modalities, together with documented IDH status. Both datasets underwent identical pre-processing and post-processing pipelines. The diagnosis of glioma for all patients was performed according to the 2021 WHO classification of tumors of the central nervous system, 5th edition (<xref ref-type="bibr" rid="B32">32</xref>). Specifically, MRI volumes were preserved in the DICOM format and standardized to a uniform dimension of <inline-formula>
<mml:math display="inline" id="im108"><mml:mrow><mml:mn>240</mml:mn><mml:mo>&#xd7;</mml:mo><mml:mn>240</mml:mn><mml:mo>&#xd7;</mml:mo><mml:mn>155</mml:mn></mml:mrow></mml:math></inline-formula>. Standard intensity normalization and artifact removal procedures were applied, followed by data augmentation strategies such as color jitter and random affine transformations to mitigate overfitting. This unified preprocessing approach allowed for direct comparisons between the two datasets and a robust evaluation of the proposed HAB-MIL framework.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Implementation details</title>
<p>The framework was implemented using Pytorch2.5.1, and all models were trained with NVIDIA 3090 GPU with CUDA 11.8. We use 3D convolutional neural networks as the deep instance generator. We set the output shape <inline-formula>
<mml:math display="inline" id="im109"><mml:mrow><mml:msup><mml:mi>H</mml:mi><mml:mo>*</mml:mo></mml:msup><mml:mo>&#xd7;</mml:mo><mml:msup><mml:mi>W</mml:mi><mml:mo>*</mml:mo></mml:msup><mml:mo>&#xd7;</mml:mo><mml:msup><mml:mi>S</mml:mi><mml:mo>*</mml:mo></mml:msup><mml:mo>&#xd7;</mml:mo><mml:mi>D</mml:mi></mml:mrow></mml:math></inline-formula> of <inline-formula>
<mml:math display="inline" id="im110"><mml:mi>&#x3c8;</mml:mi></mml:math></inline-formula>to be <inline-formula>
<mml:math display="inline" id="im111"><mml:mrow><mml:mn>2</mml:mn><mml:mo>&#xd7;</mml:mo><mml:mn>2</mml:mn><mml:mo>&#xd7;</mml:mo><mml:mn>2</mml:mn><mml:mo>&#xd7;</mml:mo><mml:mn>64</mml:mn></mml:mrow></mml:math></inline-formula> according to cross-validation. The input shapes of the MRI slices are <inline-formula>
<mml:math display="inline" id="im112"><mml:mrow><mml:mn>240</mml:mn><mml:mo>&#xd7;</mml:mo><mml:mn>240</mml:mn><mml:mo>&#xd7;</mml:mo><mml:mn>155</mml:mn></mml:mrow></mml:math></inline-formula> We set training epoch T to 100 and batch size to 2. Data enhancement strategies included color jitter and random affine transformations. The network was trained using the ADAM optimizer with an initial learning rate 1e-4 and weight decay of 1e-5. In all experiments, 60% of the subjects was used for training, 20% for model selection and hyperparameter tuning, and the remaining 20% for testing. The dataset was augmented with random flipping, random affine transformation, intensity scaling, color jitter as was done to train the segmentation network. No scans from the same subject were included in both the training and the testing samples to ensure data independence and avoid potential information leakage. Five-fold cross-validation is performed within the training set, where the validation portion in each fold is applied for model selection and hyperparameter verification. Each experiment was repeated five times to ensure fair comparisons. Evaluation metrics included accuracy (Acc), area under the curve (AUC), sensitivity (Sens), specificity (Spec) and ROC curves.</p>
</sec>
</sec>
<sec id="s4" sec-type="results">
<label>4</label>
<title>Result</title>
<sec id="s4_1">
<label>4.1</label>
<title>Ablation experiment</title>
<p>A series of ablation studies on the TCIA data set was carried out to evaluate the effectiveness of the two modules (CLE and DGA) in HAB-MIL. Initially, use only the 3D convolution and MIL modules (denoted as Ablation-1) to evaluate the performance of the backbone. In the backbone, features can still be extracted from three concatenated conventional magnetic resonance sequences for the prediction of IDH. Next, add only the CLE module to the backbone (denoted Ablation-2), using positional encoding as a feature, and train the model. Next, add only the DGA module to the backbone (denoted Ablation-3). The results of the ablation experiments are presented in <xref ref-type="table" rid="T2"><bold>Table&#xa0;2</bold></xref>. Compared to backbone results, the incorporation of tumor location encoding improved the accuracy of IDH prediction by 21.8%. This suggests that positional encoding enables our HAB-MIL to learn location-specific features, which aids in predicting consistent differences across all image regions. Then, by adding the DGA module and selecting key instances, the model&#x2019;s accuracy was further improved by 24%. Finally, combining all three modules results in the best prediction performance. Therefore, the CLE module can be used to explore the spatial relationships of pixels, and the DGA module helps select key instances, improving IDH prediction.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Classification results in the TCIA dataset of different modules in terms of AUC, ACC, SEN, SPE.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Benchmark</th>
<th valign="middle" align="center">Backbone</th>
<th valign="middle" align="center">CLE</th>
<th valign="middle" align="center">DGA</th>
<th valign="middle" align="center">AUC</th>
<th valign="middle" align="center">ACC</th>
<th valign="middle" align="center">SEN</th>
<th valign="middle" align="center">SPE</th>
<th valign="middle" align="center">Parameter size (M)</th>
<th valign="middle" align="center">Train time (h)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Ablation-1</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">0.532 &#xb1; 0.021</td>
<td valign="middle" align="center">0.514 &#xb1; 0.103</td>
<td valign="middle" align="center">0.587 &#xb1; 0.015</td>
<td valign="middle" align="center">0.612 &#xb1; 0.001</td>
<td valign="middle" align="center">1.165</td>
<td valign="middle" align="center">3.84</td>
</tr>
<tr>
<td valign="middle" align="center">Ablation-1</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">0.750 &#xb1; 0.027</td>
<td valign="middle" align="center">0.692 &#xb1; 0.001</td>
<td valign="middle" align="center">0.750 &#xb1; 0.038</td>
<td valign="middle" align="center">0.699 &#xb1; 0.018</td>
<td valign="middle" align="center">1.198</td>
<td valign="middle" align="center">5.83</td>
</tr>
<tr>
<td valign="middle" align="center">Ablation-1</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">0.772 &#xb1; 0.007</td>
<td valign="middle" align="center">0.712 &#xb1; 0.003</td>
<td valign="middle" align="center">0.781 &#xb1; 0.012</td>
<td valign="middle" align="center">0.727 &#xb1; 0.003</td>
<td valign="middle" align="center">1.255</td>
<td valign="middle" align="center">5.67</td>
</tr>
<tr>
<td valign="middle" align="center">HAB-MIL</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"><bold>0.917 &#xb1; 0.015</bold></td>
<td valign="middle" align="center"><bold>0.904 &#xb1; 0.001</bold></td>
<td valign="middle" align="center"><bold>0.858 &#xb1; 0.008</bold></td>
<td valign="middle" align="center"><bold>0.921 &#xb1; 0.016</bold></td>
<td valign="middle" align="center">1.313</td>
<td valign="middle" align="center">7.31</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Data are given as mean &#xb1; SD.</p></fn>
<fn>
<p>AUC, area under the curve; SEN, sensitivity; ACC, accuracy; SPE, specificity. All values in bold represent the optimal performance.</p></fn>
<fn>
<p>It applies to all tables unless otherwise noted.</p></fn>
</table-wrap-foot>
</table-wrap>
<p><xref ref-type="table" rid="T3"><bold>Table&#xa0;3</bold></xref> lists classification results of various methods for the prediction of IDH. Among the various methods compared, the SPE can be used as a baseline to measure the validity of segmentation. We first ablate the effects of our CLE in the HAB-MIL for the test. As can be seen, Sinusoidal Positional Encoding (SPE) which simply concatenates the sinusoidal positional encoding of the pixel locations (x, y) values directly into the encoder stage leads to a slight performance improvement of 2% in most metrics compared to those without it (denoted as &#x201c;w/o SPE&#x201d;). Bello propose neural positional encoding (NPE) for depth estimation (<xref ref-type="bibr" rid="B33">33</xref>). However, our HAB-MIL shows substantial performance improvements in all metrics by incorporating our CLE. Specifically, the AUC increased by 28.8% and the ACC by 27.1% compared to the model w/o SPE. This improvement can be attributed to the learnable auxiliary positional encoding, which allows the model to more flexibly capture and leverage position-specific feature information. From the comparison between these methods, it can be observed that CLE is more helpful for IDH prediction, as it incorporates information that accurately describe location and morphological features of tumor and peritumoral edema.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Comparison results among CLE module and different location coding functions in terms of AUC, ACC, SEN, SPE.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Method</th>
<th valign="middle" align="left">AUC</th>
<th valign="middle" align="left">ACC</th>
<th valign="middle" align="left">SEN</th>
<th valign="middle" align="left">SPE</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">w/o SPE</td>
<td valign="middle" align="left">0.629 &#xb1; 0.003</td>
<td valign="middle" align="left">0.633 &#xb1; 0.005</td>
<td valign="middle" align="left">0.618 &#xb1; 0.013</td>
<td valign="middle" align="left">0.607 &#xb1; 0.001</td>
</tr>
<tr>
<td valign="middle" align="left">SPE</td>
<td valign="middle" align="left">0.631 &#xb1; 0.025</td>
<td valign="middle" align="left">0.618 &#xb1; 0.018</td>
<td valign="middle" align="left">0.653 &#xb1; 0.001</td>
<td valign="middle" align="left">0.593 &#xb1; 0.035</td>
</tr>
<tr>
<td valign="middle" align="left">NPE</td>
<td valign="middle" align="left">0.863 &#xb1; 0.001</td>
<td valign="middle" align="left">0.813 &#xb1; 0.012</td>
<td valign="middle" align="left">0.792 &#xb1; 0.008</td>
<td valign="middle" align="left">0.901 &#xb1; 0.012</td>
</tr>
<tr>
<td valign="middle" align="left">CLE</td>
<td valign="middle" align="left"><bold>0.917 &#xb1; 0.015</bold></td>
<td valign="middle" align="left"><bold>0.904 &#xb1; 0.001</bold></td>
<td valign="middle" align="left"><bold>0.858 &#xb1; 0.008</bold></td>
<td valign="middle" align="left"><bold>0.921 &#xb1; 0.016</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>All values in bold represent the optimal performance.</p></fn>
</table-wrap-foot>
</table-wrap>
<p>To evaluate the effectiveness and feasibility of the DGA module within HAB-MIL, we considered three attention weighting methods: max pooling, average pooling, and AB-MIL approaches. These methods were compared against our proposed HAB-MIL model. Specifically, the max-pooling-based and average-pooling-based methods follow the traditional MIL assumption, where the final subject-level prediction is obtained from the most significant instance or the average of all instances within a bag, respectively. In contrast, the attention-pooling-based method leverages an attention mechanism to assign weights to each instance embedding, facilitating learning at the bag level. In contrast, HAB-MIL considers a dynamic attention mechanism to selectively focus on the most relevant instances, effectively minimizing the inclusion of irrelevant information. As shown in <xref ref-type="table" rid="T4"><bold>Table&#xa0;4</bold></xref>, our proposed HAB-MIL achieved the best overall performance, attaining an accuracy of 90.4%. Max-pooling demonstrated the lowest performance among all methods, with an AUC of 69.3%, which is expected as it only considers the single most prominent instance for prediction. In contrast, the mean-based approach, while accounting for all instances, introduces a significant amount of irrelevant information, leading to less satisfactory results. Moreover, although plain attention pooling improved AUC and ACC by 5.3% and 8.9% respectively at the instance level compared to traditional methods, our proposed HAB-MIL achieves superior performance in terms of accuracy, recall, precision and F1, indicating that the effectiveness of the DGA module.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Comparison results among DGA module and different weight functions in terms of AUC, ACC, SEN, SPE.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Method</th>
<th valign="middle" align="left">AUC</th>
<th valign="middle" align="left">ACC</th>
<th valign="middle" align="left">SEN</th>
<th valign="middle" align="left">SPE</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Max-pooling</td>
<td valign="middle" align="left">0.693 &#xb1; 0.013</td>
<td valign="middle" align="left">0.612 &#xb1; 0.008</td>
<td valign="middle" align="left">0.635 &#xb1; 0.007</td>
<td valign="middle" align="left">0.594 &#xb1; 0.018</td>
</tr>
<tr>
<td valign="middle" align="left">Mean-pooling</td>
<td valign="middle" align="left">0.712 &#xb1; 0.021</td>
<td valign="middle" align="left">0.683 &#xb1; 0.001</td>
<td valign="middle" align="left">0.712 &#xb1; 0.035</td>
<td valign="middle" align="left">0.853 &#xb1; 0.002</td>
</tr>
<tr>
<td valign="middle" align="left">Attention-based</td>
<td valign="middle" align="left">0.765 &#xb1; 0.135</td>
<td valign="middle" align="left">0.701 &#xb1; 0.100</td>
<td valign="middle" align="left">0.761 &#xb1; 0.092</td>
<td valign="middle" align="left">0.732 &#xb1; 0.001</td>
</tr>
<tr>
<td valign="middle" align="left">DGA</td>
<td valign="middle" align="left"><bold>0.917 &#xb1; 0.015</bold></td>
<td valign="middle" align="left"><bold>0.904 &#xb1; 0.001</bold></td>
<td valign="middle" align="left"><bold>0.858 &#xb1; 0.008</bold></td>
<td valign="middle" align="left"><bold>0.921 &#xb1; 0.016</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>All values in bold represent the optimal performance.</p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Different combination of MRI sequences</title>
<p>We designed experiments to demonstrate the necessity of using different sequences simultaneously in the model. As shown in <xref ref-type="table" rid="T5"><bold>Tables&#xa0;5</bold></xref> and <xref ref-type="table" rid="T6"><bold>6</bold></xref>, removing the T1w or T2w sequence from the input of the model results in a decrease in IDH prediction performance. <xref ref-type="fig" rid="f4"><bold>Figure&#xa0;4</bold></xref> shows the ROC curves of different sequences in the TCIA dataset. This indicates that the T1w and T2w sequences are crucial for the IDH prediction task, as they guide the network to focus more on the tumor region and extract features that are highly relevant for gliomas.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Classification results in the TCIA dataset of three different sequences in terms of AUC, ACC, SEN, SPE.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Method</th>
<th valign="middle" align="left">AUC</th>
<th valign="middle" align="left">ACC</th>
<th valign="middle" align="left">SEN</th>
<th valign="middle" align="left">SPE</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">T1</td>
<td valign="middle" align="left">0.583 &#xb1; 0.135</td>
<td valign="middle" align="left">0.502 &#xb1; 0.031</td>
<td valign="middle" align="left">0.623 &#xb1; 0.528</td>
<td valign="middle" align="left">0.175 &#xb1; 0.291</td>
</tr>
<tr>
<td valign="middle" align="left">T2</td>
<td valign="middle" align="left">0.667 &#xb1; 0.161</td>
<td valign="middle" align="left">0.680 &#xb1; 0.003</td>
<td valign="middle" align="left">0.608 &#xb1; 0.060</td>
<td valign="middle" align="left">0.730 &#xb1; 0.012</td>
</tr>
<tr>
<td valign="middle" align="left">FLAIR</td>
<td valign="middle" align="left">0.625 &#xb1; 0.032</td>
<td valign="middle" align="left">0.658 &#xb1; 0.031</td>
<td valign="middle" align="left">0.637 &#xb1; 0.012</td>
<td valign="middle" align="left">0.708 &#xb1; 0.139</td>
</tr>
<tr>
<td valign="middle" align="left">T1+T2</td>
<td valign="middle" align="left">0.854 &#xb1; 0.010</td>
<td valign="middle" align="left">0.711 &#xb1; 0.021</td>
<td valign="middle" align="left">0.692 &#xb1; 0.211</td>
<td valign="middle" align="left">0.789 &#xb1; 0.091</td>
</tr>
<tr>
<td valign="middle" align="left">T1+FLAIR</td>
<td valign="middle" align="left">0.792 &#xb1; 0.164</td>
<td valign="middle" align="left">0.820 &#xb1; 0.012</td>
<td valign="middle" align="left">0.703 &#xb1; 0.001</td>
<td valign="middle" align="left">0.838 &#xb1; 0.032</td>
</tr>
<tr>
<td valign="middle" align="left">T2+FLAIR</td>
<td valign="middle" align="left">0.833 &#xb1; 0.021</td>
<td valign="middle" align="left">0.767 &#xb1; 0.121</td>
<td valign="middle" align="left">0.675 &#xb1; 0.230</td>
<td valign="middle" align="left">0.776 &#xb1; 0.021</td>
</tr>
<tr>
<td valign="middle" align="left">HAB-MIL</td>
<td valign="middle" align="left"><bold>0.917 &#xb1; 0.015</bold></td>
<td valign="middle" align="left"><bold>0.904 &#xb1; 0.001</bold></td>
<td valign="middle" align="left"><bold>0.858 &#xb1; 0.008</bold></td>
<td valign="middle" align="left"><bold>0.921 &#xb1; 0.016</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>All values in bold represent the optimal performance.</p></fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>Classification results in the in-house dataset of three different sequences in terms of AUC, ACC, SEN, SPE.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Method</th>
<th valign="middle" align="left">AUC</th>
<th valign="middle" align="left">ACC</th>
<th valign="middle" align="left">SEN</th>
<th valign="middle" align="left">SPE</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">T1</td>
<td valign="middle" align="left">0.523 &#xb1; 0.031</td>
<td valign="middle" align="left">0.561 &#xb1; 0.218</td>
<td valign="middle" align="left">0.612 &#xb1; 0.015</td>
<td valign="middle" align="left">0.616 &#xb1; 0.102</td>
</tr>
<tr>
<td valign="middle" align="left">T2</td>
<td valign="middle" align="left">0.576 &#xb1; 0.005</td>
<td valign="middle" align="left">0.583 &#xb1; 0.035</td>
<td valign="middle" align="left">0.638 &#xb1; 0.005</td>
<td valign="middle" align="left">0.593 &#xb1; 0.001</td>
</tr>
<tr>
<td valign="middle" align="left">FLAIR</td>
<td valign="middle" align="left">0.608 &#xb1; 0.032</td>
<td valign="middle" align="left">0.672 &#xb1; 0.010</td>
<td valign="middle" align="left">0.665 &#xb1; 0.012</td>
<td valign="middle" align="left">0.656 &#xb1; 0.002</td>
</tr>
<tr>
<td valign="middle" align="left">T1+T2</td>
<td valign="middle" align="left">0.761 &#xb1; 0.013</td>
<td valign="middle" align="left">0.831 &#xb1; 0.105</td>
<td valign="middle" align="left">0.791 &#xb1; 0.019</td>
<td valign="middle" align="left">0.753 &#xb1; 0.052</td>
</tr>
<tr>
<td valign="middle" align="left">T1+FLAIR</td>
<td valign="middle" align="left">0.733 &#xb1; 0.031</td>
<td valign="middle" align="left">0.793 &#xb1; 0.005</td>
<td valign="middle" align="left"><bold>0.812 &#xb1; 0.324</bold></td>
<td valign="middle" align="left">0.792 &#xb1; 0.168</td>
</tr>
<tr>
<td valign="middle" align="left">T2+FLAIR</td>
<td valign="middle" align="left">0.833 &#xb1; 0.012</td>
<td valign="middle" align="left">0.809 &#xb1; 0.006</td>
<td valign="middle" align="left">0.762 &#xb1; 0.109</td>
<td valign="middle" align="left">0.837 &#xb1; 0.020</td>
</tr>
<tr>
<td valign="middle" align="left">HAB-MIL</td>
<td valign="middle" align="left"><bold>0.892 &#xb1; 0.031</bold></td>
<td valign="middle" align="left"><bold>0.857 &#xb1; 0.012</bold></td>
<td valign="middle" align="left">0.791 &#xb1; 0.003</td>
<td valign="middle" align="left"><bold>0.848 &#xb1; 0.131</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>All values in bold represent the optimal performance.</p></fn>
</table-wrap-foot>
</table-wrap>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>ROC curves of different sequences in the TCIA dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1665690-g004.tif">
<alt-text content-type="machine-generated">Receiver Operating Characteristic (ROC) curve showing true positive rate versus false positive rate for various models: T1, T2, Flair, T1+T2, T1+FLAIR, T2+FLAIR, and HAB-MIL. HAB-MIL has the highest area under the curve (AUC) of 0.917. The curves compare model performance, with the diagonal line representing random chance.</alt-text>
</graphic></fig>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Comparison with the state-of-the-art algorithms</title>
<p>To validate the proposed HAB-MIL algorithm for IDH statusclassification more effectively, we compared our proposed HAB-MIL algorithm with eight state-of-the-art IDH prediction methods, including Radiomics approaches (<xref ref-type="bibr" rid="B34">34</xref>&#x2013;<xref ref-type="bibr" rid="B36">36</xref>), MDL (<xref ref-type="bibr" rid="B37">37</xref>), MultiGeneNet (<xref ref-type="bibr" rid="B38">38</xref>), FAD (<xref ref-type="bibr" rid="B39">39</xref>), SGPNet (<xref ref-type="bibr" rid="B40">40</xref>), PS-Net (<xref ref-type="bibr" rid="B41">41</xref>), MFEF Net (<xref ref-type="bibr" rid="B24">24</xref>), MTTU Net (<xref ref-type="bibr" rid="B42">42</xref>), MTS-UNET (<xref ref-type="bibr" rid="B43">43</xref>) and GLISP (<xref ref-type="bibr" rid="B44">44</xref>). The results of the different methods are shown in the <xref ref-type="table" rid="T7"><bold>Table&#xa0;7</bold></xref>. By comparing the proposed method with other IDH prediction networks, the following results can be observed.</p>
<table-wrap id="T7" position="float">
<label>Table&#xa0;7</label>
<caption>
<p>Performance of all classifiers on testing set.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Model</th>
<th valign="middle" align="left">AUC</th>
<th valign="middle" align="left">ACC</th>
<th valign="middle" align="left">SEN</th>
<th valign="middle" align="left">SPE</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">He</td>
<td valign="middle" align="left">0.873 &#xb1; 0.050</td>
<td valign="middle" align="left">0.876 &#xb1; 0.090</td>
<td valign="middle" align="left">0.875 &#xb1; 0.110</td>
<td valign="middle" align="left">0.877 &#xb1; 0.150</td>
</tr>
<tr>
<td valign="middle" align="left">Bumes</td>
<td valign="middle" align="left">0.820</td>
<td valign="middle" align="left">0.761</td>
<td valign="middle" align="left">0.826</td>
<td valign="middle" align="left">0.727</td>
</tr>
<tr>
<td valign="middle" align="left">Kandalgaonkar</td>
<td valign="middle" align="left">0.890</td>
<td valign="middle" align="left">0.890</td>
<td valign="middle" align="left">0.800</td>
<td valign="middle" align="left">0.030</td>
</tr>
<tr>
<td valign="middle" align="left">VGG16 <xref ref-type="table-fn" rid="fnT7_1"><sup>a</sup></xref></td>
<td valign="middle" align="left">0.718 &#xb1; 0.131</td>
<td valign="middle" align="left">0.684 &#xb1; 0.172</td>
<td valign="middle" align="left">0.725 &#xb1; 0.231</td>
<td valign="middle" align="left">0.606 &#xb1; 0.108</td>
</tr>
<tr>
<td valign="middle" align="left">DenseNet <xref ref-type="table-fn" rid="fnT7_2"><sup>b</sup></xref></td>
<td valign="middle" align="left">0.791 &#xb1; 0.324</td>
<td valign="middle" align="left">0.838 &#xb1; 0.291</td>
<td valign="middle" align="left">0.776 &#xb1; 0.162</td>
<td valign="middle" align="left">0.701 &#xb1; 0.031</td>
</tr>
<tr>
<td valign="middle" align="left">ResNet-50 <xref ref-type="table-fn" rid="fnT7_3"><sup>c</sup></xref></td>
<td valign="middle" align="left">0.732 &#xb1; 0.002</td>
<td valign="middle" align="left">0.714 &#xb1; 0.015</td>
<td valign="middle" align="left">0.782 &#xb1; 0.184</td>
<td valign="middle" align="left">0.584 &#xb1; 0.320</td>
</tr>
<tr>
<td valign="middle" align="left">Inception-v3 <xref ref-type="table-fn" rid="fnT7_4"><sup>d</sup></xref></td>
<td valign="middle" align="left">0.706 &#xb1; 0.272</td>
<td valign="middle" align="left">0.668 &#xb1; 0.069</td>
<td valign="middle" align="left">0.634 &#xb1; 0.142</td>
<td valign="middle" align="left">0.596 &#xb1; 0.165</td>
</tr>
<tr>
<td valign="middle" align="left">MDL (<xref ref-type="bibr" rid="B37">37</xref>)</td>
<td valign="middle" align="left">0.872 &#xb1; 0.039</td>
<td valign="middle" align="left">0.812 &#xb1; 0.021</td>
<td valign="middle" align="left">0.740 &#xb1; 0.003</td>
<td valign="middle" align="left">0.849 &#xb1; 0.001</td>
</tr>
<tr>
<td valign="middle" align="left">MultiGeneNet (<xref ref-type="bibr" rid="B38">38</xref>)</td>
<td valign="middle" align="left">0.886 &#xb1; 0.001</td>
<td valign="middle" align="left">0.835 &#xb1; 0.035</td>
<td valign="middle" align="left">0.755 &#xb1; 0.058</td>
<td valign="middle" align="left">0.833 &#xb1; 0.036</td>
</tr>
<tr>
<td valign="middle" align="left">FAD (<xref ref-type="bibr" rid="B39">39</xref>)</td>
<td valign="middle" align="left">0.801 &#xb1; 0.031</td>
<td valign="middle" align="left">0.767 &#xb1; 0.116</td>
<td valign="middle" align="left">0.774 &#xb1; 0.236</td>
<td valign="middle" align="left">0.734 &#xb1; 0.021</td>
</tr>
<tr>
<td valign="middle" align="left">SGP Net (<xref ref-type="bibr" rid="B40">40</xref>)</td>
<td valign="middle" align="left">0.837 &#xb1; 0.056</td>
<td valign="middle" align="left">0.785 &#xb1; 0.165</td>
<td valign="middle" align="left">0.794 &#xb1; 0.003</td>
<td valign="middle" align="left">0.685 &#xb1; 0.355</td>
</tr>
<tr>
<td valign="middle" align="left">PS-NET (<xref ref-type="bibr" rid="B41">41</xref>)</td>
<td valign="middle" align="left">0.821 &#xb1; 0.013</td>
<td valign="middle" align="left">0.791 &#xb1; 0.067</td>
<td valign="middle" align="left">0.789 &#xb1; 0.147</td>
<td valign="middle" align="left">0.773 &#xb1; 0.236</td>
</tr>
<tr>
<td valign="middle" align="left">MFEF Net (<xref ref-type="bibr" rid="B24">24</xref>)</td>
<td valign="middle" align="left">0.856 &#xb1; 0.102</td>
<td valign="middle" align="left">0.802 &#xb1; 0.102</td>
<td valign="middle" align="left">0.837 &#xb1; 0.312</td>
<td valign="middle" align="left">0.813 &#xb1; 0.052</td>
</tr>
<tr>
<td valign="middle" align="left">MTTU Net (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="middle" align="left">0.903 &#xb1; 0.056</td>
<td valign="middle" align="left">0.857 &#xb1; 0.099</td>
<td valign="middle" align="left">0.812 &#xb1; 0.069</td>
<td valign="middle" align="left">0.894 &#xb1; 0.128</td>
</tr>
<tr>
<td valign="middle" align="left">MTS-UNET (<xref ref-type="bibr" rid="B43">43</xref>)</td>
<td valign="middle" align="left">0.803 &#xb1; 0.010</td>
<td valign="middle" align="left">0.868 &#xb1; 0.044</td>
<td valign="middle" align="left">0.813 &#xb1; 0.028</td>
<td valign="middle" align="left">0.796 &#xb1; 0.030</td>
</tr>
<tr>
<td valign="middle" align="left">GLISP (<xref ref-type="bibr" rid="B44">44</xref>)</td>
<td valign="middle" align="left">0.750 &#xb1; 0.028</td>
<td valign="middle" align="left">0.720 &#xb1; 0.039</td>
<td valign="middle" align="left">0.590 &#xb1; 0.037</td>
<td valign="middle" align="left">0.680 &#xb1; 0.106</td>
</tr>
<tr>
<td valign="middle" align="left">HAB-MIL</td>
<td valign="middle" align="left"><bold>0.917 &#xb1; 0.015</bold></td>
<td valign="middle" align="left"><bold>0.904 &#xb1; 0.001</bold></td>
<td valign="middle" align="left"><bold>0.858 &#xb1; 0.008</bold></td>
<td valign="middle" align="left"><bold>0.921 &#xb1; 0.016</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="fnT7_1"><label>a</label>
<p>Visual Geometry Group 16-layer network.</p></fn>
<fn id="fnT7_2"><label>b</label>
<p>Densely Connected Convolutional Network.</p></fn>
<fn id="fnT7_3"><label>c</label>
<p>Residual Network 50-layer.</p></fn>
<fn id="fnT7_4"><label>d</label>
<p>Inception Version 3.</p></fn>
<fn>
<p>All values in bold represent the optimal performance.</p></fn>
</table-wrap-foot>
</table-wrap>
<p>He, Bumes, and Kandalgaonkar et&#xa0;al. applied radiomics-based approaches in their studies. In terms of data acquisition, Bumes utilized magnetic resonance spectroscopy (MRS), while He employed contrast-enhanced T1-weighted imaging (T1C). Although multimodal data were incorporated, insufficient feature standardization or inter-modality registration may have hindered the effective integration of information, resulting in amplified noise and reduced model performance. Moreover, conventional radiomics methods are limited to extracting shallow, handcrafted features, which may fail to capture the complex pathological heterogeneity underlying gliomas.</p>
<p>Firstly, GLISP model for patch-level prediction employs a lightweight CNN architecture. While this design ensures computational efficiency, it may fall short in capturing deeper and more complex pathological features compared to more advanced models. Secondly, the prediction accuracy of MultiGeneNet is higher than that of FAD. This is because MultiGeneNet is designed to simultaneously predict multiple key genetic mutations. It employs a shared feature extraction backbone with separate classification branches for each task, enabling effective feature sharing while preserving task-specific distinctions&#x2014;ultimately improving overall performance. However, the model processes patch samples from whole-slide images without explicitly modeling intra-tumoral heterogeneity, which may result in the loss of important regional variations within the tumor.</p>
<p>Then MTS-UNET architecture lacks an explicit positional encoding mechanism, which may hinder its ability to capture spatial context&#x2014;such as the relative positioning of lesions&#x2014;particularly in scenarios involving high tumor heterogeneity. The poor performance of MTDL may be due to the loss of information on intra-tumoral heterogeneity and the variation in tumor size between patients. The model searches for the largest tumor bounding box and then crops the input image to a fixed large size without incorporating the tumor mask information. This approach can introduce irrelevant background information into the model, particularly when the tumor size is small.</p>
<p>Although the MFEF net model incorporates advanced modules such as segmentation-guided feature extraction, asymmetric amplification, and dual-attention feature fusion, the asymmetric amplification module may amplify not only pathological features, but also noise. Additionally, the feature spaces of the T2w and Flair sequences may differ significantly. Calculating these differences directly can introduce inconsistencies, affecting the stability of feature representation and prediction performance, resulting in suboptimal outcomes. MTTU Net achieved good results using a CNN transformer encoder, but it employs an uncertainty-aware pseudo-label selection method to generate and select pseudo-label. If the initial model generates a significant number of incorrect pseudo-labels, these errors may propagate and amplify with each iteration, causing the model to gradually deviate from the correct target and fall into a negative feedback loop.</p>
<p>Hence, the proposed HAB-MIL model is designed to address the limitations of approaches by explicitly accounting for the high heterogeneity of gliomas. It integrates positional encoding during the feature extraction phase and leverages the DGA module to emphasize the significance of key instances. As a result, it achieves superior performance in IDH status prediction compared to all other evaluated methods.</p>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Interpretative visualization</title>
<p><xref ref-type="fig" rid="f5"><bold>Figure&#xa0;5</bold></xref> illustrates how our proposed HAB-MIL model localizes discriminative regions for the prediction of IDH status. In particular, the three columns of images are shown for each subject: (1) Original image of different patients, (2) key patches identified by HAB-MIL, and (3) Grad-CAM visualizations produced by HAB-MIL. According to <xref ref-type="fig" rid="f5"><bold>Figure&#xa0;5</bold></xref>, the Key patches in wild-type IDH tumors are more widely distributed, encompassing both the tumor core and surrounding regions. In contrast, key patches in mutant-type IDH tumors are predominantly confined to the tumor core or adjacent areas, indicating a more localized distribution and reduced invasiveness.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Interpretability of the HAB-MIL model. The first column is FLAIR slices extracted from different patients. Meanwhile, the second column is key patches from HAB-MIL, whereas the third column is their corresponding Grad-CAMs with the HAB-MIL.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1665690-g005.tif">
<alt-text content-type="machine-generated">Brain MRI images comparing IDH-Wild Type and IDH-Mutant Type. Left column shows original images, middle column highlights key patches, and right column presents Grad-CAMs with heat maps indicating areas of importance.</alt-text>
</graphic></fig>
<p>The boundaries of wild-type IDH tumors appear relatively indistinct, and some tumor regions show low contrast to surrounding tissues, suggesting a tendency to invasive growth. In comparison, mutant-type IDH tumors show well-defined boundaries, concentrated lesions, and more prominent high signals in the tumor core, indicating a denser internal structure. These observations highlight significant differences in morphological characteristics, invasiveness, and growth patterns between the two types of tumors, offering valuable imaging-based insights to guide treatment decisions and prognosis assessment.</p>
</sec>
</sec>
<sec id="s5" sec-type="discussion">
<label>5</label>
<title>Discussion</title>
<p>Despite the significant success of deep neural networks in the field of IDH prediction, it still encounters challenges related to weak interpretability and low credibility. In clinical practice, classification of IDH status is crucial for guiding subsequent treatment strategies and predicting the prognosis of patients. Currently, the only way to determine IDH status is to obtain pathological tissue by open surgery. Consequently, if we can describe the imaging distinctions between IDH-wild and IDH-mutant and integrate these characteristics into the diagnostic model, the interpretability of the model would be enhanced, and its classification performance could be further improved. Following this strategy, a novel Hierarchical Attention-Based Multiple Instance Learning (HAB-MIL) framework is proposed. To effectively capture tumor location information, auxiliary positional encoding was employed to encode imaging features, which were then concatenated with the feature maps derived from a deep instance-level feature extractor, as illustrated in <xref ref-type="fig" rid="f1"><bold>Figure&#xa0;1</bold></xref>. Tumor location, serving as a non-invasive biomarker, not only significantly enhances the accuracy of the classification model but also provides valuable insights for preoperative prediction of IDH status. Additionally, we developed a weakly supervised learning-based classification network, which substantially reduces annotation costs and better accommodates the spatial heterogeneity of gliomas. This approach facilitates the automatic identification of key regional features closely associated with IDH prediction, thereby improving the model&#x2019;s accuracy, generalization performance, and clinical interpretability.</p>
<p><xref ref-type="table" rid="T2"><bold>Table&#xa0;2</bold></xref> presents the results of the ablation study on different modules. As shown in table, the model&#x2019;s performance improves to a certain extent with the addition of both modules compared to using the backbone alone. <xref ref-type="table" rid="T3"><bold>Table&#xa0;3</bold></xref> presents the ablation study results for different positional encoding methods. Testing with various positional encodings led to slight variations in the performance of the proposed algorithm, demonstrating the flexibility and effectiveness of GeLU. <xref ref-type="table" rid="T4"><bold>Table&#xa0;4</bold></xref> compares the performance of different attention mechanisms. The dynamic attention mechanism used in this study improves the precision of feature selection through a gating mechanism and dropout regularization, while also enhancing the model&#x2019;s generalization capability. This approach is particularly well-suited for applications in few-shot learning or multi-instance learning scenarios.</p>
<p><xref ref-type="table" rid="T5"><bold>Tables&#xa0;5</bold></xref> and <xref ref-type="table" rid="T6"><bold>6</bold></xref> present the results of different modality combinations on the TCIA dataset and the in-house dataset, respectively. The performance on the TCIA dataset is slightly better than that on the in-house dataset, likely due to the strict quality control typically applied to public datasets prior to release. In contrast, the quality of in-house data may be affected by factors such as the acquisition environment, equipment variability, or human error. In both datasets, the combination of three modalities consistently yields better performance, which attributed to the varying sensitivities of different modalities to tissue structures, lesions, and fluids. By integrating multiple modalities, more comprehensive and informative imaging data can be obtained. Additionally, the network shows increased robustness to missing modalities when setting the T1 and T2 or FLAIR sequence to zero while training. There is only a small decrease in performance when only providing the T1w, T2w and Flair scans compared to all three MRI as input. This is especially useful as not all three MRI modalities are available for all patients in the Xi&#x2019;an Jiaotong University Hospital dataset. This way an accurate prediction could still be obtained for these patients. The very high specificity and slightly lower sensitivity indicate that prediction inaccuracies are due to parts of the surrounding edema that are not detected by the network.</p>
<p>An in-depth analysis of <xref ref-type="fig" rid="f4"><bold>Figure&#xa0;4</bold></xref> and the attention weights learned by the model reveals a consistent anatomical distribution pattern of lDH-mutant gliomas. Specifically, gliomas located in the thalamus and cerebellum are predominantly IDH wild-type, whereas those in the insular cortex are more likely to harbor IDH mutations. This observation is consistent with previous radiological and histopathological studies, further validating the reliability of our model. Compared to existing methods, our approach offers three distinct advantages: first, HAB-MIL accurately identifies subtle imaging differences between IDH wild-type and lDH-mutant gliomas through key instance selection, helping clinicians pinpoint regions that are most informative for distinguishing IDH status. Second, the process of identifying key instances is intuitive and straightforward to implement. Finally, the clinical application of this model requires only routine MRI sequences (T1w, T2w, and FLAIR) as input, without the need for contrast agents, thereby reducing potential risks in practical use. Furthermore, the model processes the entire MRI image directly, requiring only skull-stripping and registration steps, without manual tumor delineation by radiologists. This design minimizes reliance on specialized expertise and facilitates broader clinical implementation. This study still has several limitations. First, the HAB-MIL model has not yet been validated in large-scale, multicenter clinical settings; therefore, its feasibility and stability in real-world clinical practice remain to be further confirmed. Second, some cases in this study lacked key molecular markers such as TERT promoter mutation, EGFR amplification, and chromosome 7 gain/chromosome 10 loss (+7/&#x2212;10).</p>
</sec>
<sec id="s6" sec-type="conclusions">
<label>6</label>
<title>Conclusion</title>
<p>In this study, we proposed a HAB-MIL framework, contributing to the prediction of IDH status using only routinely acquired preoperative MRI. Compared with conventional clinical workflows that require labor-intensive, slice-by-slice tumor annotation, HAB-MIL employs a weakly supervised learning strategy that relies solely on case-level labels, eliminating the need for detailed lesion delineation. This approach significantly reduces annotation time and manual effort while preserving model accuracy. Moreover, the model requires only routine MRI sequences such as T1-weighted, T2-weighted, and FLAIR images, without the need for contrast agents, thereby minimizing potential procedural risks and reducing the overall financial burden on patients.</p>
<p>In addition, future work will focus on further refining and expanding this study. First, we plan to validate the HAB-MIL model in large-scale, multicenter clinical settings to evaluate its feasibility and stability in real-world clinical practice. Second, we will collect more diverse datasets encompassing cases acquired from different imaging devices and scanning protocols, as well as additional modalities such as MR perfusion and DTI, to further enhance the model&#x2019;s generalization and robustness. Finally, we will conduct comparative analyses with other noninvasive prediction methods to comprehensively evaluate the model&#x2019;s performance.</p>
</sec>
</body>
<back>
<sec id="s7" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material
. Further inquiries can be directed to the corresponding author.</p></sec>
<sec id="s8" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The studies involving humans were approved by The Ethics Committee of the First Affiliated Hospital of Xi&#x2019;an Jiaotong University. The studies were conducted in accordance with the local legislation and institutional requirements. Written informed consent for participation was not required from the participants or the participants&#x2019; legal guardians/next of kin in accordance with the national legislation and institutional requirements.</p></sec>
<sec id="s9" sec-type="author-contributions">
<title>Author contributions</title>
<p>QX: Formal analysis, Software, Validation, Visualization, Writing &#x2013; original draft. YHS: Software, Writing &#x2013; review &amp; editing. YL: Data curation, Writing &#x2013; review &amp; editing. YS: Data curation, Writing &#x2013; review &amp; editing. HW: Methodology, Software, Writing &#x2013; review &amp; editing. FW: Methodology, Writing &#x2013; review &amp; editing. RW: Resources, Supervision, Writing &#x2013; review &amp; editing. BC: Resources, Supervision, Writing &#x2013; review &amp; editing. MZ: Resources, Supervision, Writing &#x2013; review &amp; editing. CN: Funding acquisition, Resources, Writing &#x2013; review &amp; editing.</p></sec>
<sec id="s11" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declared that this work was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p></sec>
<sec id="s12" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declared that generative AI was not used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p></sec>
<sec id="s13" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p></sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Schaff</surname> <given-names>LR</given-names></name>
<name><surname>Mellinghoff</surname> <given-names>IK</given-names></name>
</person-group>. 
<article-title>Glioblastoma and other primary brain Malignancies in adults: A review</article-title>. <source>JAMA</source>. (<year>2023</year>) <volume>329</volume>:<fpage>574</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1001/jama.2023.0023</pub-id>, PMID: <pub-id pub-id-type="pmid">36809318</pub-id>
</mixed-citation>
</ref>
<ref id="B2">
<label>2</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Weller</surname> <given-names>M</given-names></name>
<name><surname>Wen</surname> <given-names>PY</given-names></name>
<name><surname>Chang</surname> <given-names>SM</given-names></name>
<name><surname>Dirven</surname> <given-names>L</given-names></name>
<name><surname>Lim</surname> <given-names>M</given-names></name>
<name><surname>Monje</surname> <given-names>M</given-names></name>
<etal/>
</person-group>. 
<article-title>Glioma</article-title>. <source>Nat Rev Dis Primers</source>. (<year>2024</year>) <volume>10</volume>:<fpage>33</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41572-024-00516-y</pub-id>, PMID: <pub-id pub-id-type="pmid">38724526</pub-id>
</mixed-citation>
</ref>
<ref id="B3">
<label>3</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Singh</surname> <given-names>S</given-names></name>
<name><surname>Dey</surname> <given-names>D</given-names></name>
<name><surname>Barik</surname> <given-names>D</given-names></name>
<name><surname>Mohapatra</surname> <given-names>I</given-names></name>
<name><surname>Kim</surname> <given-names>S</given-names></name>
<name><surname>Sharma</surname> <given-names>M</given-names></name>
<etal/>
</person-group>. 
<article-title>Glioblastoma at the crossroads: current understanding and future therapeutic horizons</article-title>. <source>Sig Transduct Target Ther</source>. (<year>2025</year>) <volume>10</volume>:<fpage>213</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41392-025-02299-4</pub-id>, PMID: <pub-id pub-id-type="pmid">40628732</pub-id>
</mixed-citation>
</ref>
<ref id="B4">
<label>4</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Carosi</surname> <given-names>F</given-names></name>
<name><surname>Broseghini</surname> <given-names>E</given-names></name>
<name><surname>Fabbri</surname> <given-names>L</given-names></name>
<name><surname>Corradi</surname> <given-names>G</given-names></name>
<name><surname>Gili</surname> <given-names>R</given-names></name>
<name><surname>Forte</surname> <given-names>V</given-names></name>
<etal/>
</person-group>. 
<article-title>Targeting isocitrate dehydrogenase (IDH) in solid tumors: current evidence and future perspectives</article-title>. <source>Cancers</source>. (<year>2024</year>) <volume>16</volume>:<elocation-id>2752</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/cancers16152752</pub-id>, PMID: <pub-id pub-id-type="pmid">39123479</pub-id>
</mixed-citation>
</ref>
<ref id="B5">
<label>5</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Rud&#xe0;</surname> <given-names>R</given-names></name>
<name><surname>Bruno</surname> <given-names>F</given-names></name>
<name><surname>Ius</surname> <given-names>T</given-names></name>
<name><surname>Silvani</surname> <given-names>A</given-names></name>
<name><surname>Minniti</surname> <given-names>G</given-names></name>
<name><surname>Pace</surname> <given-names>A</given-names></name>
<etal/>
</person-group>. 
<article-title>IDH wild-type grade 2 diffuse astrocytomas: prognostic factors and impact of treatments within molecular subgroups</article-title>. <source>Neuro-Oncology</source>. (<year>2022</year>) <volume>24</volume>:<page-range>809&#x2013;20</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/neuonc/noab239</pub-id>, PMID: <pub-id pub-id-type="pmid">34651653</pub-id>
</mixed-citation>
</ref>
<ref id="B6">
<label>6</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Ramos-Fresnedo</surname> <given-names>A</given-names></name>
<name><surname>Pullen</surname> <given-names>MW</given-names></name>
<name><surname>Perez-Vega</surname> <given-names>C</given-names></name>
<name><surname>Domingo</surname> <given-names>RA</given-names></name>
<name><surname>Akinduro</surname> <given-names>OO</given-names></name>
<name><surname>Almeida</surname> <given-names>JP</given-names></name>
<etal/>
</person-group>. 
<article-title>The survival outcomes of molecular glioblastoma IDH-wildtype: a multicenter study</article-title>. <source>J Neurooncol</source>. (<year>2022</year>) <volume>157</volume>:<page-range>177&#x2013;85</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11060-022-03960-6</pub-id>, PMID: <pub-id pub-id-type="pmid">35175545</pub-id>
</mixed-citation>
</ref>
<ref id="B7">
<label>7</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Solomou</surname> <given-names>G</given-names></name>
<name><surname>Finch</surname> <given-names>A</given-names></name>
<name><surname>Asghar</surname> <given-names>A</given-names></name>
<name><surname>Bardella</surname> <given-names>C</given-names></name>
</person-group>. 
<article-title>Mutant IDH in gliomas: role in cancer and treatment options</article-title>. <source>Cancers</source>. (<year>2023</year>) <volume>15</volume>:<elocation-id>2883</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/cancers15112883</pub-id>, PMID: <pub-id pub-id-type="pmid">37296846</pub-id>
</mixed-citation>
</ref>
<ref id="B8">
<label>8</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Miller</surname> <given-names>JJ</given-names></name>
<name><surname>Gonzalez Castro</surname> <given-names>LN</given-names></name>
<name><surname>McBrayer</surname> <given-names>S</given-names></name>
<name><surname>Weller</surname> <given-names>M</given-names></name>
<name><surname>Cloughesy</surname> <given-names>T</given-names></name>
<name><surname>Portnow</surname> <given-names>J</given-names></name>
<etal/>
</person-group>. 
<article-title>Isocitrate dehydrogenase (IDH) mutant gliomas: A Society for Neuro-Oncology (SNO) consensus review on diagnosis, management, and future directions</article-title>. <source>Neuro-Oncology</source>. (<year>2023</year>) <volume>25</volume>:<fpage>4</fpage>&#x2013;<lpage>25</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/neuonc/noac207</pub-id>, PMID: <pub-id pub-id-type="pmid">36239925</pub-id>
</mixed-citation>
</ref>
<ref id="B9">
<label>9</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Yu</surname> <given-names>D</given-names></name>
<name><surname>Zhong</surname> <given-names>Q</given-names></name>
<name><surname>Xiao</surname> <given-names>Y</given-names></name>
<name><surname>Feng</surname> <given-names>Z</given-names></name>
<name><surname>Tang</surname> <given-names>F</given-names></name>
<name><surname>Feng</surname> <given-names>S</given-names></name>
<etal/>
</person-group>. 
<article-title>Combination of MRI-based prediction and CRISPR/Cas12a-based detection for IDH genotyping in glioma</article-title>. <source>NPJ Precis Onc</source>. (<year>2024</year>) <volume>8</volume>:<fpage>140</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41698-024-00632-8</pub-id>, PMID: <pub-id pub-id-type="pmid">38951603</pub-id>
</mixed-citation>
</ref>
<ref id="B10">
<label>10</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Jeon</surname> <given-names>YH</given-names></name>
<name><surname>Choi</surname> <given-names>KS</given-names></name>
<name><surname>Lee</surname> <given-names>KH</given-names></name>
<name><surname>Jeong</surname> <given-names>SY</given-names></name>
<name><surname>Lee</surname> <given-names>JY</given-names></name>
<name><surname>Ham</surname> <given-names>T</given-names></name>
<etal/>
</person-group>. 
<article-title>Deep learning-based quantification of T2-FLAIR mismatch sign: extending IDH mutation prediction in adult-type diffuse lower-grade glioma</article-title>. <source>Eur Radiol</source>. (<year>2025</year>) <volume>35</volume>:<page-range>5193&#x2013;202</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00330-025-11475-7</pub-id>, PMID: <pub-id pub-id-type="pmid">40050456</pub-id>
</mixed-citation>
</ref>
<ref id="B11">
<label>11</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Bangalore Yogananda</surname> <given-names>CG</given-names></name>
<name><surname>Truong</surname> <given-names>NCD</given-names></name>
<name><surname>Wagner</surname> <given-names>BC</given-names></name>
<name><surname>Xi</surname> <given-names>Y</given-names></name>
<name><surname>Bowerman</surname> <given-names>J</given-names></name>
<name><surname>Reddy</surname> <given-names>DD</given-names></name>
<etal/>
</person-group>. 
<article-title>Bridging the clinical gap: Confidence informed IDH prediction in brain gliomas using MRI and deep learning</article-title>. <source>Neuro-Oncol Adv</source>. (<year>2025</year>) <volume>7</volume>:<elocation-id>vdaf142</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/noajnl/vdaf142</pub-id>, PMID: <pub-id pub-id-type="pmid">40842645</pub-id>
</mixed-citation>
</ref>
<ref id="B12">
<label>12</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Negro</surname> <given-names>A</given-names></name>
<name><surname>Gemini</surname> <given-names>L</given-names></name>
<name><surname>Tortora</surname> <given-names>M</given-names></name>
<name><surname>Pace</surname> <given-names>G</given-names></name>
<name><surname>Iaccarino</surname> <given-names>R</given-names></name>
<name><surname>Marchese</surname> <given-names>M</given-names></name>
<etal/>
</person-group>. 
<article-title>VASARI 2.0: a new updated MRI VASARI lexicon to predict grading and IDH status in brain glioma</article-title>. <source>Front Oncol</source>. (<year>2024</year>) <volume>14</volume>:<elocation-id>1449982</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fonc.2024.1449982</pub-id>, PMID: <pub-id pub-id-type="pmid">39763601</pub-id>
</mixed-citation>
</ref>
<ref id="B13">
<label>13</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Jopek</surname> <given-names>MA</given-names></name>
<name><surname>Pastuszak</surname> <given-names>K</given-names></name>
<name><surname>Cygert</surname> <given-names>S</given-names></name>
<name><surname>Best</surname> <given-names>MG</given-names></name>
<name><surname>Wurdinger</surname> <given-names>T</given-names></name>
<name><surname>Jassem</surname> <given-names>J</given-names></name>
<etal/>
</person-group>. 
<article-title>Deep learning-based, multiclass approach to cancer classification on liquid biopsy data</article-title>. <source>IEEE J Transl Eng Health Med</source>. (<year>2024</year>) <volume>12</volume>:<page-range>306&#x2013;13</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/JTEHM.2024.3360865</pub-id>
</mixed-citation>
</ref>
<ref id="B14">
<label>14</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Elwenspoek</surname> <given-names>MMC</given-names></name>
<name><surname>Sheppard</surname> <given-names>AL</given-names></name>
<name><surname>McInnes</surname> <given-names>MDF</given-names></name>
<name><surname>Merriel</surname> <given-names>SWD</given-names></name>
<name><surname>Rowe</surname> <given-names>EWJ</given-names></name>
<name><surname>Bryant</surname> <given-names>RJ</given-names></name>
<etal/>
</person-group>. 
<article-title>Comparison of multiparametric magnetic resonance imaging and targeted biopsy with systematic biopsy alone for the diagnosis of prostate cancer: A systematic review and meta-analysis</article-title>. <source>JAMA Netw Open</source>. (<year>2019</year>) <volume>2</volume>:<fpage>e198427</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1001/jamanetworkopen.2019.8427</pub-id>, PMID: <pub-id pub-id-type="pmid">31390032</pub-id>
</mixed-citation>
</ref>
<ref id="B15">
<label>15</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Gong</surname> <given-names>Y</given-names></name>
<name><surname>Liu</surname> <given-names>G</given-names></name>
<name><surname>Xue</surname> <given-names>Y</given-names></name>
<name><surname>Li</surname> <given-names>R</given-names></name>
<name><surname>Meng</surname> <given-names>L</given-names></name>
</person-group>. 
<article-title>A survey on dataset quality in machine learning</article-title>. <source>Inf Softw Technol</source>. (<year>2023</year>) <volume>162</volume>:<elocation-id>107268</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.infsof.2023.107268</pub-id>
</mixed-citation>
</ref>
<ref id="B16">
<label>16</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Gadermayr</surname> <given-names>M</given-names></name>
<name><surname>Tschuchnig</surname> <given-names>M</given-names></name>
</person-group>. 
<article-title>Multiple instance learning for digital pathology: A review of the state-of-the-art, limitations &amp; future potential</article-title>. <source>Comput Med Imaging Graphics</source>. (<year>2024</year>) <volume>112</volume>:<elocation-id>102337</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compmedimag.2024.102337</pub-id>, PMID: <pub-id pub-id-type="pmid">38228020</pub-id>
</mixed-citation>
</ref>
<ref id="B17">
<label>17</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Li</surname> <given-names>Z</given-names></name>
<name><surname>Wang</surname> <given-names>Y</given-names></name>
<name><surname>Zhu</surname> <given-names>Y</given-names></name>
<name><surname>Xu</surname> <given-names>J</given-names></name>
<name><surname>Wei</surname> <given-names>J</given-names></name>
<name><surname>Xie</surname> <given-names>J</given-names></name>
<etal/>
</person-group>. 
<article-title>Modality-based attention and dual-stream multiple instance convolutional neural network for predicting microvascular invasion of hepatocellular carcinoma</article-title>. <source>Front Oncol</source>. (<year>2023</year>) <volume>13</volume>:<elocation-id>1195110</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fonc.2023.1195110</pub-id>, PMID: <pub-id pub-id-type="pmid">37434971</pub-id>
</mixed-citation>
</ref>
<ref id="B18">
<label>18</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Mahadevkar</surname> <given-names>SV</given-names></name>
<name><surname>Khemani</surname> <given-names>B</given-names></name>
<name><surname>Patil</surname> <given-names>S</given-names></name>
<name><surname>Kotecha</surname> <given-names>K</given-names></name>
<name><surname>Vora</surname> <given-names>DR</given-names></name>
<name><surname>Abraham</surname> <given-names>A</given-names></name>
<etal/>
</person-group>. 
<article-title>A review on machine learning styles in computer vision&#x2014;Techniques and future directions</article-title>. <source>IEEE Access</source>. (<year>2022</year>) <volume>10</volume>:<page-range>107293&#x2013;329</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2022.3209825</pub-id>
</mixed-citation>
</ref>
<ref id="B19">
<label>19</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Liang</surname> <given-names>X</given-names></name>
<name><surname>Li</surname> <given-names>X</given-names></name>
<name><surname>Li</surname> <given-names>F</given-names></name>
<name><surname>Jiang</surname> <given-names>J</given-names></name>
<name><surname>Dong</surname> <given-names>Q</given-names></name>
<name><surname>Wang</surname> <given-names>W</given-names></name>
<etal/>
</person-group>. 
<article-title>MedFILIP: medical fine-grained language-image pre-training</article-title>. <source>IEEE J BioMed Health Inform</source>. (<year>2025</year>) <volume>29</volume>:<page-range>3587&#x2013;97</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/JBHI.2025.3528196</pub-id>, PMID: <pub-id pub-id-type="pmid">40030972</pub-id>
</mixed-citation>
</ref>
<ref id="B20">
<label>20</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Shi</surname> <given-names>X</given-names></name>
<name><surname>Xing</surname> <given-names>F</given-names></name>
<name><surname>Xie</surname> <given-names>Y</given-names></name>
<name><surname>Zhang</surname> <given-names>Z</given-names></name>
<name><surname>Cui</surname> <given-names>L</given-names></name>
<name><surname>Yang</surname> <given-names>L</given-names></name>
</person-group>. 
<article-title>Loss-based attention for deep multiple instance learning</article-title>. <source>AAAI</source>. (<year>2020</year>) <volume>34</volume>:<page-range>5742&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1609/aaai.v34i04.6030</pub-id>
</mixed-citation>
</ref>
<ref id="B21">
<label>21</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Konstantinov</surname> <given-names>AV</given-names></name>
<name><surname>Utkin</surname> <given-names>LV</given-names></name>
</person-group>. 
<article-title>Multi-attention multiple instance learning</article-title>. <source>Neural Comput Applic</source>. (<year>2022</year>) <volume>34</volume>:<page-range>14029&#x2013;51</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00521-022-07259-5</pub-id>
</mixed-citation>
</ref>
<ref id="B22">
<label>22</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Kim</surname> <given-names>D</given-names></name>
<name><surname>Lee</surname> <given-names>J</given-names></name>
<name><surname>Jung</surname> <given-names>M</given-names></name>
<name><surname>Yim</surname> <given-names>K</given-names></name>
<name><surname>Hwang</surname> <given-names>G</given-names></name>
<name><surname>Yoon</surname> <given-names>H</given-names></name>
<etal/>
</person-group>. 
<article-title>Whole slide image-level classification of Malignant effusion cytology using clustering-constrained attention multiple instance learning</article-title>. <source>Lung Cancer</source>. (<year>2025</year>) <volume>204</volume>:<elocation-id>108552</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.lungcan.2025.108552</pub-id>, PMID: <pub-id pub-id-type="pmid">40311308</pub-id>
</mixed-citation>
</ref>
<ref id="B23">
<label>23</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Lu</surname> <given-names>MY</given-names></name>
<name><surname>Williamson</surname> <given-names>DFK</given-names></name>
<name><surname>Chen</surname> <given-names>TY</given-names></name>
<name><surname>Chen</surname> <given-names>RJ</given-names></name>
<name><surname>Barbieri</surname> <given-names>M</given-names></name>
<name><surname>Mahmood</surname> <given-names>F</given-names></name>
</person-group>. 
<article-title>Data-efficient and weakly supervised computational pathology on whole-slide images</article-title>. <source>Nat BioMed Eng</source>. (<year>2021</year>) <volume>5</volume>:<page-range>555&#x2013;70</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41551-020-00682-w</pub-id>, PMID: <pub-id pub-id-type="pmid">33649564</pub-id>
</mixed-citation>
</ref>
<ref id="B24">
<label>24</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Zhang</surname> <given-names>J</given-names></name>
<name><surname>Cao</surname> <given-names>J</given-names></name>
<name><surname>Tang</surname> <given-names>F</given-names></name>
<name><surname>Xie</surname> <given-names>T</given-names></name>
<name><surname>Feng</surname> <given-names>Q</given-names></name>
<name><surname>Huang</surname> <given-names>M</given-names></name>
</person-group>. 
<article-title>Multi-level feature exploration and fusion network for prediction of IDH status in gliomas from MRI</article-title>. <source>IEEE J BioMed Health Inform</source>. (<year>2024</year>) <volume>28</volume>:<fpage>42</fpage>&#x2013;<lpage>53</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/JBHI.2023.3279433</pub-id>, PMID: <pub-id pub-id-type="pmid">37247314</pub-id>
</mixed-citation>
</ref>
<ref id="B25">
<label>25</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Rani</surname> <given-names>V</given-names></name>
<name><surname>Kumar</surname> <given-names>M</given-names></name>
<name><surname>Gupta</surname> <given-names>A</given-names></name>
<name><surname>Sachdeva</surname> <given-names>M</given-names></name>
<name><surname>Mittal</surname> <given-names>A</given-names></name>
<name><surname>Kumar</surname> <given-names>K</given-names></name>
</person-group>. 
<article-title>Self-supervised learning for medical image analysis: a comprehensive review</article-title>. <source>Evolving Syst</source>. (<year>2024</year>) <volume>15</volume>:<page-range>1607&#x2013;33</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s12530-024-09581-w</pub-id>
</mixed-citation>
</ref>
<ref id="B26">
<label>26</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Dominguez-Morales</surname> <given-names>JP</given-names></name>
<name><surname>Duran-Lopez</surname> <given-names>L</given-names></name>
<name><surname>Marini</surname> <given-names>N</given-names></name>
<name><surname>Vicente-Diaz</surname> <given-names>S</given-names></name>
<name><surname>Linares-Barranco</surname> <given-names>A</given-names></name>
<name><surname>Atzori</surname> <given-names>M</given-names></name>
<etal/>
</person-group>. 
<article-title>A systematic comparison of deep learning methods for Gleason grading and scoring</article-title>. <source>Med Image Anal</source>. (<year>2024</year>) <volume>95</volume>:<elocation-id>103191</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.media.2024.103191</pub-id>, PMID: <pub-id pub-id-type="pmid">38728903</pub-id>
</mixed-citation>
</ref>
<ref id="B27">
<label>27</label>
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name><surname>Ding</surname> <given-names>Y</given-names></name>
<name><surname>Zhao</surname> <given-names>L</given-names></name>
<name><surname>Yuan</surname> <given-names>L</given-names></name>
<name><surname>Wen</surname> <given-names>X</given-names></name>
</person-group>. (<year>2022</year>). 
<article-title>Deep multi-instance learning with adaptive recurrent pooling for medical image classification</article-title>, in: <conf-name>2022 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</conf-name>, <conf-loc>Las Vegas, NV, USA</conf-loc>. pp. <page-range>3335&#x2013;42</page-range>. 
<publisher-name>IEEE</publisher-name>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/BIBM55620.2022.9995191</pub-id>
</mixed-citation>
</ref>
<ref id="B28">
<label>28</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Cheplygina</surname> <given-names>V</given-names></name>
<name><surname>de Bruijne</surname> <given-names>M</given-names></name>
<name><surname>Pluim</surname> <given-names>JPW</given-names></name>
</person-group>. 
<article-title>Not-so-supervised: A survey of semi-supervised, multi-instance, and transfer learning in medical image analysis</article-title>. <source>Med Image Anal</source>. (<year>2019</year>) <volume>54</volume>:<page-range>280&#x2013;96</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.media.2019.03.009</pub-id>, PMID: <pub-id pub-id-type="pmid">30959445</pub-id>
</mixed-citation>
</ref>
<ref id="B29">
<label>29</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Liang</surname> <given-names>X</given-names></name>
<name><surname>Han</surname> <given-names>L</given-names></name>
<name><surname>Zhang</surname> <given-names>X</given-names></name>
<name><surname>Li</surname> <given-names>X</given-names></name>
<name><surname>Sun</surname> <given-names>Y</given-names></name>
<name><surname>Tong</surname> <given-names>T</given-names></name>
<etal/>
</person-group>. 
<article-title>Singular value decomposition based under-sampling pattern optimization for MRI reconstruction</article-title>. <source>Med Phys</source>. (<year>2025</year>) <volume>52</volume>:<fpage>e17860</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/mp.17860</pub-id>, PMID: <pub-id pub-id-type="pmid">40296184</pub-id>
</mixed-citation>
</ref>
<ref id="B30">
<label>30</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Musigmann</surname> <given-names>M</given-names></name>
<name><surname>Bilgin</surname> <given-names>M</given-names></name>
<name><surname>Bilgin</surname> <given-names>SS</given-names></name>
<name><surname>Kr&#xe4;hling</surname> <given-names>H</given-names></name>
<name><surname>Heindel</surname> <given-names>W</given-names></name>
<name><surname>Mannil</surname> <given-names>M</given-names></name>
</person-group>. 
<article-title>Completely non-invasive prediction of IDH mutation status based on preoperative native CT images</article-title>. <source>Sci Rep</source>. (<year>2024</year>) <volume>14</volume>:<fpage>26763</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-024-77789-6</pub-id>, PMID: <pub-id pub-id-type="pmid">39501053</pub-id>
</mixed-citation>
</ref>
<ref id="B31">
<label>31</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Hosseini</surname> <given-names>S</given-names></name>
<name><surname>Hosseini</surname> <given-names>E</given-names></name>
<name><surname>Hajianfar</surname> <given-names>G</given-names></name>
<name><surname>Shiri</surname> <given-names>I</given-names></name>
<name><surname>Servaes</surname> <given-names>S</given-names></name>
<name><surname>Rosa-Neto</surname> <given-names>P</given-names></name>
<etal/>
</person-group>. 
<article-title>MRI-based radiomics combined with deep learning for distinguishing IDH-mutant WHO grade 4 astrocytomas from IDH-wild-type glioblastomas</article-title>. <source>Cancers</source>. (<year>2023</year>) <volume>15</volume>:<elocation-id>951</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/cancers15030951</pub-id>, PMID: <pub-id pub-id-type="pmid">36765908</pub-id>
</mixed-citation>
</ref>
<ref id="B32">
<label>32</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Calabrese</surname> <given-names>E</given-names></name>
<name><surname>Villanueva-Meyer</surname> <given-names>JE</given-names></name>
<name><surname>Rudie</surname> <given-names>JD</given-names></name>
<name><surname>Rauschecker</surname> <given-names>AM</given-names></name>
<name><surname>Baid</surname> <given-names>U</given-names></name>
<name><surname>Bakas</surname> <given-names>S</given-names></name>
<etal/>
</person-group>. 
<article-title>The university of california san francisco preoperative diffuse glioma MRI dataset</article-title>. <source>Radiol: Artif Intell</source>. (<year>2022</year>) <volume>4</volume>:<fpage>e220058</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1148/ryai.220058</pub-id>, PMID: <pub-id pub-id-type="pmid">36523646</pub-id>
</mixed-citation>
</ref>
<ref id="B33">
<label>33</label>
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name><surname>Bello</surname> <given-names>JLG</given-names></name>
<name><surname>Kim</surname> <given-names>M</given-names></name>
</person-group>. (<year>2021</year>). 
<article-title>PLADE-net: towards pixel-level accuracy for self-supervised single-view depth estimation with neural positional encoding and distilled matting loss</article-title>, in: <conf-name>2021 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <conf-loc>Nashville, TN, USA</conf-loc>. pp. <page-range>6847&#x2013;56</page-range>. 
<publisher-name>IEEE</publisher-name>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR46437.2021.00678</pub-id>
</mixed-citation>
</ref>
<ref id="B34">
<label>34</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>He</surname> <given-names>A</given-names></name>
<name><surname>Wang</surname> <given-names>P</given-names></name>
<name><surname>Zhu</surname> <given-names>A</given-names></name>
<name><surname>Liu</surname> <given-names>Y</given-names></name>
<name><surname>Chen</surname> <given-names>J</given-names></name>
<name><surname>Liu</surname> <given-names>L</given-names></name>
</person-group>. 
<article-title>Predicting IDH mutation status in low-grade gliomas based on optimal radiomic features combined with multi-sequence magnetic resonance imaging</article-title>. <source>Diagnostics</source>. (<year>2022</year>) <volume>12</volume>:<elocation-id>2995</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/diagnostics12122995</pub-id>, PMID: <pub-id pub-id-type="pmid">36553002</pub-id>
</mixed-citation>
</ref>
<ref id="B35">
<label>35</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Bumes</surname> <given-names>E</given-names></name>
<name><surname>Fellner</surname> <given-names>C</given-names></name>
<name><surname>Fellner</surname> <given-names>FA</given-names></name>
<name><surname>Fleischanderl</surname> <given-names>K</given-names></name>
<name><surname>H&#xe4;ckl</surname> <given-names>M</given-names></name>
<name><surname>Lenz</surname> <given-names>S</given-names></name>
<etal/>
</person-group>. 
<article-title>Validation study for non-invasive prediction of IDH mutation status in patients with glioma using <italic>in vivo</italic> 1H-magnetic resonance spectroscopy and machine learning</article-title>. <source>Cancers</source>. (<year>2022</year>) <volume>14</volume>:<elocation-id>2762</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/cancers14112762</pub-id>, PMID: <pub-id pub-id-type="pmid">35681741</pub-id>
</mixed-citation>
</ref>
<ref id="B36">
<label>36</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Kandalgaonkar</surname> <given-names>P</given-names></name>
<name><surname>Sahu</surname> <given-names>A</given-names></name>
<name><surname>Saju</surname> <given-names>AC</given-names></name>
<name><surname>Joshi</surname> <given-names>A</given-names></name>
<name><surname>Mahajan</surname> <given-names>A</given-names></name>
<name><surname>Thakur</surname> <given-names>M</given-names></name>
<etal/>
</person-group>. 
<article-title>Predicting IDH subtype of grade 4 astrocytoma and glioblastoma from tumor radiomic patterns extracted from multiparametric magnetic resonance images using a machine learning approach</article-title>. <source>Front Oncol</source>. (<year>2022</year>) <volume>12</volume>:<elocation-id>879376</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fonc.2022.879376</pub-id>, PMID: <pub-id pub-id-type="pmid">36276136</pub-id>
</mixed-citation>
</ref>
<ref id="B37">
<label>37</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Wu</surname> <given-names>X</given-names></name>
<name><surname>Zhang</surname> <given-names>S</given-names></name>
<name><surname>Zhang</surname> <given-names>Z</given-names></name>
<name><surname>He</surname> <given-names>Z</given-names></name>
<name><surname>Xu</surname> <given-names>Z</given-names></name>
<name><surname>Wang</surname> <given-names>W</given-names></name>
<etal/>
</person-group>. 
<article-title>Biologically interpretable multi-task deep learning pipeline predicts molecular alterations, grade, and prognosis in glioma patients</article-title>. <source>NPJ Precis Onc</source>. (<year>2024</year>) <volume>8</volume>:<fpage>181</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41698-024-00670-2</pub-id>, PMID: <pub-id pub-id-type="pmid">39152182</pub-id>
</mixed-citation>
</ref>
<ref id="B38">
<label>38</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Liu</surname> <given-names>X</given-names></name>
<name><surname>Hu</surname> <given-names>W</given-names></name>
<name><surname>Diao</surname> <given-names>S</given-names></name>
<name><surname>Abera</surname> <given-names>DE</given-names></name>
<name><surname>Racoceanu</surname> <given-names>D</given-names></name>
<name><surname>Qin</surname> <given-names>W</given-names></name>
</person-group>. 
<article-title>Multi-scale feature fusion for prediction of IDH1 mutations in glioma histopathological images</article-title>. <source>Comput Methods Programs Biomed</source>. (<year>2024</year>) <volume>248</volume>:<elocation-id>108116</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cmpb.2024.108116</pub-id>, PMID: <pub-id pub-id-type="pmid">38518408</pub-id>
</mixed-citation>
</ref>
<ref id="B39">
<label>39</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Choi</surname> <given-names>YS</given-names></name>
<name><surname>Bae</surname> <given-names>S</given-names></name>
<name><surname>Chang</surname> <given-names>JH</given-names></name>
<name><surname>Kang</surname> <given-names>S-G</given-names></name>
<name><surname>Kim</surname> <given-names>SH</given-names></name>
<name><surname>Kim</surname> <given-names>J</given-names></name>
<etal/>
</person-group>. 
<article-title>Fully automated hybrid approach to predict the <italic>IDH</italic> mutation status of gliomas via deep learning and radiomics</article-title>. <source>Neuro-Oncology</source>. (<year>2021</year>) <volume>23</volume>:<page-range>304&#x2013;13</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/neuonc/noaa177</pub-id>, PMID: <pub-id pub-id-type="pmid">32706862</pub-id>
</mixed-citation>
</ref>
<ref id="B40">
<label>40</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Wang</surname> <given-names>Y</given-names></name>
<name><surname>Wang</surname> <given-names>Y</given-names></name>
<name><surname>Guo</surname> <given-names>C</given-names></name>
<name><surname>Zhang</surname> <given-names>S</given-names></name>
<name><surname>Yang</surname> <given-names>L</given-names></name>
</person-group>. 
<article-title>SGPNet: A three-dimensional multitask residual framework for segmentation and IDH genotype prediction of gliomas</article-title>. <source>Comput Intell Neurosci</source>. (<year>2021</year>) <volume>2021</volume>:<elocation-id>5520281</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1155/2021/5520281</pub-id>
</mixed-citation>
</ref>
<ref id="B41">
<label>41</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>van der Voort</surname> <given-names>SR</given-names></name>
<name><surname>Incekara</surname> <given-names>F</given-names></name>
<name><surname>Wijnenga</surname> <given-names>MMJ</given-names></name>
<name><surname>Kapsas</surname> <given-names>G</given-names></name>
<name><surname>Gahrmann</surname> <given-names>R</given-names></name>
<name><surname>Schouten</surname> <given-names>JW</given-names></name>
<etal/>
</person-group>. 
<article-title>Combined molecular subtyping, grading, and segmentation of glioma using multi-task deep learning</article-title>. <source>Neuro-Oncology</source>. (<year>2023</year>) <volume>25</volume>:<page-range>279&#x2013;89</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/neuonc/noac166</pub-id>, PMID: <pub-id pub-id-type="pmid">35788352</pub-id>
</mixed-citation>
</ref>
<ref id="B42">
<label>42</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Cheng</surname> <given-names>J</given-names></name>
<name><surname>Liu</surname> <given-names>J</given-names></name>
<name><surname>Kuang</surname> <given-names>H</given-names></name>
<name><surname>Wang</surname> <given-names>J</given-names></name>
</person-group>. 
<article-title>A fully automated multimodal MRI-based multi-task learning for glioma segmentation and IDH genotyping</article-title>. <source>IEEE Trans Med Imaging</source>. (<year>2022</year>) <volume>41</volume>:<page-range>1520&#x2013;32</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TMI.2022.3142321</pub-id>, PMID: <pub-id pub-id-type="pmid">35020590</pub-id>
</mixed-citation>
</ref>
<ref id="B43">
<label>43</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Farahani</surname> <given-names>S</given-names></name>
<name><surname>Hejazi</surname> <given-names>M</given-names></name>
<name><surname>Di Ieva</surname> <given-names>A</given-names></name>
<name><surname>Fatemizadeh</surname> <given-names>E</given-names></name>
<name><surname>Liu</surname> <given-names>S</given-names></name>
</person-group>. 
<article-title>Towards a multimodal MRI-based foundation model for multi-level feature exploration in segmentation, molecular subtyping, and grading of glioma</article-title>. (<year>2025</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2503.06828</pub-id>
</mixed-citation>
</ref>
<ref id="B44">
<label>44</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Le</surname> <given-names>M-K</given-names></name>
<name><surname>Kawai</surname> <given-names>M</given-names></name>
<name><surname>Masui</surname> <given-names>K</given-names></name>
<name><surname>Komori</surname> <given-names>T</given-names></name>
<name><surname>Kawamata</surname> <given-names>T</given-names></name>
<name><surname>Muragaki</surname> <given-names>Y</given-names></name>
<etal/>
</person-group>. 
<article-title>Glioma image-level and slide-level gene predictor (GLISP) for molecular diagnosis and predicting genetic events of adult diffuse glioma</article-title>. <source>Bioengineering</source>. (<year>2024</year>) <volume>12</volume>:<elocation-id>12</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/bioengineering12010012</pub-id>, PMID: <pub-id pub-id-type="pmid">39851286</pub-id>
</mixed-citation>
</ref>
</ref-list>
<fn-group>
<fn id="n1" fn-type="custom" custom-type="edited-by">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/955237">Francesco Bruno</ext-link>, University and City of Health and Science Hospital, Italy</p></fn>
<fn id="n2" fn-type="custom" custom-type="reviewed-by">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2865926">Giulia Berzero</ext-link>, IRCCS Ospedale San Raffaele, Italy</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3244203">Yu Liang</ext-link>, Harbin University of Science and Technology, China</p></fn>
</fn-group>
</back>
</article>