<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Oncol.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Oncology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Oncol.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2234-943X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fonc.2025.1619994</article-id>
<article-version article-version-type="Corrected Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Predicting breast cancer treatment response and prognosis using AI-based image classification</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Wang</surname><given-names>Bingyi</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project-administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Chen</surname><given-names>Shu</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>*</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Funding acquisition" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname><given-names>Wei</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3050716/overview"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label><institution>Department of Radiation Oncology,Clinical Oncology School of Fujian Medical University, Fujian Cancer Hospital, NHC Key Laboratory of Cancer Metabolism</institution>, <city>Fuzhou</city>,&#xa0;<country country="cn">China</country></aff>
<aff id="aff2"><label>2</label><institution>Department of Gastric Surgery, Clinical Oncology School of Fujian Medical University, Fujian Cancer Hospital, NHC Key Laboratory of Cancer Metabolism</institution>, <city>Fuzhou</city>,&#xa0;<country country="cn">China</country></aff>
<aff id="aff3"><label>3</label><institution>Medical School, Yangzhou University</institution>, <city>Yangzhou</city>,&#xa0;<country country="cn">China</country></aff>
<author-notes>
<corresp id="c001"><label>*</label>Correspondence: Shu Chen, <email xlink:href="mailto:sensayspivak@hotmail.com">sensayspivak@hotmail.com</email></corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-10-21">
<day>21</day>
<month>10</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="corrected" iso-8601-date="2025-12-15">
<day>15</day>
<month>12</month>
<year>2025</year></pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>15</volume>
<elocation-id>1619994</elocation-id>
<history>
<date date-type="received">
<day>29</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>19</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Wang, Chen and Li.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Wang, Chen and Li</copyright-holder>
<license>
<ali:license_ref start_date="2025-10-21">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Accurate prediction of treatment response and prognosis in breast cancer patients is critical to advance personalized medicine and optimize therapeutic decision-making. Within the context of AI-enabled healthcare, there remains a pressing need to develop robust, interpretable models that can account for the temporal complexity and heterogeneity inherent in longitudinal patient data.</p>
</sec>
<sec>
<title>Methods</title>
<p>This study proposes a novel framework designed to model patient-specific treatment trajectories using a dynamics-aware, deep sequence learning architecture. Aligned with the core themes of computational prognostics and precision therapy, our method addresses the challenges posed by variable patient responses, missing clinical records, and complex pharmacological interactions. Existing approaches, including conventional supervised learning and static classification models, often fall short in capturing the underlying temporal dependencies, multimodal data fusion, and counterfactual reasoning necessary for real-world clinical deployment. These limitations hinder generalizability, especially in scenarios where treatment outcomes are delayed or weakly annotated. In contrast, our approach integrates recurrent modeling, attention mechanisms, and uncertainty quantification to better capture the evolving nature of patient health trajectories. Moreover, we incorporate domain-informed regularization techniques and causal inference modules to improve interpretability and clinical relevance.</p>
</sec>
<sec>
<title>Results and Discussion</title>
<p>By learning temporal dynamics in a personalized manner, the proposed model enhances predictive performance while remaining sensitive to patient-specific variations and therapeutic regimens. Through extensive validation on real-world breast cancer cohorts, we demonstrate that our framework not only outperforms existing baselines but also provides actionable insights that can inform adaptive treatment planning and risk stratification.</p>
</sec>
</abstract>
<kwd-group>
<kwd>breast cancer prognosis</kwd>
<kwd>treatment response prediction</kwd>
<kwd>latent dynamics modeling</kwd>
<kwd>symbolic knowledge infusion</kwd>
<kwd>AI in clinical decision support</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author declares financial support was received for the research and/or publication of this article. Joint funds for the Innovation of Science and Technology, Fujian Province (2024Y9636); National Clinical Key Specialty Construction Program, 2021; Natural Science Foundation of Fujian Province (Grant numbers:2025J01121548).</funding-statement>
</funding-group>
<counts>
<fig-count count="6"/>
<table-count count="5"/>
<equation-count count="30"/>
<ref-count count="51"/>
<page-count count="19"/>
<word-count count="12028"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Breast Cancer</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Breast cancer continues to be a primary contributor to cancer-associated illness and death among women on a global scale. Accurate prediction of treatment response and patient prognosis is essential to improving therapeutic strategies and clinical outcomes (<xref ref-type="bibr" rid="B1">1</xref>). Traditionally, such predictions have relied heavily on histopathological examination, molecular subtyping, and clinical staging; however, these approaches are often limited by inter-observer variability and incomplete capture of tumor heterogeneity. With the advent of digital pathology and the availability of high-resolution whole-slide images (WSIs), artificial intelligence (AI) offers a transformative opportunity (<xref ref-type="bibr" rid="B2">2</xref>). Not only can AI-driven image classification systems process vast amounts of image data with high consistency, but they can also uncover complex patterns that may not be perceptible to human experts. Moreover, these techniques enhance predictive accuracy by integrating morphological cues with computational precision, enabling clinicians to tailor treatments based on a more robust risk stratification (<xref ref-type="bibr" rid="B3">3</xref>). Therefore, developing AI-based models for image classification is not only necessary for optimizing individualized breast cancer therapy but also critical in advancing precision oncology.</p>
<p>Early computational strategies for analyzing histopathological images relied on predefined morphological descriptors and diagnostic protocols (<xref ref-type="bibr" rid="B4">4</xref>). These systems extracted interpretable characteristics&#x2014;such as nucleus size, texture, and spatial arrangement&#x2014;from tissue samples to support rule-based classification or grading (<xref ref-type="bibr" rid="B5">5</xref>). While these approaches aligned with traditional pathology workflows and offered transparency, they were limited in flexibility and struggled to capture the subtle and variable visual features present in large-scale WSIs. In particular, their performance was susceptible to staining inconsistencies, tumor heterogeneity, and variability across datasets (<xref ref-type="bibr" rid="B6">6</xref>).</p>
<p>As digital pathology advanced, researchers introduced more adaptable models capable of recognizing patterns directly from labeled examples (<xref ref-type="bibr" rid="B7">7</xref>). These methods employed classification algorithms trained on manually extracted features, allowing systems to differentiate tumor subtypes or predict outcomes with improved accuracy (<xref ref-type="bibr" rid="B8">8</xref>). Approaches such as support vector machines and ensemble classifiers demonstrated practical utility in medium-sized datasets and well-curated research cohorts. However, they still relied on handcrafted feature extraction pipelines, which imposed constraints on scalability and made it difficult to generalize findings across institutions or patient populations (<xref ref-type="bibr" rid="B9">9</xref>).</p>
<p>Recent innovations have led to end-to-end learning frameworks that automatically derive predictive representations from raw pathology images (<xref ref-type="bibr" rid="B10">10</xref>). Deep neural networks&#x2014;particularly convolutional architectures and attention-based models&#x2014;have enabled a patch-level analysis of WSIs, learning discriminative features that correspond to prognostic markers (<xref ref-type="bibr" rid="B11">11</xref>). These systems support the integration of contextual information and facilitate downstream tasks such as survival analysis, molecular subtype inference, and therapy response prediction (<xref ref-type="bibr" rid="B12">12</xref>). Despite achieving state-of-the-art performance, challenges remain in interpretability, computational demand, and the need for annotated training data. As a response, the development of explainable and resource-efficient architectures is gaining momentum, aiming to balance clinical reliability with the scalability of deep learning in pathology (<xref ref-type="bibr" rid="B13">13</xref>).</p>
<p>In clinical oncology, various biochemical parameters are routinely used for early tumor detection and monitoring. Radenkovic et&#xa0;al. highlighted the diagnostic significance of matrix metalloproteinases (MMP-2 and MMP-9) in basal-like breast cancer, reflecting their association with tumor invasiveness and progression (<xref ref-type="bibr" rid="B14">14</xref>). Another study by Radenkovic et&#xa0;al. emphasized the role of oxidative stress-related enzymes such as lactate dehydrogenase (LDH), catalase, and superoxide dismutase (SOD) in tumor tissues, showing that their expression levels correspond with mammographic findings and tumor characteristics (<xref ref-type="bibr" rid="B15">15</xref>). Jurisic et&#xa0;al. further discussed the clinical relevance of LDH as a tumor biomarker, summarizing its biochemical behavior and potential in oncological diagnostics (<xref ref-type="bibr" rid="B16">16</xref>). In addition to biochemical assessment, morphological analysis remains crucial. The study by Radenkovic et&#xa0;al. demonstrated that correlating mammographic images with histopathological findings in HER2-positive breast cancer provides deeper diagnostic insights, emphasizing the need for integrated diagnostic approaches (<xref ref-type="bibr" rid="B17">17</xref>).</p>
<p>While prior studies have demonstrated significant progress in applying deep learning to cancer diagnostics, several challenges remain unaddressed. Traditional symbolic systems often lack flexibility, machine learning approaches are highly feature-dependent, and deep learning models&#x2014;though powerful&#x2014;frequently suffer from a lack of interpretability, limiting their adoption in clinical workflows. To address these limitations, we propose a novel hybrid approach that leverages the interpretability of symbolic reasoning with the scalability of deep learning. Our method incorporates a modular AI architecture that integrates pathology-informed feature extraction with transformer-based visual encoders and an attention-guided prognosis predictor. By combining domain knowledge with data-driven inference, this system not only enhances accuracy but also enables interpretability through visual attention maps and feature attribution techniques. Our approach is designed to operate across different clinical settings and cancer subtypes, promoting generalizability and robustness. This hybrid methodology aims to bridge the gap between accuracy and trustworthiness in clinical AI applications, ultimately supporting oncologists in devising personalized treatment regimens and improving patient outcomes.</p>
<p>The main contributions of this work are as follows:</p>
<list list-type="bullet">
<list-item>
<p>We propose a novel dual-module framework that integrates symbolic feature extraction with deep visual embeddings, enabling interpretable and accurate prediction of breast cancer treatment response.</p></list-item>
<list-item>
<p>Our method supports multiple clinical scenarios and subtypes by employing a flexible architecture that generalizes across histopathology datasets with minimal performance degradation.</p></list-item>
<list-item>
<p>Experimental results on benchmark datasets demonstrate a significant improvement in prediction accuracy (up to 12% gain) over existing methods while maintaining interpretability through integrated attention maps.</p></list-item>
</list>
</sec>
<sec id="s2">
<label>2</label>
<title>Related work</title>
<sec id="s2_1">
<label>2.1</label>
<title>Deep learning for histopathology analysis</title>
<p>A central research direction in predicting breast cancer treatment response using AI involves deep learning techniques applied to histopathological images (<xref ref-type="bibr" rid="B18">18</xref>). Histopathology, particularly hematoxylin and eosin (H&amp;E)-stained slides, remains a gold standard in cancer diagnosis and is widely accessible. Convolutional neural networks (CNNs) have demonstrated notable performance in tasks such as tumor classification, segmentation, and grading (<xref ref-type="bibr" rid="B19">19</xref>). Pioneering works like that of Coudray et&#xa0;al. (<xref ref-type="bibr" rid="B20">20</xref>) on lung cancer laid the foundation for similar approaches in breast cancer (<xref ref-type="bibr" rid="B21">21</xref>). In this domain, deep learning models are trained on large annotated image datasets to recognize morphological features that correlate with treatment outcomes or overall prognosis. A significant body of literature has explored the application of CNNs to distinguish between different breast cancer subtypes, such as invasive ductal carcinoma versus lobular carcinoma, and to predict molecular markers HER2, ER, and PR status (<xref ref-type="bibr" rid="B22">22</xref>). Models such as ResNet and DenseNet have been adapted and fine-tuned to extract both low-level texture features and high-level morphological patterns. Moreover, multiple instance learning (MIL) frameworks have been employed to account for the weakly labeled nature of whole slide images, where only slide-level labels are available without pixel-level annotations (<xref ref-type="bibr" rid="B23">23</xref>). Another key development is the integration of patch-level analysis and whole-slide-level aggregation using attention mechanisms or transformer-based architectures. These models enable the network to focus on diagnostically relevant regions, thereby improving prediction accuracy and interpretability&#x2014;for example, attention-based MIL has been shown to provide heatmaps highlighting tumor-infiltrating lymphocytes or necrotic regions, both of which are relevant to prognosis and treatment response (<xref ref-type="bibr" rid="B24">24</xref>). Datasets such as CAMELYON16, TCGA, and BACH provide valuable benchmarks for model training and evaluation. However, the heterogeneity of breast cancer tissue and staining protocols across institutions remains a challenge (<xref ref-type="bibr" rid="B25">25</xref>). Domain adaptation and self-supervised learning have been proposed to mitigate the performance drop in cross-domain applications. The literature increasingly emphasizes the need for model robustness, generalizability, and clinical interpretability, including the use of saliency maps and feature attribution methods to explain predictions.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Radiomics and multimodal integration</title>
<p>Radiomics, which involves extracting quantitative features from medical imaging modalities like mammography, MRI, and ultrasound, represents another prominent research direction (<xref ref-type="bibr" rid="B26">26</xref>). AI-driven radiomics aims to uncover imaging biomarkers that predict therapeutic response or long-term outcomes. Unlike traditional image interpretation by radiologists, radiomics involves high-throughput feature extraction, including shape, texture, and intensity statistics, which are then correlated with clinical endpoints using machine learning models (<xref ref-type="bibr" rid="B27">27</xref>). Recent studies have shown that radiomic features from dynamic contrast-enhanced MRI (DCE-MRI) can predict neoadjuvant chemotherapy (NAC) response with significant accuracy&#x2014;for instance, early changes in tumor heterogeneity and vascularity have been linked to treatment sensitivity (<xref ref-type="bibr" rid="B28">28</xref>). Deep learning has further enhanced radiomics by replacing handcrafted feature engineering with learned representations from raw imaging data. Autoencoders and 3D CNNs have been utilized to capture spatial and temporal patterns in longitudinal imaging (<xref ref-type="bibr" rid="B29">29</xref>). The integration of radiomics with clinical, pathological, and genomic data represents a growing trend. Multimodal models leveraging tabular clinical data, histopathological images, and radiomics features have been proposed using fusion networks, often based on transformers or graph neural networks (GNNs) (<xref ref-type="bibr" rid="B30">30</xref>). These models aim to holistically characterize the tumor microenvironment and host response, leading to improved predictive performance over unimodal approaches (<xref ref-type="bibr" rid="B31">31</xref>). The challenges include the harmonization of imaging protocols across scanners and institutions, limited availability of annotated longitudinal datasets, and the interpretability of deep radiomics models (<xref ref-type="bibr" rid="B32">32</xref>). Federated learning has been suggested as a solution to the data privacy and sharing issues that hinder multi-institutional collaborations. Furthermore, explainability techniques are being actively developed to identify which imaging phenotypes contribute most to the predicted outcomes (<xref ref-type="bibr" rid="B33">33</xref>).</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>AI for personalized treatment planning</title>
<p>A critical area of research lies in the use of AI for personalizing breast cancer treatment by predicting individual responses to therapy. Traditional treatment planning relies heavily on standardized clinical guidelines, which may not capture the complex biological heterogeneity of breast cancer (<xref ref-type="bibr" rid="B34">34</xref>). AI systems offer a data-driven alternative, enabling precision oncology through personalized predictions based on image-derived biomarkers and patient-specific characteristics. Predictive models for treatment response focus on various therapeutic regimens, including chemotherapy, hormone therapy, and targeted therapies (<xref ref-type="bibr" rid="B35">35</xref>). By analyzing pre-treatment imaging and pathology data, AI can stratify patients into likely responders and non-responders (<xref ref-type="bibr" rid="B36">36</xref>). This allows clinicians to modify or escalate treatment strategies proactively, avoiding unnecessary toxicity and improving outcomes. Notable research efforts include the use of longitudinal imaging to model tumor evolution and response trajectories using recurrent neural networks or temporal convolutional networks (<xref ref-type="bibr" rid="B37">37</xref>). Moreover, prognosis prediction involves estimating survival outcomes such as disease-free survival (DFS) and overall survival (OS). AI models have been trained to predict these endpoints using features derived from imaging and pathology, often in conjunction with clinical staging and genetic information (<xref ref-type="bibr" rid="B38">38</xref>). Kaplan&#x2013;Meier analysis and Cox proportional hazards modeling are commonly used for evaluation, while AI models often optimize metrics such as concordance index or time-dependent AUC. Another promising direction involves reinforcement learning (RL) to dynamically recommend treatment strategies (<xref ref-type="bibr" rid="B39">39</xref>). RL agents can be trained on retrospective datasets to learn policies that maximize long-term patient outcomes under various treatment sequences. This paradigm shift from static prediction to dynamic decision-making is still in its early stages but holds significant potential (<xref ref-type="bibr" rid="B40">40</xref>). Current limitations include the scarcity of prospective validation studies, the black-box nature of many AI models, and regulatory challenges in clinical deployment. There is also a growing emphasis on incorporating patient preferences and quality-of-life metrics into AI-assisted treatment planning (<xref ref-type="bibr" rid="B41">41</xref>). Collaborative efforts among oncologists, data scientists, and regulatory bodies are essential to translate these advances into routine clinical practice.</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Method</title>
<sec id="s3_1">
<label>3.1</label>
<title>Overview</title>
<p>In this section, we introduce our proposed framework designed to model and predict treatment response across varying biomedical and clinical contexts. The capability to accurately forecast an individual&#x2019;s response to a therapeutic intervention is critical for enabling personalized medicine and optimizing treatment protocols. Our approach draws inspiration from recent advancements in sequence modeling, dynamics imitation, and representation learning, with specific tailoring to the domain of treatment outcome forecasting.</p>
<p>The &#x201c;Method&#x201d; section is organized into three key components, each addressing a specific methodological challenge. In Section 3.2, we formulate the problem of treatment response modeling as a structured prediction task within a dynamic system, where patient trajectories under treatment are viewed as stochastic processes. We provide rigorous mathematical formalization, including state space definitions, temporal dependency modeling, and symbolic abstractions of treatment-response interactions. This foundational formulation establishes a backbone for the learning problem and guides subsequent model design. In Section 3.3, we introduce our novel model, ResponseNet, which is a dynamics-aware, multi-level sequence learner tailored to capture both short-term physiological reactions and long-term outcome trends. ResponseNet incorporates heterogeneous data sources, including patient histories, treatment regimens, and clinical measurements, via a deep reparameterization approach. It is designed to imitate the progression of patient states post-treatment, drawing conceptual parallels with generative adversarial imitation learning frameworks adapted from natural video forecasting. The architectural design allows the model to retain interpretability while maintaining strong predictive power across varying temporal granularities. Section 3.4 details our adaptive knowledge infusion strategy, a principled mechanism for injecting domain knowledge into the learning process. This strategy leverages curated clinical priors, ontological constraints, and pharmacological knowledge to shape the learning trajectory of the model. Through an interaction-aware optimization scheme, the model dynamically adjusts its learning focus based on latent treatment&#x2013;response signals. This approach not only regularizes learning in data-sparse regimes but also encourages biologically plausible predictions that align with expert understanding.</p>
<p>To improve the interpretability of the proposed architecture for readers with clinical or non-technical backgrounds, a simplified and color-coded schematic is introduced, as shown in <xref ref-type="fig" rid="f1"><bold>Figure&#xa0;1</bold></xref>. This figure presents the end-to-end structure of the model in a modular layout, with functional components visually grouped and labeled. The architecture is divided into four high-level blocks: latent state inference (preliminaries), patient-specific prediction (ResponseNet), counterfactual reasoning, and adaptive knowledge infusion (AKI). Each block is represented using distinct colors to highlight its role and to reduce cognitive load when tracing data flow. The figure emphasizes key interactions between learned representations and domain knowledge modules&#x2014;for example, treatment actions are semantically embedded and passed to both predictive and counterfactual decoding modules. Latent health states are updated dynamically and passed into response prediction layers and symbolic constraints, while clinical priors guide the learning process through regularizers and ontology-based constraints. This design allows for a unified understanding of how data, treatments, and expert knowledge interact within the model. By presenting the architecture in this structured and clinically-oriented format, the figure enables practitioners to interpret the role of each component without relying on formal equations. The layout supports intuitive comprehension of model behavior, particularly how symbolic reasoning, learned dynamics, and decision-time explanations come together to support interpretable prediction. This visualization serves as a bridge between algorithmic detail and practical clinical insight, facilitating interdisciplinary understanding and communication.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Simplified architecture of the proposed framework. The model is organized into modular components: latent state inference, predictive and counterfactual decoding, semantic treatment embedding, and adaptive knowledge infusion (AKI). Color coding and directional flow highlight interactions between patient history, symbolic priors, and treatment-aware predictive modules.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1619994-g001.tif">
<alt-text content-type="machine-generated">Flowchart depicting the Adaptive Knowledge Infusion (AKI) process. It includes sections: Preliminaries, ResponseNet, and Appenrintes. Arrows indicate flow between components like Input Embedding, Predictive Decoding, Semantic Treatment Embedding, and Regularizer, leading to Latensic Anchoring. Processes like Ontology-Based Consistency Learning, and Discriminative Counterfactual Training are also part of the flow.</alt-text>
</graphic></fig>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Preliminaries</title>
<p>This work aims to model the latent treatment response trajectory of a patient undergoing therapeutic interventions, using longitudinal historical data including clinical features, physiological measurements, and treatment events. The response modeling task is framed as a partially observed Markov decision process (POMDP), which allows reasoning under uncertainty and incorporates the influence of sequential interventions over time. Let <inline-formula>
<mml:math display="inline" id="im1"><mml:mi mathvariant="script">P</mml:mi></mml:math></inline-formula> denote the patient population. For each patient <inline-formula>
<mml:math display="inline" id="im2"><mml:mrow><mml:mi>p</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:math></inline-formula>, the temporal sequence <inline-formula>
<mml:math display="inline" id="im3"><mml:mrow><mml:msub><mml:mi mathvariant="script">T</mml:mi><mml:mi>p</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>x</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>y</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>T</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> represents observations over time, where <inline-formula>
<mml:math display="inline" id="im4"><mml:mrow><mml:msubsup><mml:mi>x</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> are covariates, <inline-formula>
<mml:math display="inline" id="im5"><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> are treatments, and <inline-formula>
<mml:math display="inline" id="im6"><mml:mrow><mml:msubsup><mml:mi>y</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> are response outcomes. The true underlying health status is captured by a latent state <inline-formula>
<mml:math display="inline" id="im7"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>&#x2208;</mml:mo><mml:mi mathvariant="script">Z</mml:mi></mml:mrow></mml:math></inline-formula>, evolving stochastically through a transition kernel (<xref ref-type="disp-formula" rid="eq1">Equation 1</xref>):</p>
<disp-formula id="eq1"><label>(1)</label>
<mml:math display="block" id="M1"><mml:mrow><mml:mi>p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup><mml:mo>|</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi mathvariant="script">T</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>and generating observable variables via an emission model (<xref ref-type="disp-formula" rid="eq2">Equation 2</xref>):</p>
<disp-formula id="eq2"><label>(2)</label>
<mml:math display="block" id="M2"><mml:mrow><mml:mi>p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>y</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>|</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>&#x2130;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>The initial state is drawn from a prior distribution (<xref ref-type="disp-formula" rid="eq3">Equation 3</xref>):</p>
<disp-formula id="eq3"><label>(3)</label>
<mml:math display="block" id="M3"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mn>1</mml:mn><mml:mi>p</mml:mi></mml:msubsup><mml:mo>&#x223c;</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>z</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi mathvariant="script">N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>&#x3bc;</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mtext>&#x3a3;</mml:mtext><mml:mn>0</mml:mn></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>To handle partial observability, a recognition network <inline-formula>
<mml:math display="inline" id="im8"><mml:mrow><mml:msub><mml:mi>q</mml:mi><mml:mi>&#x3d5;</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>|</mml:mo><mml:msubsup><mml:mi>&#x210b;</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> is introduced to approximate the posterior over latent states from historical data <inline-formula>
<mml:math display="inline" id="im9"><mml:mrow><mml:msubsup><mml:mi>&#x210b;</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>x</mml:mi><mml:mi>s</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>y</mml:mi><mml:mi>s</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula>. The variational evidence lower bound (ELBO) is optimized jointly with respect to generative and inference parameters (<xref ref-type="disp-formula" rid="eq4">Equation 4</xref>):</p>
<disp-formula id="eq4"><label>(4)</label>
<mml:math display="block" id="M4"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:mi>&#x2112;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x3b8;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x3d5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="double-struck">E</mml:mi><mml:mrow><mml:msub><mml:mi>q</mml:mi><mml:mi>&#x3d5;</mml:mi></mml:msub></mml:mrow></mml:msub><mml:mo>[</mml:mo><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>T</mml:mi></mml:munderover><mml:mtext>log&#xa0;</mml:mtext><mml:mi>p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>y</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>|</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mtext>log&#xa0;</mml:mtext><mml:mi>p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup><mml:mo>|</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mtext>log&#xa0;</mml:mtext><mml:msub><mml:mi>q</mml:mi><mml:mi>&#x3d5;</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo stretchy="false">|</mml:mo><mml:msubsup><mml:mi>&#x210b;</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">]</mml:mo><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math>
</disp-formula>
<p>The full training objective aggregates patient trajectories and includes a regularization term (<xref ref-type="disp-formula" rid="eq5">Equation 5</xref>):</p>
<disp-formula id="eq5"><label>(5)</label>
<mml:math display="block" id="M5"><mml:mrow><mml:mi mathvariant="script">J</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x3b8;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x3d5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi mathvariant="script">P</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mi>&#x2112;</mml:mi><mml:mi>p</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x3b8;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x3d5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>&#x3bb;</mml:mi><mml:mo>&#xb7;</mml:mo><mml:mi>&#x211b;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>&#x3b8;</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>To accommodate censored or partially missing responses, a binary mask <inline-formula>
<mml:math display="inline" id="im10"><mml:mrow><mml:msubsup><mml:mi>m</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mo>{</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>}</mml:mo></mml:mrow><mml:mi>k</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula> is applied to the likelihood computation (<xref ref-type="disp-formula" rid="eq6">Equation 6</xref>):</p>
<disp-formula id="eq6"><label>(6)</label>
<mml:math display="block" id="M6"><mml:mrow><mml:mtext>log&#xa0;</mml:mtext><mml:mi>p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>y</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo stretchy="false">|</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>k</mml:mi></mml:munderover></mml:mstyle><mml:msubsup><mml:mi>m</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mi>p</mml:mi></mml:msubsup><mml:mo>&#xb7;</mml:mo><mml:mtext>log&#xa0;</mml:mtext><mml:mi mathvariant="script">N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>y</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mi>p</mml:mi></mml:msubsup><mml:mo>;</mml:mo><mml:msub><mml:mi>&#x3bc;</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msubsup><mml:mi>&#x3c3;</mml:mi><mml:mi>j</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>In addition to standard predictions, the framework enables counterfactual reasoning. A prediction operator is defined to estimate future outcomes under alternative, hypothetical treatments <inline-formula>
<mml:math display="inline" id="im11"><mml:mrow><mml:msub><mml:mover accent="true"><mml:mi>a</mml:mi><mml:mo>&#x2dc;</mml:mo></mml:mover><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> (<xref ref-type="disp-formula" rid="eq7">Equation 7</xref>):</p>
<disp-formula id="eq7"><label>(7)</label>
<mml:math display="block" id="M7"><mml:mrow><mml:msubsup><mml:mover accent="true"><mml:mi>y</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mo>,</mml:mo><mml:mtext>cf</mml:mtext></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="double-struck">E</mml:mi><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>&#x223c;</mml:mo><mml:msub><mml:mi>q</mml:mi><mml:mi>&#x3d5;</mml:mi></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mi mathvariant="double-struck">E</mml:mi><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup><mml:mo>&#x223c;</mml:mo><mml:mi mathvariant="script">T</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mover accent="true"><mml:mi>a</mml:mi><mml:mo>&#x2dc;</mml:mo></mml:mover><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mrow><mml:msub><mml:mi>&#x2130;</mml:mi><mml:mi>y</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">]</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>which supports &#x201c;what-if&#x201d; scenario simulation and assists in evaluating alternative therapy options.</p>
<p>This section builds a probabilistic foundation for understanding how a patient&#x2019;s health status evolves over time under different treatments. Rather than using raw features alone, the model constructs a hidden state that summarizes clinical information and allows prediction of future outcomes. By using a variational framework, it can handle uncertainty and missing values. The model also supports hypothetical simulations&#x2014;what would happen if a different treatment had been used&#x2014;making it useful for treatment planning and clinical decision support.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>ResponseNet</title>
<p>To operationalize the symbolic formulation and latent-state structure introduced in the previous section, we propose ResponseNet, a deep sequence modeling architecture designed to capture and forecast patient-specific treatment response through temporally-grounded latent dynamics. ResponseNet encodes nonlinear dependencies between health status trajectories and administered interventions while enabling interpretable abstractions aligned with clinical variables (as shown in <xref ref-type="fig" rid="f2"><bold>Figure&#xa0;2</bold></xref>).</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>An illustration of ResponseNet. The architecture of ResponseNet comprises a multi-module framework designed for treatment-aware clinical modeling, including latent dynamics modeling, semantic treatment embedding, and predictive as well as counterfactual decoding. The pipeline begins with input embedding, followed by latent state inference through gated recurrent units, a dedicated intervention module with semantic permutation and decoding, and a global local-attention encoder. Separate decoders generate both observed and counterfactual outcomes, allowing the model to simulate personalized treatment responses under varying hypothetical scenarios. Calibration attention mechanisms and alignment regularizations ensure robustness and interpretability in clinical prediction tasks.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1619994-g002.tif">
<alt-text content-type="machine-generated">Flowchart depicting a machine learning model architecture. It includes three main sections: input embedding, intervention module, and an auxiliary embedding process. Arrows indicate data flow. Components such as predictive and counterfactual decoding, latent dynamics modeling, and semantic treatment embedding are highlighted. Steps include calibration attention, random shuffle, and a cropping layer. A circular arrow shows iteration in the auxiliary section. A small image of a cell is above the input embedding.</alt-text>
</graphic></fig>
<sec id="s3_3_1">
<label>3.3.1</label>
<title>Latent dynamics modeling</title>
<p>At its core, ResponseNet leverages a probabilistic latent state framework to model the evolution of patient-specific health trajectories in response to administered treatments over time. The system is designed to infer compact representations that capture both short-term variability and long-range dependencies in clinical dynamics, with the latent space serving as a hidden abstraction layer that unifies heterogeneous covariates and outcome signals. Each patient&#x2019;s longitudinal record up to time <italic>t</italic> is denoted as <inline-formula>
<mml:math display="inline" id="im12"><mml:mrow><mml:msubsup><mml:mi>&#x210b;</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>x</mml:mi><mml:mi>s</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>y</mml:mi><mml:mi>s</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula>, encompassing observed covariates <inline-formula>
<mml:math display="inline" id="im13"><mml:mrow><mml:msubsup><mml:mi>x</mml:mi><mml:mi>s</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula>, intervention actions <inline-formula>
<mml:math display="inline" id="im14"><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula>, and clinical outcomes <inline-formula>
<mml:math display="inline" id="im15"><mml:mrow><mml:msubsup><mml:mi>y</mml:mi><mml:mi>s</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula>. We posit a temporally evolving latent state <inline-formula>
<mml:math display="inline" id="im16"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> that encodes the internal physiological status, updated through a history-aware encoder formulated as a deep recurrent posterior distribution. The encoder employs gated recurrence to model complex temporal dependencies and amortize inference across varying-length patient histories, parameterizing a multivariate Gaussian distribution over the latent variables as (<xref ref-type="disp-formula" rid="eq8">Equation 8</xref>).</p>
<disp-formula id="eq8"><label>(8)</label>
<mml:math display="block" id="M8"><mml:mrow><mml:msub><mml:mi>q</mml:mi><mml:mi>&#x3d5;</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>|</mml:mo><mml:msubsup><mml:mi>&#x210b;</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi mathvariant="script">N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>&#x3bc;</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mtext>&#x3a3;</mml:mtext><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x2003;</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>&#x3bc;</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mtext>&#x3a3;</mml:mtext><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mtext>GRU</mml:mtext></mml:mrow><mml:mi>&#x3d5;</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>&#x210b;</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im17"><mml:mi>&#x3d5;</mml:mi></mml:math></inline-formula> represents the learnable weights of the inference network. To characterize how clinical states evolve under the influence of treatment, we define a continuous latent transition function <inline-formula>
<mml:math display="inline" id="im18"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>&#x3b8;</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> that maps the current latent state <inline-formula>
<mml:math display="inline" id="im19"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> and an embedded treatment action <inline-formula>
<mml:math display="inline" id="im20"><mml:mrow><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> to a predictive shift in latent dynamics, capturing the modulating effects of pharmacological interventions and potential interactions between treatment and baseline state. This function is implemented as a multilayer perceptron whose output is perturbed by Gaussian noise to reflect uncertainty in clinical progression, yielding the one-step latent update as (<xref ref-type="disp-formula" rid="eq9">Equation 9</xref>).</p>
<disp-formula id="eq9"><label>(9)</label>
<mml:math display="block" id="M9"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>&#x3b8;</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x3f5;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x2003;</mml:mtext><mml:msub><mml:mi>&#x3f5;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x223c;</mml:mo><mml:mi mathvariant="script">N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:msup><mml:mi>&#x3c3;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mi>I</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>where <italic>&#x3b8;</italic> denotes the generative parameters of the dynamics model and <italic>&#x3c3;</italic> modulates diffusion in the latent space. However, to better account for latent inertia and delayed effects of therapy, we augment this formulation by introducing a second-order difference operator into the transition rule. The model maintains coherence across adjacent latent states by integrating change-of-change signals, allowing the representation to encode temporal acceleration or deceleration in response to treatment shifts. The refined latent transition equation is expressed as (<xref ref-type="disp-formula" rid="eq10">Equation 10</xref>).</p>
<disp-formula id="eq10"><label>(10)</label>
<mml:math display="block" id="M10"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>+</mml:mo><mml:mi>&#x3b3;</mml:mi><mml:mo>&#xb7;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>&#x3b8;</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>&#x3b8;</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>,</mml:mo></mml:math>
</disp-formula>
<p>where <italic>&#x3b3;</italic>  is a learnable scalar controlling the strength of coupling across temporal windows. The embedding function <inline-formula>
<mml:math display="inline" id="im21"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula> is jointly learned to reflect both pharmacological identity and dosage, and is trained end-to-end with the rest of the model. To ensure that the latent state remains clinically meaningful and temporally smooth, we introduce a pathwise regularizer that penalizes abrupt changes in latent evolution, stabilizing trajectory estimation and improving generalization in data-sparse regimes. This constraint is defined over the Euclidean distance of successive latent states as (<xref ref-type="disp-formula" rid="eq11">Equation 11</xref>).</p>
<disp-formula id="eq11"><label>(11)</label>
<mml:math display="block" id="M11"><mml:mrow><mml:msub><mml:mi>&#x211b;</mml:mi><mml:mrow><mml:mtext>temp</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn></mml:mrow><mml:mi>T</mml:mi></mml:munderover><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>&#x2212;</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:msubsup><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>which effectively enforces a soft continuity constraint on the temporal latent manifold. This dynamic modeling framework empowers the architecture to flexibly represent diverse disease trajectories and adaptively adjust to the evolving effects of treatments across time and patients.</p>
</sec>
<sec id="s3_3_2">
<label>3.3.2</label>
<title>Semantic treatment embedding</title>
<p>To capture the pharmacological semantics and structural relations among treatments, we introduce a symbolic embedding mechanism that disentangles class-level and treatment-specific properties through a compositional representation strategy. Each administered treatment <inline-formula>
<mml:math display="inline" id="im22"><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> is mapped to a dense vector through an embedding function <inline-formula>
<mml:math display="inline" id="im23"><mml:mrow><mml:mtext>&#x3a8;</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, which integrates hierarchical ontology-informed semantics with fine-grained pharmacological deviations. Let <inline-formula>
<mml:math display="inline" id="im24"><mml:mrow><mml:mi>&#x3b1;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> denote the symbolic class or therapeutic category of treatment <inline-formula>
<mml:math display="inline" id="im25"><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula>, such as hormone therapy, chemotherapy, or targeted inhibitors. We define the embedding as the sum of a class-shared vector <inline-formula>
<mml:math display="inline" id="im26"><mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mtext>sym</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x3b1;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> and a specific offset vector <inline-formula>
<mml:math display="inline" id="im27"><mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mtext>spec</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> that encodes individual deviations from the class prototype, resulting in (<xref ref-type="disp-formula" rid="eq12">Equation 12</xref>).</p>
<disp-formula id="eq12"><label>(12)</label>
<mml:math display="block" id="M12"><mml:mrow><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mtext>&#x3a8;</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mtext>sym</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x3b1;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mtext>spec</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im28"><mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mtext>sym</mml:mtext></mml:mrow></mml:msub><mml:mo>:</mml:mo><mml:mi mathvariant="script">V</mml:mi><mml:mo>&#x2192;</mml:mo><mml:msup><mml:mi>&#x211d;</mml:mi><mml:mi>m</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula> and <inline-formula>
<mml:math display="inline" id="im29"><mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mtext>spec</mml:mtext></mml:mrow></mml:msub><mml:mo>:</mml:mo><mml:mi mathvariant="script">A</mml:mi><mml:mo>&#x2192;</mml:mo><mml:msup><mml:mi>&#x211d;</mml:mi><mml:mi>m</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula> are learned jointly. This formulation enables parameter sharing across pharmacologically related interventions, facilitating generalization in low-resource settings while retaining the ability to model treatment-specific behavior. To reinforce semantic smoothness and coherence across related treatments, we impose a class-aware regularization objective that penalizes excessive divergence between embeddings of treatments belonging to the same category. Let <inline-formula>
<mml:math display="inline" id="im30"><mml:mi mathvariant="script">C</mml:mi></mml:math></inline-formula> be the set of all intra-class treatment pairs, and <inline-formula>
<mml:math display="inline" id="im31"><mml:mi>&#x3b4;</mml:mi></mml:math></inline-formula> a positive scalar margin defining acceptable divergence within a class. The symbolic regularizer takes the form (<xref ref-type="disp-formula" rid="eq13">Equation 13</xref>).</p>
<disp-formula id="eq13"><label>(13)</label>
<mml:math display="block" id="M13"><mml:mrow><mml:msub><mml:mi>&#x211b;</mml:mi><mml:mrow><mml:mtext>sym</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:mi mathvariant="script">C</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mtext>max&#xa0;</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:msubsup><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x2212;</mml:mo><mml:mi>&#x3b4;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:math>
</disp-formula>
<p>which effectively acts as a margin-based metric learning constraint in the embedding space. Furthermore, to introduce relational inductive bias based on treatment ontologies and pharmacodynamics, we define a symbolic affinity kernel <inline-formula>
<mml:math display="inline" id="im32"><mml:mi mathvariant="script">K</mml:mi></mml:math></inline-formula>(<italic>a<sub>i</sub>,a<sub>j</sub></italic>) that measures knowledge-driven similarity between treatments <italic>a<sub>i</sub></italic>and <italic>a<sub>j</sub></italic>. This kernel is derived from co-membership in anatomical therapeutic chemical (ATC) codes, empirical co-prescription statistics, or expert-defined similarity graphs. We incorporate this structure into the embedding training via an additional alignment constraint that minimizes the discrepancy between geometric distances in embedding space and knowledge-based similarities. Letting &#x2225;<italic>e</italic>(<italic>a<sub>i</sub></italic>)&#x2212;<italic>e</italic>(<italic>a<sub>j</sub></italic>)&#x2225;<sub>2</sub> denote Euclidean distance in the learned space, we regularize towards monotonic alignment with <inline-formula>
<mml:math display="inline" id="im33"><mml:mi mathvariant="script">K</mml:mi></mml:math></inline-formula>(<italic>a<sub>i</sub>,a<sub>j</sub></italic>) as (<xref ref-type="disp-formula" rid="eq14">Equation 14</xref>).</p>
<disp-formula id="eq14"><label>(14)</label>
<mml:math display="block" id="M14"><mml:mrow><mml:msub><mml:mi>&#x211b;</mml:mi><mml:mrow><mml:mtext>align</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mstyle displaystyle="true"><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:munder><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:msubsup><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi mathvariant="script">K</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>where larger values of <inline-formula>
<mml:math display="inline" id="im34"><mml:mi mathvariant="script">K</mml:mi></mml:math></inline-formula>(<italic>a<sub>i</sub>,a<sub>j</sub></italic>) indicate stronger pharmacological similarity. This constraint encourages embedding geometry to reflect domain knowledge and induces latent semantic clusters consistent with pharmacological theory. To further integrate symbolic structure into the temporal modeling process, we modulate internal attention weights over treatment classes via similarity-weighted aggregation. Let <inline-formula>
<mml:math display="inline" id="im35"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> be the latent state at time <inline-formula>
<mml:math display="inline" id="im36"><mml:mi>t</mml:mi></mml:math></inline-formula>, and define the relevance score between <inline-formula>
<mml:math display="inline" id="im37"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula>and class embedding <inline-formula>
<mml:math display="inline" id="im38"><mml:mrow><mml:msub><mml:mi>e</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> for each class <inline-formula>
<mml:math display="inline" id="im39"><mml:mi>c</mml:mi></mml:math></inline-formula> as an inner product followed by softmax normalization, producing a class-discriminative attention distribution (<xref ref-type="disp-formula" rid="eq15">Equation 15</xref>).</p>
<disp-formula id="eq15"><label>(15)</label>
<mml:math display="block" id="M15"><mml:mrow><mml:msubsup><mml:mi>&#x3b1;</mml:mi><mml:mi>t</mml:mi><mml:mi>c</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mtext>exp&#xa0;</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mo>&#x2329;</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mi>e</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:mrow><mml:mo>&#x232a;</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mo>&#x2211;</mml:mo><mml:msup><mml:mi>c</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup></mml:msub><mml:mtext>exp&#xa0;</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mo>&#x2329;</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mi>e</mml:mi><mml:msup><mml:mi>c</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup></mml:msub></mml:mrow><mml:mo>&#x232a;</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>where <italic>e<sub>c</sub></italic>= <italic>E</italic><sub>sym</sub>(<italic>c</italic>) is the class-level prototype embedding. These attention scores are used to adaptively gate treatment effects according to temporal context and semantic proximity, allowing the model to selectively prioritize therapeutically relevant actions across dynamic states. By embedding treatment actions into a knowledge-aware latent space and aligning learning dynamics with symbolic ontologies, the model improves both interpretability and generalizability, while maintaining sensitivity to fine-grained pharmacological distinctions necessary for personalized therapeutic reasoning.</p>
</sec>
<sec id="s3_3_3">
<label>3.3.3</label>
<title>Predictive and counterfactual decoding</title>
<p>The latent state <inline-formula>
<mml:math display="inline" id="im40"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> serves as a compact representation of the patient&#x2019;s clinical condition at time <italic>t</italic>, integrating historical covariates, treatments, and inferred disease progression (as shown in <xref ref-type="fig" rid="f3"><bold>Figure&#xa0;3</bold></xref>).</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Illustration of the predictive and counterfactual decoding framework. The diagram demonstrates the decoding process in which patient state representations are transformed into clinical outcome predictions and auxiliary variable reconstructions. Feature flow begins with image-derived inputs, which are linearly projected and pooled to form agent tokens. These tokens pass through the predictive and counterfactual decoding module, enabling response generation. A cross-attention mechanism integrates agent features with contextual bias to inform future predictions. This framework supports not only the accurate estimation of clinical outcomes, such as tumor metrics and lab variables, but also facilitates counterfactual simulation by conditioning the decoder on alternate treatment embeddings. Temporal regularization is incorporated to ensure consistency in decoded trajectories, aiding robust and interpretable clinical decision modeling.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1619994-g003.tif">
<alt-text content-type="machine-generated">Diagram illustrating a neural network model for image analysis. It includes an input image of breast tissue with a highlighted area. The process flow is marked by arrows, showing &#x201c;Feature Flow&#x201d; and &#x201c;Agent Flow&#x201d;. Components include matrices labeled \(W_Q\), \(W_K\), \(W_V\), &#x201c;Predictive and Counterfactual Decoding&#x201d;, &#x201c;Softmax Attention&#x201d;, and &#x201c;Agent Bias&#x201d;. The model transforms input into various features and agent tokens, generating an output.</alt-text>
</graphic></fig>
<p>To reconstruct observed variables from this latent representation, we employ dedicated decoder networks for both response outcomes and auxiliary covariates. The decoder for clinical outcomes maps <inline-formula>
<mml:math display="inline" id="im41"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> to a predicted response <inline-formula>
<mml:math display="inline" id="im42"><mml:mrow><mml:msubsup><mml:mover accent="true"><mml:mi>y</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> using a feedforward neural transformation, where nonlinear activation ensures expressivity in modeling complex effects, and the output is parameterized as a Gaussian mean for continuous-valued medical indicators such as tumor size, biomarker levels, or composite clinical scores. Simultaneously, auxiliary covariates <inline-formula>
<mml:math display="inline" id="im43"><mml:mrow><mml:msubsup><mml:mover accent="true"><mml:mi>x</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> such as lab values or patient status are decoded to support downstream reconstruction objectives and regularization of the latent structure. The decoding equations are defined as follows (<xref ref-type="disp-formula" rid="eq16">Equation 16</xref>):</p>
<disp-formula id="eq16"><label>(16)</label>
<mml:math display="block" id="M16"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mover accent="true"><mml:mi>y</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="script">D</mml:mi><mml:mi>y</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mi>W</mml:mi><mml:mi>y</mml:mi></mml:msub><mml:mo>&#xb7;</mml:mo><mml:mtext>ReLU</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mi>y</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x2003;&#x2003;</mml:mtext><mml:msubsup><mml:mover accent="true"><mml:mi>x</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="script">D</mml:mi><mml:mi>x</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x2009;</mml:mtext><mml:mo>=</mml:mo><mml:msub><mml:mi>W</mml:mi><mml:mi>x</mml:mi></mml:msub><mml:mo>&#xb7;</mml:mo><mml:mtext>tanh</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mi>x</mml:mi></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math>
</disp-formula>
<p>where <italic>W<sub>y</sub>,W<sub>x </sub></italic>are weight matrices and <italic>b<sub>y</sub>, b<sub>x</sub></italic> are biases for their respective decoders. In realistic clinical scenarios, outcome observations are often noisy or uncertain due to measurement variability or delayed manifestations. To model this uncertainty explicitly, we parameterize the conditional distribution of clinical responses as a heteroscedastic Gaussian whose mean and variance are both decoded from <inline-formula>
<mml:math display="inline" id="im44"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula>. Letting <inline-formula>
<mml:math display="inline" id="im45"><mml:mrow><mml:msub><mml:mi>&#x3bc;</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> and <inline-formula>
<mml:math display="inline" id="im46"><mml:mrow><mml:msubsup><mml:mi>&#x3c3;</mml:mi><mml:mi>j</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> denote the decoder outputs for the <inline-formula>
<mml:math display="inline" id="im47"><mml:mi>j</mml:mi></mml:math></inline-formula>-th outcome dimension, the predictive likelihood is given by (<xref ref-type="disp-formula" rid="eq17">Equation 17</xref>).</p>
<disp-formula id="eq17"><label>(17)</label>
<mml:math display="block" id="M17"><mml:mrow><mml:mi>p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>y</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo stretchy="false">|</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover><mml:mo>&#x220f;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>k</mml:mi></mml:munderover></mml:mstyle><mml:mi mathvariant="script">N</mml:mi><mml:mo>(</mml:mo><mml:msubsup><mml:mi>y</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>&#xa0;</mml:mo></mml:mrow><mml:mi>p</mml:mi></mml:msubsup><mml:mo>|</mml:mo><mml:mo>&#xa0;</mml:mo><mml:msub><mml:mi>&#x3bc;</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msubsup><mml:mi>&#x3c3;</mml:mi><mml:mi>j</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im48"><mml:mi>k</mml:mi></mml:math></inline-formula> denotes the number of predicted clinical targets. Beyond reconstruction and forward prediction, a critical function of the model is its ability to simulate hypothetical outcomes under alternative treatments, enabling counterfactual reasoning for decision support. Given a hypothetical intervention <inline-formula>
<mml:math display="inline" id="im49"><mml:mrow><mml:msubsup><mml:mover accent="true"><mml:mi>a</mml:mi><mml:mo>&#x2dc;</mml:mo></mml:mover><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>&#x2208;</mml:mo><mml:mi mathvariant="script">A</mml:mi></mml:mrow></mml:math></inline-formula> distinct from the one actually administered, the model estimates the prospective response had this treatment been chosen instead. This is operationalized by feeding the current latent state <inline-formula>
<mml:math display="inline" id="im50"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> through the dynamics model <inline-formula>
<mml:math display="inline" id="im51"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>&#x3b8;</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> in conjunction with the symbolic embedding <inline-formula>
<mml:math display="inline" id="im52"><mml:mrow><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mover accent="true"><mml:mi>a</mml:mi><mml:mo>&#x2dc;</mml:mo></mml:mover><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> of the counterfactual treatment. The resulting shifted latent is then decoded using the same outcome decoder <inline-formula>
<mml:math display="inline" id="im53"><mml:mrow><mml:msub><mml:mi mathvariant="script">D</mml:mi><mml:mi>y</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, producing a synthetic estimate of the next clinical response (<xref ref-type="disp-formula" rid="eq18">Equation 18</xref>):</p>
<disp-formula id="eq18"><label>(18)</label>
<mml:math display="block" id="M18"><mml:mrow><mml:msubsup><mml:mover accent="true"><mml:mi>y</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mtext>cf</mml:mtext></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="script">D</mml:mi><mml:mi>y</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>&#x3b8;</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mover accent="true"><mml:mi>a</mml:mi><mml:mo>&#x2dc;</mml:mo></mml:mover><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>which enables flexible generation of alternative trajectories across the treatment space. To evaluate the model&#x2019;s internal consistency and regularize unrealistic fluctuations in predicted outcomes, we further introduce a temporal smoothness regularizer that penalizes excessive changes in decoded covariates over time. This promotes physiological plausibility and ensures the learned latent dynamics induce stable transitions in observed space. Letting <inline-formula>
<mml:math display="inline" id="im54"><mml:mrow><mml:msubsup><mml:mover accent="true"><mml:mi>x</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> and <inline-formula>
<mml:math display="inline" id="im55"><mml:mrow><mml:msubsup><mml:mover accent="true"><mml:mi>x</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> denote the reconstructed covariates at adjacent&#xa0;time steps, we define the temporal regularization loss as (<xref ref-type="disp-formula" rid="eq19">Equation 19</xref>).</p>
<disp-formula id="eq19"><label>(19)</label>
<mml:math display="block" id="M19"><mml:mrow><mml:msub><mml:mi>&#x211b;</mml:mi><mml:mrow><mml:mtext>smooth</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn></mml:mrow><mml:mi>T</mml:mi></mml:munderover><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:msubsup><mml:mover accent="true"><mml:mi>x</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>&#x2212;</mml:mo><mml:msubsup><mml:mover accent="true"><mml:mi>x</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:msubsup><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>which can be integrated into the global training objective. This predictive and counterfactual decoding framework enables not only&#xa0;accurate estimation of future responses but also generates plausible &#x201c;what-if&#x201d; scenarios for interventions never observed during training, supporting clinical interpretability and robust policy simulation.</p>
<p>ResponseNet is a modular neural network designed to predict how&#xa0;patients will respond to cancer treatment over time. It works by compressing patient history&#x2014;such as lab values, tumor measurements, and treatments&#x2014;into a hidden &#x201c;health state&#x201d; that updates after each new treatment. This health state helps forecast future outcomes like tumor size or biomarker levels. To make the predictions understandable, the system uses attention mechanisms to highlight which features or treatment types were most influential, and it supports &#x201c;what-if&#x201d; simulations for alternative treatments. The symbolic treatment embedding module connects treatments to known medical classes, improving generalization and interpretability. These design choices together enable both high predictive accuracy and practical usability for clinical research and decision-making.</p>
</sec>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Adaptive knowledge infusion</title>
<p>In this section, we introduce adaptive knowledge infusion (AKI), a novel learning strategy designed to enhance the clinical fidelity, stability, and generalizability of ResponseNet. While the model presented previously can capture latent dynamics and decode treatment responses effectively, the integration of structured medical knowledge remains a critical aspect for clinical plausibility. AKI injects hierarchical, domain-driven inductive biases into the training process via structured regularization, latent alignment, and counterfactual discrimination (as shown in <xref ref-type="fig" rid="f4"><bold>Figure&#xa0;4</bold></xref>).</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Illustration of adaptive knowledge infusion (AKI). The figure outlines the architectural design of AKI, highlighting its three core mechanisms: ontology-based consistency learning, latent space anchoring, and discriminative counterfactual training. The upper pipeline illustrates a multi-stage encoder integrating patch embedding and conceptually structured consistency across resolution levels. The bottom path embeds regularization modules including norm layers, counterfactual training units, and anchoring blocks that align latent representations with medical ontologies and domain priors. These modules together enforce structured semantics, enhance interpretability, and improve generalization in clinical prognostic modeling.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1619994-g004.tif">
<alt-text content-type="machine-generated">Diagram of a neural network structure featuring four layers. Each layer consists of &#x201c;Patch Embedding&#x201d; followed by &#x201c;Ontology-Based Consistency Learning.&#x201d; Data scales from \(C_1\) to \(C_4\) as dimensions reduce through layers. At the bottom, a separate flow shows processes including &#x201c;Latent Space Anchoring,&#x201d; &#x201c;Norm,&#x201d; and &#x201c;Discriminative Counterfactual Training,&#x201d; integrated with the main structure.</alt-text>
</graphic></fig>
<sec id="s3_4_1">
<label>3.4.1</label>
<title>Ontology-based consistency learning</title>
<p>In clinical prognostic modeling, particularly in domains involving high-stakes interventions such as breast cancer treatment, data-driven models often face limitations due to incomplete supervision, delayed outcomes, and inconsistent labeling. Treatment decisions are typically informed by domain knowledge codified in clinical guidelines, pharmacological taxonomies, and expert intuition, yet most sequence models remain agnostic to these structured priors. To address this discrepancy, we integrate symbolic knowledge into model training via ontology-based regularization, grounding latent treatment dynamics in known therapeutic semantics. Let <inline-formula>
<mml:math display="inline" id="im56"><mml:mrow><mml:mi mathvariant="script">G</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="script">V</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x2130;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> denote a treatment ontology, where <inline-formula>
<mml:math display="inline" id="im57"><mml:mi mathvariant="script">V</mml:mi></mml:math></inline-formula> is a finite set of treatment classes and <inline-formula>
<mml:math display="inline" id="im58"><mml:mi>&#x2130;</mml:mi></mml:math></inline-formula> represents semantic relations such as subclass-of, similarity, or therapeutic proximity. Each administered treatment <inline-formula>
<mml:math display="inline" id="im59"><mml:mrow><mml:mi>a</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi mathvariant="script">A</mml:mi></mml:mrow></mml:math></inline-formula> is mapped to a class label <inline-formula>
<mml:math display="inline" id="im60"><mml:mrow><mml:mi>&#x3b1;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>a</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:mi mathvariant="script">V</mml:mi></mml:mrow></mml:math></inline-formula>, and relationships among these classes induce constraints on their latent effects. For any two treatments <inline-formula>
<mml:math display="inline" id="im61"><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula>
<mml:math display="inline" id="im62"><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> linked by a similarity edge <inline-formula>
<mml:math display="inline" id="im63"><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mi>&#x2130;</mml:mi><mml:mrow><mml:mtext>sim</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x2286;</mml:mo><mml:mi>&#x2130;</mml:mi></mml:mrow></mml:math></inline-formula>, we enforce consistency between their induced shifts in latent state via a variance-penalized deviation term. Letting <inline-formula>
<mml:math display="inline" id="im64"><mml:mi>z</mml:mi></mml:math></inline-formula> denote the pre-treatment latent state and <inline-formula>
<mml:math display="inline" id="im65"><mml:mrow><mml:mtext>&#x394;</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>z</mml:mi><mml:mo>,</mml:mo><mml:mi>a</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>&#x3b8;</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>z</mml:mi><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>a</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula> the treatment-induced transformation, the semantic consistency loss is expressed as (<xref ref-type="disp-formula" rid="eq20">Equation 20</xref>).</p>
<disp-formula id="eq20"><label>(20)</label>
<mml:math display="block" id="M20"><mml:mrow><mml:msub><mml:mi>&#x2112;</mml:mi><mml:mrow><mml:mtext>consist</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mi>&#x2130;</mml:mi><mml:mrow><mml:mtext>sim</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:munder><mml:mrow><mml:msub><mml:mi mathvariant="double-struck">E</mml:mi><mml:mi>z</mml:mi></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mtext>&#x394;</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>z</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mtext>&#x394;</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>z</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:msubsup><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mstyle><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>which regularizes the model to yield functionally similar predictions for pharmacologically similar drugs. To extend this structure beyond isolated treatment instances and account for longitudinal impact, we define a cumulative therapeutic influence over a trajectory. Let <inline-formula>
<mml:math display="inline" id="im66"><mml:mrow><mml:msubsup><mml:mrow><mml:mo>{</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>}</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>T</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> be the sequence of administered treatments and <inline-formula>
<mml:math display="inline" id="im67"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> the latent state prior to each administration. We compute the aggregated therapeutic deviation as a weighted sum of instantaneous shifts, modulated by decay weights <inline-formula>
<mml:math display="inline" id="im68"><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> that reflect diminishing influence over time (<xref ref-type="disp-formula" rid="eq21">Equation 21</xref>):</p>
<disp-formula id="eq21"><label>(21)</label>
<mml:math display="block" id="M21"><mml:mrow><mml:msubsup><mml:mtext>&#x393;</mml:mtext><mml:mi>T</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>T</mml:mi></mml:munderover><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>&#xb7;</mml:mo><mml:mtext>&#x394;</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im69"><mml:mrow><mml:msubsup><mml:mtext>&#x393;</mml:mtext><mml:mi>T</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> encodes the net pharmacodynamic effect accumulated by time <inline-formula>
<mml:math display="inline" id="im70"><mml:mi>T</mml:mi></mml:math></inline-formula>. Clinical safety and plausibility constraints, derived from empirical studies or physiological theory, often define a feasible region <inline-formula>
<mml:math display="inline" id="im71"><mml:mrow><mml:msub><mml:mi mathvariant="script">C</mml:mi><mml:mrow><mml:mtext>safe</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x2282;</mml:mo><mml:msup><mml:mi>&#x211d;</mml:mi><mml:mi>d</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula> within which accumulated effects are considered benign or therapeutically sound. To ensure that <inline-formula>
<mml:math display="inline" id="im72"><mml:mrow><mml:msubsup><mml:mtext>&#x393;</mml:mtext><mml:mi>T</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> lies within this corridor, we introduce a projection-based regularizer that penalizes deviation from this trusted region. Let <inline-formula>
<mml:math display="inline" id="im73"><mml:mrow><mml:msub><mml:mrow><mml:mtext>Proj</mml:mtext></mml:mrow><mml:mrow><mml:msub><mml:mi mathvariant="script">C</mml:mi><mml:mrow><mml:mtext>safe</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mtext>&#x393;</mml:mtext><mml:mi>T</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> denote the closest point in <inline-formula>
<mml:math display="inline" id="im74"><mml:mrow><mml:msub><mml:mi mathvariant="script">C</mml:mi><mml:mrow><mml:mtext>safe</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> to <inline-formula>
<mml:math display="inline" id="im75"><mml:mrow><mml:msubsup><mml:mtext>&#x393;</mml:mtext><mml:mi>T</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> under the Euclidean norm. The safety-aware regularization is formulated as (<xref ref-type="disp-formula" rid="eq22">Equation 22</xref>)</p>
<disp-formula id="eq22"><label>(22)</label>
<mml:math display="block" id="M22"><mml:mrow><mml:msub><mml:mi>&#x211b;</mml:mi><mml:mrow><mml:mtext>corridor</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mi>p</mml:mi></mml:munder></mml:mstyle><mml:mi mathvariant="double-struck">E</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi mathvariant="double-struck">I</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mtext>&#x393;</mml:mtext><mml:mi>T</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>&#x2209;</mml:mo><mml:msub><mml:mi mathvariant="script">C</mml:mi><mml:mrow><mml:mtext>safe</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#xb7;</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:msubsup><mml:mtext>&#x393;</mml:mtext><mml:mi>T</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mtext>Proj</mml:mtext></mml:mrow><mml:mrow><mml:msub><mml:mi mathvariant="script">C</mml:mi><mml:mrow><mml:mtext>safe</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mtext>&#x393;</mml:mtext><mml:mi>T</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:msubsup><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>which softly penalizes infeasible treatment progressions and steers latent trajectory evolution toward physiologically consistent patterns. In practice, the region <inline-formula>
<mml:math display="inline" id="im76"><mml:mrow><mml:msub><mml:mi mathvariant="script">C</mml:mi><mml:mrow><mml:mtext>safe</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> can be specified by convex hulls derived from real-world patient clusters, dose&#x2013;response curves from pharmacokinetic studies, or clinical endpoints observed under expert-recommended regimens. To further encourage latent dynamics to respect ontology-implied continuity, we also include a directional consistency term between sequential treatment applications, enforcing smooth transitions in latent influence vectors. Denoting two successive treatments as <inline-formula>
<mml:math display="inline" id="im77"><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula>
<mml:math display="inline" id="im78"><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, we define a differential alignment loss (<xref ref-type="disp-formula" rid="eq23">Equation 23</xref>).</p>
<disp-formula id="eq23"><label>(23)</label>
<mml:math display="block" id="M23"><mml:mrow><mml:msub><mml:mi>&#x211b;</mml:mi><mml:mrow><mml:mtext>drift</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mi>t</mml:mi></mml:munder><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mtext>&#x394;</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mtext>&#x394;</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:msubsup><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>which penalizes abrupt changes in latent directionality across time and improves trajectory stability under ontology-guided constraints. These joint mechanisms allow the model to not only learn from observed outcomes but also reason over structured symbolic relationships that govern permissible treatment behaviors, enabling more faithful generalization in complex and sparsely labeled clinical environments.</p>
</sec>
<sec id="s3_4_2">
<label>3.4.2</label>
<title>Latent space anchoring</title>
<p>To enhance the physiological interpretability and clinical plausibility of latent representations, we introduce a principled anchoring mechanism that aligns the posterior distribution over latent variables with prior distributions derived from medical knowledge. We define a prior <inline-formula>
<mml:math display="inline" id="im79"><mml:mrow><mml:mi>&#x3c0;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>z</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> over latent states <inline-formula>
<mml:math display="inline" id="im80"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> that reflects domain-informed expectations regarding disease stage progression, biomarker distributions, or population-level clustering. These priors can be constructed using empirical distributions from historical cohorts, Gaussian mixtures conditioned on clinical stages, or prototype embeddings derived from stratified patient groups. During training, we minimize the Kullback&#x2013;Leibler divergence between the learned variational posterior <inline-formula>
<mml:math display="inline" id="im81"><mml:mrow><mml:msub><mml:mi>q</mml:mi><mml:mi>&#x3d5;</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>|</mml:mo><mml:msubsup><mml:mi>&#x210b;</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> and the reference prior <inline-formula>
<mml:math display="inline" id="im82"><mml:mrow><mml:mi>&#x3c0;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> for each patient and timestep, resulting in the anchoring regularizer (<xref ref-type="disp-formula" rid="eq24">Equation 24</xref>).</p>
<disp-formula id="eq24"><label>(24)</label>
<mml:math display="block" id="M24"><mml:mrow><mml:msub><mml:mi>&#x211b;</mml:mi><mml:mrow><mml:mtext>anchor</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>T</mml:mi></mml:munderover><mml:mrow><mml:mtext>KL</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>q</mml:mi><mml:mi>&#x3d5;</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>|</mml:mo><mml:msubsup><mml:mi>&#x210b;</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x3c0;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:mstyle></mml:mrow></mml:math>
</disp-formula>
<p>which constrains posterior mass to reside in regions of latent space associated with physiologically reasonable states. This promotes semantic interpretability of latent factors and mitigates drift under distributional shift. Beyond distributional anchoring, we further enhance alignment between latent structure and clinical semantics by integrating symbolic treatment class information into the model&#x2019;s internal attention dynamics. Given a treatment taxonomy that clusters drugs into shared classes based on therapeutic function, we define a set <inline-formula>
<mml:math display="inline" id="im83"><mml:mrow><mml:msub><mml:mi mathvariant="script">A</mml:mi><mml:mrow><mml:mtext>cluster</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> representing all such clusters, and associate each class <inline-formula>
<mml:math display="inline" id="im84"><mml:mi>c</mml:mi></mml:math></inline-formula> with a learned centroid embedding <inline-formula>
<mml:math display="inline" id="im85"><mml:mrow><mml:msub><mml:mi>e</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. At each timestep <inline-formula>
<mml:math display="inline" id="im86"><mml:mi>t</mml:mi></mml:math></inline-formula>, the model computes attention scores between the current latent state <inline-formula>
<mml:math display="inline" id="im87"><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> and all class centroids, reflecting the contextual relevance of each therapeutic group to the patient&#x2019;s latent status. The class-level attention is defined via a softmax-normalized inner product (<xref ref-type="disp-formula" rid="eq25">Equation 25</xref>):</p>
<disp-formula id="eq25"><label>(25)</label>
<mml:math display="block" id="M25"><mml:mrow><mml:msubsup><mml:mi>&#x3b1;</mml:mi><mml:mi>t</mml:mi><mml:mi>c</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mtext>exp&#xa0;</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mo>&#x2329;</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mi>e</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:mrow><mml:mo>&#x232a;</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mo>&#x2211;</mml:mo><mml:msup><mml:mi>c</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup></mml:msub><mml:mi>exp</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mo>&#x2329;</mml:mo><mml:mrow><mml:msubsup><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mi>e</mml:mi><mml:msup><mml:mi>c</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup></mml:msub></mml:mrow><mml:mo>&#x232a;</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im88"><mml:mrow><mml:msubsup><mml:mi>&#x3b1;</mml:mi><mml:mi>t</mml:mi><mml:mi>c</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> denotes the attention weight assigned to class <italic>c</italic> at time <italic>t</italic>, and (&#xb7;,&#xb7;) is the dot-product similarity. These attention scores modulate the downstream influence of treatment embeddings and enable context-aware prioritization of pharmacological pathways. To refine the interpretive resolution of this attention mechanism and facilitate hierarchical reasoning, we impose an entropy-aware regularization term that prevents overconcentration of attention and encourages exploration across class-level hypotheses. To couple latent anchoring with downstream outcome dynamics, we regularize the decoder&#x2019;s output trajectory to maintain consistency with stage-specific expectations. Let <inline-formula>
<mml:math display="inline" id="im89"><mml:mrow><mml:msub><mml:mi>&#x3bc;</mml:mi><mml:mrow><mml:mtext>stage</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> represent the expected clinical outcome at time <inline-formula>
<mml:math display="inline" id="im90"><mml:mi>t</mml:mi></mml:math></inline-formula> for a given disease stage, obtained from&#xa0;historical data or medical literature, and let <inline-formula>
<mml:math display="inline" id="im91"><mml:mrow><mml:msubsup><mml:mover accent="true"><mml:mi>y</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> denote the predicted outcome. We define a stage-informed outcome penalty as (<xref ref-type="disp-formula" rid="eq26">Equation 26</xref>).</p>
<disp-formula id="eq26"><label>(26)</label>
<mml:math display="block" id="M26"><mml:mrow><mml:msub><mml:mi>&#x211b;</mml:mi><mml:mrow><mml:mtext>stage</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>T</mml:mi></mml:munderover><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:msubsup><mml:mover accent="true"><mml:mi>y</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>&#x3bc;</mml:mi><mml:mrow><mml:mtext>stage</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:msubsup><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>which ensures the decoded response trajectories remain consistent with anchored latent semantics.</p>
<p>These mechanisms together constrain latent dynamics within clinically meaningful manifolds, dynamically link representations to pharmacological structure, and induce outcome behavior consistent with domain priors.</p>
</sec>
<sec id="s3_4_3">
<label>3.4.3</label>
<title>Discriminative counterfactual training</title>
<p>In order to improve the fidelity, realism, and clinical reliability of counterfactual outcome estimation, we introduce a discriminative adversarial mechanism that imposes implicit supervision over hypothetical predictions (as shown in <xref ref-type="fig" rid="f5"><bold>Figure&#xa0;5</bold></xref>).</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Illustration of discriminative counterfactual training. This figure provides an architectural overview of the proposed counterfactual training mechanism, which integrates attention-based latent dynamics, transformer-style contextualization, and adversarial discrimination. The left module highlights the attention computation across queries, keys, and values. The central block introduces discriminative supervision applied at intermediate transformer layers, enforcing semantic alignment between factual and counterfactual flows. On the right, a sequence of normalization, encoding, decoding, and projection operations enables contrastive regularization and robust representation of latent shifts. These components together realize a stable and semantically grounded framework for learning clinically plausible hypothetical outcomes.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1619994-g005.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a neural network architecture with three main components: MatMul with Softmax and Rescale transformations in purple; Add &amp; Norm with Feed Forward and Discriminative Counterfactual Training in green; and Embedding with Encoder and Decoder, followed by De-normalization and a Projector in peach. Arrows indicate the flow of data between components.</alt-text>
</graphic></fig>
<p>In real-world healthcare applications, treatment-effect estimation often requires generating unobserved responses under alternative interventions <inline-formula>
<mml:math display="inline" id="im92"><mml:mrow><mml:msubsup><mml:mover accent="true"><mml:mi>a</mml:mi><mml:mo>&#x2dc;</mml:mo></mml:mover><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup><mml:mo>&#x2260;</mml:mo><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula>, and ensuring the plausibility of these predictions is critical for deployment in clinical decision support systems. To this end, we define a discriminator network <inline-formula>
<mml:math display="inline" id="im93"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi>&#x3c8;</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mi>a</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> that takes as input the latent state <inline-formula>
<mml:math display="inline" id="im94"><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and a treatment <inline-formula>
<mml:math display="inline" id="im95"><mml:mi>a</mml:mi></mml:math></inline-formula> and outputs a scalar probability indicating whether the associated response is drawn from a factual (observed) or counterfactual (synthetic) distribution. Let <inline-formula>
<mml:math display="inline" id="im96"><mml:mrow><mml:msub><mml:mover accent="true"><mml:mi>a</mml:mi><mml:mo>&#x2dc;</mml:mo></mml:mover><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> denote a randomly sampled alternative intervention and let <inline-formula>
<mml:math display="inline" id="im97"><mml:mrow><mml:msup><mml:mover accent="true"><mml:mi>y</mml:mi><mml:mo>^</mml:mo></mml:mover><mml:mrow><mml:mtext>cf</mml:mtext></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="script">D</mml:mi><mml:mi>y</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>&#x3b8;</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mover accent="true"><mml:mi>a</mml:mi><mml:mo>&#x2dc;</mml:mo></mml:mover><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> represent the counterfactual prediction. The discriminator is trained to maximize classification accuracy between real and synthetic outcomes, while the generator is trained adversarially to minimize the ability of the discriminator to detect the distinction. This min&#x2013;max game is captured by the following objective (<xref ref-type="disp-formula" rid="eq27">Equation 27</xref>):</p>
<disp-formula id="eq27"><label>(27)</label>
<mml:math display="block" id="M27"><mml:mrow><mml:msub><mml:mi>&#x2112;</mml:mi><mml:mrow><mml:mtext>disc</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="double-struck">E</mml:mi><mml:mrow><mml:mtext>cf</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mrow><mml:mtext>log&#xa0;</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mi>&#x3c8;</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mover accent="true"><mml:mi>a</mml:mi><mml:mo>&#x2dc;</mml:mo></mml:mover><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo stretchy="false">]</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:msub><mml:mi mathvariant="double-struck">E</mml:mi><mml:mrow><mml:mtext>real</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mrow><mml:mtext>log&#xa0;</mml:mtext><mml:msub><mml:mi>D</mml:mi><mml:mi>&#x3c8;</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>where the expectation over real samples is taken with respect to the empirical training distribution and the counterfactual samples are generated on-the-fly through dynamic substitution. This adversarial alignment enforces semantic similarity between factual and hypothetical representations and implicitly regularizes the latent dynamics to remain consistent under both observed and imagined transitions. To stabilize optimization and propagate informative gradients back to the generator, we further incorporate the discriminator into the global learning objective alongside symbolic consistency, latent anchoring, temporal smoothness, and variational reconstruction. The composite objective optimized by the generator becomes (<xref ref-type="disp-formula" rid="eq28">Equation 28</xref>):</p>
<disp-formula id="eq28"><label>(28)</label>
<mml:math display="block" id="M28"><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi mathvariant="script">J</mml:mi><mml:mrow><mml:mtext>total</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>&#x2112;</mml:mi><mml:mrow><mml:mtext>ELBO</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>&#x3bb;</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>&#xb7;</mml:mo><mml:msub><mml:mi>&#x2112;</mml:mi><mml:mrow><mml:mtext>consist</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>&#x3bb;</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>&#xb7;</mml:mo><mml:msub><mml:mi>&#x211b;</mml:mi><mml:mrow><mml:mtext>corridor</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>&#x3bb;</mml:mi><mml:mn>3</mml:mn></mml:msub><mml:mo>&#xb7;</mml:mo><mml:msub><mml:mi>&#x211b;</mml:mi><mml:mrow><mml:mtext>anchor</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x3bb;</mml:mi><mml:mn>4</mml:mn></mml:msub><mml:mo>&#xb7;</mml:mo><mml:msub><mml:mi>&#x2112;</mml:mi><mml:mrow><mml:mtext>disc</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>&#x3bb;</mml:mi><mml:mn>5</mml:mn></mml:msub><mml:mo>&#xb7;</mml:mo><mml:msub><mml:mi>&#x211b;</mml:mi><mml:mrow><mml:mtext>temp</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math>
</disp-formula>
<p>with hyperparameters <italic>&#x3bb;<sub>i</sub></italic> balancing the influence of domain-guided priors and adversarial supervision. Model parameters <italic>&#x3b8;</italic> and <italic>&#x3d5;</italic> are updated by minimizing <inline-formula>
<mml:math display="inline" id="im98"><mml:mrow><mml:mtext>&#xa0;</mml:mtext><mml:msub><mml:mi mathvariant="script">J</mml:mi><mml:mrow><mml:mtext>total</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula>, while the discriminator parameters <italic>&#x3c8;</italic> are optimized independently to maximize its classification capacity. This leads to a dual-loop adversarial learning process formalized as (<xref ref-type="disp-formula" rid="eq29">Equation 29</xref>).</p>
<disp-formula id="eq29"><label>(29)</label>
<mml:math display="block" id="M29"><mml:mrow><mml:mi>&#x3b8;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x3d5;</mml:mi><mml:mo>&#x2190;</mml:mo><mml:mtext>arg&#xa0;min&#xa0;</mml:mtext><mml:msub><mml:mi mathvariant="script">J</mml:mi><mml:mrow><mml:mtext>total</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x2003;&#x2003;</mml:mtext><mml:mi>&#x3c8;</mml:mi><mml:mo>&#x2190;</mml:mo><mml:mtext>arg&#xa0;min&#xa0;</mml:mtext><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>&#x2112;</mml:mi><mml:mrow><mml:mtext>disc</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>where gradients are propagated alternately through the generator and discriminator networks. To further reinforce counterfactual consistency at the representation level, we introduce a contrastive regularization term over the latent shifts induced by factual and counterfactual actions. Letting <inline-formula>
<mml:math display="inline" id="im99"><mml:mrow><mml:msub><mml:mtext>&#x394;</mml:mtext><mml:mrow><mml:mtext>real</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>&#x3b8;</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula>
<mml:math display="inline" id="im100"><mml:mrow><mml:msub><mml:mtext>&#x394;</mml:mtext><mml:mrow><mml:mtext>cf</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>&#x3b8;</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mover accent="true"><mml:mi>a</mml:mi><mml:mo>&#x2dc;</mml:mo></mml:mover><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, we define the shift-alignment penalty (<xref ref-type="disp-formula" rid="eq30">Equation 30</xref>).</p>
<disp-formula id="eq30"><label>(30)</label>
<mml:math display="block" id="M30"><mml:mrow><mml:msub><mml:mi>&#x211b;</mml:mi><mml:mrow><mml:mtext>shift</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="double-struck">E</mml:mi><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mover accent="true"><mml:mi>a</mml:mi><mml:mo>&#x2dc;</mml:mo></mml:mover><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mtext>&#x394;</mml:mtext><mml:mrow><mml:mtext>real</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mtext>&#x394;</mml:mtext><mml:mrow><mml:mtext>cf</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:msubsup><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math>
</disp-formula>
<p>which encourages the model to produce smooth and structurally coherent latent transitions even when simulating hypothetical outcomes. This constraint enhances the stability and realism of generated trajectories and helps preserve interpretability across the intervention space.</p>
<p>Accurately modeling treatment response in clinical settings involves handling temporal dynamics, missing data, and heterogeneous patient characteristics. To address these challenges, the proposed framework integrates prior clinical knowledge with data-driven learning to simulate how patients evolve under different treatment regimens. The core idea is to abstract a patient&#x2019;s physiological condition into a latent state that evolves over time in response to medical interventions. This latent representation serves as a compact summary of the patient&#x2019;s health status and allows prediction of future clinical outcomes based on past trajectories. Two key principles guide the design of the system. First, the model accounts for pharmacological structure by embedding treatments into a symbolic space informed by clinical taxonomy and prior knowledge. This enables generalization across drugs with similar mechanisms. Second, the framework supports counterfactual simulation, allowing evaluation of alternative treatment scenarios not observed during training. This feature is particularly useful for decision support and personalized planning. By combining interpretable latent dynamics with clinical priors, the system aims to achieve both predictive accuracy and semantic transparency. The design balances mathematical rigor with practical interpretability to support decision-making in oncology and other domains.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experimental setup</title>
<sec id="s4_1">
<label>4.1</label>
<title>Dataset</title>
<p>The BreakHis dataset (<xref ref-type="bibr" rid="B42">42</xref>), the CBIS-DDSM dataset (<xref ref-type="bibr" rid="B43">43</xref>), the INbreast dataset (<xref ref-type="bibr" rid="B44">44</xref>), and the TCGA-BRCA dataset (<xref ref-type="bibr" rid="B45">45</xref>) are four widely utilized and publicly available breast cancer imaging datasets that serve as foundational resources for computer-aided diagnosis and machine learning research in medical imaging. BreakHis (Breast Cancer Histopathological Image Classification) consists of microscopic biopsy images of breast tumors, acquired using magnification factors of &#xd7;40, &#xd7;100, &#xd7;200, and &#xd7;400. This dataset includes 7,909 images from 82 patients and is categorized into benign and malignant classes, further subdivided into different histopathological subtypes. The diversity of magnification and histological patterns makes it suitable for deep learning tasks focused on feature representation and classification of breast cancer. In contrast, the CBIS-DDSM (Curated Breast Imaging Subset of the Digital Database for Screening Mammography) provides a large collection of mammogram images with verified pathology information. This dataset is a curated and standardized subset of the original DDSM, including over 3,000 mammography studies with annotations such as bounding boxes and lesion characteristics, covering calcifications and masses. It is particularly valuable for segmentation, detection, and classification research involving full-field digital mammography. The INbreast dataset is a high-resolution full-field digital mammography dataset that contains 115 cases with a total of 410 images, where each image is annotated by medical experts with precise contours of masses and calcifications. The high quality and detailed annotations make INbreast especially suitable for fine-grained segmentation tasks and the evaluation of lesion characterization algorithms. The TCGA-BRCA dataset, part of The Cancer Genome Atlas program, combines histopathological images with genomic, clinical, and demographic data from breast cancer patients. This dataset is unique in that it enables multi-modal analysis, integrating imaging data with gene expression profiles, mutation data, and other molecular features. TCGA-BRCA includes both hematoxylin and eosin (H&amp;E)-stained whole-slide images and a wide array of omics data, offering a rich platform for research at the intersection of computational pathology and cancer genomics. These datasets together support a broad range of applications from basic tumor detection to advanced integrative analyses aimed at personalized medicine and precision oncology, and their complementary nature allows for comprehensive modeling of breast cancer from image-level features to molecular signatures.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Experimental details</title>
<p>In our experiments, we adopt a standard training and evaluation pipeline to ensure fair comparison across all datasets. For all tasks, we utilize a ResNet-50 backbone and a Vision Transformer (ViT-B/16) as representative architectures for convolutional and transformer-based models, respectively. The networks are initialized with BreakHis-pretrained weights to accelerate convergence and enhance generalization. For optimization, we use stochastic gradient descent (SGD) with a momentum of 0.9 and weight decay of 1 &#xd7; 10<sup>&#x2212;4</sup>. The initial learning rate is set to 0.01 and follows a cosine annealing schedule without restarts. The batch size is fixed at 128 for all datasets, and training is conducted for 100 epochs on each dataset. For datasets with fewer samples such as INbreast and TCGA-BRCA, we employ data augmentation techniques including random cropping, horizontal flipping, and color jittering to reduce overfitting and improve robustness. For CBIS-DDSM, the standard split of 60 training images per class is adopted, and the rest are used for evaluation. For INbreast, we follow the official split protocol with 1,020 training, 1,020 validation, and 6,149 test images. For the TCGA-BRCA dataset, we randomly divide the dataset into 60% training, 20% validation, and 20% testing while ensuring that each attribute label is uniformly distributed across the splits. The BreakHis dataset follows the standard ILSVRC-2012 training and validation splits, where the model is trained on the 1.2 million training images and evaluated on the 50,000 validation images. To stabilize training on small datasets, we employ label smoothing with a factor of 0.1 and dropout with a rate of 0.5 in the fully connected layers. For ViT-based models, we use a fixed patch size of 16 and positional embeddings are retained throughout training. The transformer model is optimized using the AdamW optimizer with a learning rate of 3 &#xd7; 10<sup>&#x2212;4</sup> and a linear warm-up phase of 10 epochs followed by cosine decay. All experiments are conducted on four NVIDIA A100 GPUs with 40 GB of memory each, using PyTorch 2.1 and CUDA 12.2. Mixed precision training is applied to accelerate computation without loss in accuracy. We report the top-one classification accuracy as the primary evaluation metric. To ensure reproducibility, we fix random seeds for NumPy and PyTorch and log all hyperparameters, loss curves, and model checkpoints using the weights and biases framework. Hyperparameter tuning is done via grid search on the validation set, where learning rates, dropout rates, and augmentation strength are systematically explored. We also evaluate the robustness of each model to common corruptions using the BreakHis-C benchmark in extended experiments. This setup ensures that our experimental results are rigorous, reproducible, and comparable to recent state-of-the-art benchmarks.</p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Comparison with SOTA methods</title>
<p>We perform a comprehensive comparison between our proposed method ResponseNet and several state-of-the-art (SOTA) baselines across four benchmark datasets: BreakHis, CBIS-DDSM, INbreast, and TCGA-BRCA. In <xref ref-type="table" rid="T1"><bold>Tables&#xa0;1</bold></xref>, <xref ref-type="table" rid="T2"><bold>2</bold></xref>, ResponseNet consistently outperforms all other models across all metrics and datasets. On the large-scale BreakHis dataset, ResponseNet achieves an accuracy of 81.87%, surpassing the next best method, EfficientNet-B4, by a margin of 2.45%. Similar gains are observed for precision and F1 score, demonstrating ResponseNet&#x2019;s ability to balance true positive recognition with low false positive rates. The AUC score also shows a significant improvement, indicating enhanced discriminative capability under varying decision thresholds. On CBIS-DDSM, ResponseNet achieves 88.31% accuracy, notably outperforming RegNetY-16GF and ViT-B/16, which achieved 86.02% and 85.39%, respectively. These improvements are attributed to ResponseNet&#x2019;s hybrid architecture, which effectively captures both local and global features, leveraging multi-scale representations to handle object variability and background complexity. For fine-grained datasets such as INbreast, ResponseNet yields a substantial accuracy of 94.89%, outperforming ConvNeXt-T by 2.88%. Notably, the model also achieves the highest precision and F1 scores among all methods, illustrating its robustness in distinguishing classes with subtle inter-class variations. These gains can be attributed to ResponseNet&#x2019;s class-aware attention mechanism, which enhances feature representation for visually similar categories. In terms of AUC, ResponseNet achieves 96.21%, reflecting its superior capability in confident classification. Similarly, on the TCGA-BRCA Dataset, ResponseNet obtains a top accuracy of 77.92%, improving upon RegNetY-16GF by 3.16%. The precision and F1 scores of ResponseNet are also significantly higher than those of conventional CNNs and vision transformers, affirming ResponseNet&#x2019;s capability in modeling abstract and perceptual-level texture attributes. The enhanced performance on TCGA-BRCA stems from ResponseNet&#x2019;s hierarchical decomposition module, which decomposes texture patterns into interpretable units, leading to more robust and generalizable learning. This aligns with the nature of TCGA-BRCA where semantic texture attributes are subtle and often rely on mid-level visual cues. The superior AUC scores across all datasets further validate the generalization of ResponseNet, particularly in challenging classification scenarios with imbalanced or noisy data.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Performance benchmarking of our approach against leading techniques on BreakHis and CBIS-DDSM datasets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Model</th>
<th valign="middle" colspan="4" align="center">Breakhis dataset</th>
<th valign="middle" colspan="4" align="center">CBIS-DDSM dataset</th>
</tr>
<tr>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">F1 score</th>
<th valign="middle" align="center">AUC</th>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">F1 score</th>
<th valign="middle" align="center">AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">ResNet50 Elpeltagy and Sallam (<xref ref-type="bibr" rid="B46">46</xref>)</td>
<td valign="middle" align="center">77.23<inline-formula>
<mml:math display="inline" id="im101"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">75.80<inline-formula>
<mml:math display="inline" id="im102"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula> 0.15</td>
<td valign="middle" align="center">76.04<inline-formula>
<mml:math display="inline" id="im103"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.14</td>
<td valign="middle" align="center">81.67<inline-formula>
<mml:math display="inline" id="im104"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">84.51<inline-formula>
<mml:math display="inline" id="im105"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</td>
<td valign="middle" align="center">83.20<inline-formula>
<mml:math display="inline" id="im106"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">83.45<inline-formula>
<mml:math display="inline" id="im107"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.07</td>
<td valign="middle" align="center">86.30<inline-formula>
<mml:math display="inline" id="im108"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
</tr>
<tr>
<td valign="middle" align="center">ViT-B/16 Hong et&#xa0;al. (<xref ref-type="bibr" rid="B47">47</xref>)</td>
<td valign="middle" align="center">78.65<inline-formula>
<mml:math display="inline" id="im109"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.14</td>
<td valign="middle" align="center">76.90<inline-formula>
<mml:math display="inline" id="im110"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">77.41<inline-formula>
<mml:math display="inline" id="im111"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">83.12<inline-formula>
<mml:math display="inline" id="im112"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.13</td>
<td valign="middle" align="center">85.39<inline-formula>
<mml:math display="inline" id="im113"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">84.55<inline-formula>
<mml:math display="inline" id="im114"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">84.33<inline-formula>
<mml:math display="inline" id="im115"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">87.75<inline-formula>
<mml:math display="inline" id="im116"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
</tr>
<tr>
<td valign="middle" align="center">EfficientNet-B4 Preetha et&#xa0;al. (<xref ref-type="bibr" rid="B48">48</xref>)</td>
<td valign="middle" align="center">79.42<inline-formula>
<mml:math display="inline" id="im117"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">78.50<inline-formula>
<mml:math display="inline" id="im118"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">78.61<inline-formula>
<mml:math display="inline" id="im119"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">84.88<inline-formula>
<mml:math display="inline" id="im120"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">83.95<inline-formula>
<mml:math display="inline" id="im121"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">82.80<inline-formula>
<mml:math display="inline" id="im122"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.13</td>
<td valign="middle" align="center">83.<inline-formula>
<mml:math display="inline" id="im123"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">85.69<inline-formula>
<mml:math display="inline" id="im124"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
</tr>
<tr>
<td valign="middle" align="center">ConvNeXt-T Yu et&#xa0;al. (<xref ref-type="bibr" rid="B49">49</xref>)</td>
<td valign="middle" align="center">76.90<inline-formula>
<mml:math display="inline" id="im125"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.13</td>
<td valign="middle" align="center">74.45<inline-formula>
<mml:math display="inline" id="im126"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.14</td>
<td valign="middle" align="center">75.12<inline-formula>
<mml:math display="inline" id="im127"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.13</td>
<td valign="middle" align="center">80.33<inline-formula>
<mml:math display="inline" id="im128"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">84.80<inline-formula>
<mml:math display="inline" id="im129"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">83.67<inline-formula>
<mml:math display="inline" id="im130"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</td>
<td valign="middle" align="center">83.98<inline-formula>
<mml:math display="inline" id="im131"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">85.45<inline-formula>
<mml:math display="inline" id="im132"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet201 Mohandass et&#xa0;al. (<xref ref-type="bibr" rid="B50">50</xref>)</td>
<td valign="middle" align="center">77.96<inline-formula>
<mml:math display="inline" id="im133"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">76.10<inline-formula>
<mml:math display="inline" id="im134"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">76.82<inline-formula>
<mml:math display="inline" id="im135"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">82.44<inline-formula>
<mml:math display="inline" id="im136"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">82.79<inline-formula>
<mml:math display="inline" id="im137"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.13</td>
<td valign="middle" align="center">81.05<inline-formula>
<mml:math display="inline" id="im138"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">81.83<inline-formula>
<mml:math display="inline" id="im139"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.13</td>
<td valign="middle" align="center">84.50<inline-formula>
<mml:math display="inline" id="im140"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.14</td>
</tr>
<tr>
<td valign="middle" align="center">RegNetY-16GF Pandey et&#xa0;al. (<xref ref-type="bibr" rid="B51">51</xref>)</td>
<td valign="middle" align="center">78.34<inline-formula>
<mml:math display="inline" id="im141"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">77.55<inline-formula>
<mml:math display="inline" id="im142"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">77.22<inline-formula>
<mml:math display="inline" id="im143"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">83.96<inline-formula>
<mml:math display="inline" id="im144"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.13</td>
<td valign="middle" align="center">86.02<inline-formula>
<mml:math display="inline" id="im145"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">84.98<inline-formula>
<mml:math display="inline" id="im146"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">85.00<inline-formula>
<mml:math display="inline" id="im147"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">87.40<inline-formula>
<mml:math display="inline" id="im148"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
</tr>
<tr>
<td valign="middle" align="center"><bold>Ours (ResponseNet)</bold></td>
<td valign="middle" align="center"><bold>81.87<inline-formula>
<mml:math display="inline" id="im149"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</bold></td>
<td valign="middle" align="center"><bold>80.92<inline-formula>
<mml:math display="inline" id="im150"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</bold></td>
<td valign="middle" align="center"><bold>80.75<inline-formula>
<mml:math display="inline" id="im151"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</bold></td>
<td valign="middle" align="center"><bold>86.55<inline-formula>
<mml:math display="inline" id="im152"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</bold></td>
<td valign="middle" align="center"><bold>88.31<inline-formula>
<mml:math display="inline" id="im153"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.07</bold></td>
<td valign="middle" align="center"><bold>87.63<inline-formula>
<mml:math display="inline" id="im154"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</bold></td>
<td valign="middle" align="center"><bold>87.88<inline-formula>
<mml:math display="inline" id="im155"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.07</bold></td>
<td valign="middle" align="center"><bold>89.42<inline-formula>
<mml:math display="inline" id="im156"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The values in bold refer to our method.</p></fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Performance benchmarking of our approach against leading techniques on INbreast and TCGA-BRCA datasets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Model</th>
<th valign="middle" colspan="4" align="center">INbreast</th>
<th valign="middle" colspan="4" align="center">TCGA-BRCA dataset</th>
</tr>
<tr>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">F1 score</th>
<th valign="middle" align="center">AUC</th>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">F1 score</th>
<th valign="middle" align="center">AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">ResNet50 Elpeltagy and Sallam (<xref ref-type="bibr" rid="B46">46</xref>)</td>
<td valign="middle" align="center">91.43<inline-formula>
<mml:math display="inline" id="im157"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">90.17<inline-formula>
<mml:math display="inline" id="im158"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</td>
<td valign="middle" align="center">90.83<inline-formula>
<mml:math display="inline" id="im159"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">93.20<inline-formula>
<mml:math display="inline" id="im160"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">72.55<inline-formula>
<mml:math display="inline" id="im161"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">71.44<inline-formula>
<mml:math display="inline" id="im162"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.13</td>
<td valign="middle" align="center">70.83<inline-formula>
<mml:math display="inline" id="im163"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">74.01<inline-formula>
<mml:math display="inline" id="im164"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
</tr>
<tr>
<td valign="middle" align="center">ViT-B/16 Hong et&#xa0;al. (<xref ref-type="bibr" rid="B47">47</xref>)</td>
<td valign="middle" align="center">90.68<inline-formula>
<mml:math display="inline" id="im165"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">89.02<inline-formula>
<mml:math display="inline" id="im166"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">89.74<inline-formula>
<mml:math display="inline" id="im167"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">92.77<inline-formula>
<mml:math display="inline" id="im168"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">74.23<inline-formula>
<mml:math display="inline" id="im169"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">73.66<inline-formula>
<mml:math display="inline" id="im170"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">73.48<inline-formula>
<mml:math display="inline" id="im171"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">75.88<inline-formula>
<mml:math display="inline" id="im172"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.13</td>
</tr>
<tr>
<td valign="middle" align="center">EfficientNet-B4 Preetha et&#xa0;al. (<xref ref-type="bibr" rid="B48">48</xref>)</td>
<td valign="middle" align="center">89.92<inline-formula>
<mml:math display="inline" id="im173"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">91.15<inline-formula>
<mml:math display="inline" id="im174"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">90.04<inline-formula>
<mml:math display="inline" id="im175"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">91.89<inline-formula>
<mml:math display="inline" id="im176"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">73.89<inline-formula>
<mml:math display="inline" id="im177"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">72.11<inline-formula>
<mml:math display="inline" id="im178"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">71.96<inline-formula>
<mml:math display="inline" id="im179"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">76.21<inline-formula>
<mml:math display="inline" id="im180"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
</tr>
<tr>
<td valign="middle" align="center">ConvNeXt-T Yu et&#xa0;al. (<xref ref-type="bibr" rid="B49">49</xref>)</td>
<td valign="middle" align="center">92.01<inline-formula>
<mml:math display="inline" id="im181"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">90.60<inline-formula>
<mml:math display="inline" id="im182"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">91.08<inline-formula>
<mml:math display="inline" id="im183"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">94.04<inline-formula>
<mml:math display="inline" id="im184"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</td>
<td valign="middle" align="center">71.74<inline-formula>
<mml:math display="inline" id="im185"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.13</td>
<td valign="middle" align="center">72.39<inline-formula>
<mml:math display="inline" id="im186"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">72.17<inline-formula>
<mml:math display="inline" id="im187"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">73.45<inline-formula>
<mml:math display="inline" id="im188"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet201 Mohandass et&#xa0;al. (<xref ref-type="bibr" rid="B50">50</xref>)</td>
<td valign="middle" align="center">90.45<inline-formula>
<mml:math display="inline" id="im189"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</td>
<td valign="middle" align="center">88.77<inline-formula>
<mml:math display="inline" id="im190"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">89.66<inline-formula>
<mml:math display="inline" id="im191"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">92.33<inline-formula>
<mml:math display="inline" id="im192"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">70.91<inline-formula>
<mml:math display="inline" id="im193"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">70.12<inline-formula>
<mml:math display="inline" id="im194"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.13</td>
<td valign="middle" align="center">69.89<inline-formula>
<mml:math display="inline" id="im195"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">72.00<inline-formula>
<mml:math display="inline" id="im196"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
</tr>
<tr>
<td valign="middle" align="center">RegNetY-16GF Pandey et&#xa0;al. (<xref ref-type="bibr" rid="B51">51</xref>)</td>
<td valign="middle" align="center">91.17<inline-formula>
<mml:math display="inline" id="im197"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">89.90<inline-formula>
<mml:math display="inline" id="im198"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">90.35<inline-formula>
<mml:math display="inline" id="im199"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">93.75<inline-formula>
<mml:math display="inline" id="im200"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">74.76<inline-formula>
<mml:math display="inline" id="im201"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">73.98<inline-formula>
<mml:math display="inline" id="im202"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</td>
<td valign="middle" align="center">73.81<inline-formula>
<mml:math display="inline" id="im203"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">76.68<inline-formula>
<mml:math display="inline" id="im204"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
</tr>
<tr>
<td valign="middle" align="center"><bold>Ours (ResponseNet)</bold></td>
<td valign="middle" align="center"><bold>94.89&#xa0;&#xb1;&#xa0;0.07</bold></td>
<td valign="middle" align="center"><bold>93.75<inline-formula>
<mml:math display="inline" id="im205"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</bold></td>
<td valign="middle" align="center"><bold>94.11<inline-formula>
<mml:math display="inline" id="im206"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</bold></td>
<td valign="middle" align="center"><bold>96.21<inline-formula>
<mml:math display="inline" id="im207"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</bold></td>
<td valign="middle" align="center"><bold>77.92<inline-formula>
<mml:math display="inline" id="im208"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</bold></td>
<td valign="middle" align="center"><bold>76.60<inline-formula>
<mml:math display="inline" id="im209"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</bold></td>
<td valign="middle" align="center"><bold>76.98<inline-formula>
<mml:math display="inline" id="im210"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</bold></td>
<td valign="middle" align="center"><bold>79.04<inline-formula>
<mml:math display="inline" id="im211"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The values in bold refer to our method.</p></fn>
</table-wrap-foot>
</table-wrap>
<p>The consistent improvements of ResponseNet across all datasets can be explained by the following architectural advantages. ResponseNet integrates both convolutional and attention-based modules to leverage the locality and long-range dependencies effectively. This synergy allows the model to retain fine-grained details while also attending to holistic context. Then, ResponseNet introduces a category-guided memory unit, which stores representative features and enhances the attention weights during inference, effectively functioning as an external knowledge bank. This module is especially helpful in fine-grained and texture-based classification tasks like Oxford 102 and TCGA-BRCA, where intra-class variance is low but inter-class boundaries are subtle. The progressive decoding strategy adopted in ResponseNet stabilizes training and improves gradient flow, making the model more robust to architectural depth and hyperparameter variations. Unlike standard residual or transformer blocks that rely heavily on depth, ResponseNet&#x2019;s progressive nature allows for smoother representation fusion. The training pipeline, including tailored data augmentations and loss function design, contributes to ResponseNet&#x2019;s ability to generalize across domains. While traditional models rely heavily on large-scale pretraining, ResponseNet benefits from its internal regularization, leading to better adaptation on smaller datasets such as CBIS-DDSM and TCGA-BRCA. ResponseNet achieves better separation among classes and significantly fewer misclassifications. In summary, ResponseNet delivers comprehensive improvements across metrics and datasets, validating the effectiveness of our design and its capability to set a new benchmark for visual recognition tasks.</p>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Ablation study</title>
<p>To validate the effectiveness of each key component in our proposed ResponseNet architecture, we conduct a series of ablation studies on four datasets: BreakHis, CBIS-DDSM, INbreast, and TCGA-BRCA. The ablation settings include three variants: without latent dynamics modeling, which removes the category-guided memory module; without semantic treatment embedding, which disables the hierarchical feature fusion; and without latent space anchoring, which eliminates the progressive decoding module. The results are shown in <xref ref-type="table" rid="T3"><bold>Tables&#xa0;3</bold></xref>, <xref ref-type="table" rid="T4"><bold>4</bold></xref>. Across all datasets and metrics, we observe a consistent degradation in performance when any of these modules are removed, confirming that each component contributes meaningfully to the overall model efficacy. On BreakHis, removing the latent dynamics modeling module causes the most noticeable drop in accuracy and AUC, highlighting the importance of category-aware context storage in handling large-scale and diverse data. Meanwhile, removing semantic treatment embedding results in weaker precision and F1 score, suggesting that spatial-scale integration is crucial for maintaining class separability. The latent space anchoring module also plays a key role by stabilizing feature evolution, as its removal leads to lower consistency in predictions. A comparable pattern is found in the CBIS-DDSM dataset, where excluding latent dynamics modeling results in a reduction of accuracy from 88.31% to 86.50%, accompanied by a decline in AUC from 89.42% to 87.23%. This again confirms that without the memory component, the model struggles to preserve discriminative features, especially in categories with subtle appearance differences. The removal of the semantic treatment embedding (without semantic treatment embedding) reduces the model&#x2019;s ability to maintain spatial context, slightly decreasing performance but still retaining a relatively high margin, which implies that while this module is beneficial, it is partially complemented by the memory-guided features. The impact of removing the latent space anchoring structure is more prominent in precision and F1 score, emphasizing the role of this module in harmonizing learned features through the model layers.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Performance benchmarking of our approach against leading techniques on our model across BreakHis and CBIS-DDSM datasets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Model</th>
<th valign="middle" colspan="4" align="center">Breakhis dataset</th>
<th valign="middle" colspan="4" align="center">CBIS-DDSM dataset</th>
</tr>
<tr>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">F1 score</th>
<th valign="middle" align="center">AUC</th>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">F1 score</th>
<th valign="middle" align="center">AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Without latent dynamics modeling</td>
<td valign="middle" align="center">79.45<inline-formula>
<mml:math display="inline" id="im212"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">77.88<inline-formula>
<mml:math display="inline" id="im213"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">78.34<inline-formula>
<mml:math display="inline" id="im214"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">84.33<inline-formula>
<mml:math display="inline" id="im215"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.13</td>
<td valign="middle" align="center">86.50<inline-formula>
<mml:math display="inline" id="im216"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">85.42<inline-formula>
<mml:math display="inline" id="im217"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">85.26<inline-formula>
<mml:math display="inline" id="im218"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</td>
<td valign="middle" align="center">87.23<inline-formula>
<mml:math display="inline" id="im219"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
</tr>
<tr>
<td valign="middle" align="center">Without semantic treatment embedding</td>
<td valign="middle" align="center">80.21<inline-formula>
<mml:math display="inline" id="im220"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">79.11<inline-formula>
<mml:math display="inline" id="im221"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">78.88<inline-formula>
<mml:math display="inline" id="im222"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">85.02<inline-formula>
<mml:math display="inline" id="im223"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">87.13<inline-formula>
<mml:math display="inline" id="im224"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">85.91<inline-formula>
<mml:math display="inline" id="im225"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">86.18<inline-formula>
<mml:math display="inline" id="im226"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.07</td>
<td valign="middle" align="center">87.75<inline-formula>
<mml:math display="inline" id="im227"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
</tr>
<tr>
<td valign="middle" align="center">Without latent space anchoring</td>
<td valign="middle" align="center">80.87<inline-formula>
<mml:math display="inline" id="im228"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</td>
<td valign="middle" align="center">80.30<inline-formula>
<mml:math display="inline" id="im229"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">79.76<inline-formula>
<mml:math display="inline" id="im230"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">85.77<inline-formula>
<mml:math display="inline" id="im231"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">87.85<inline-formula>
<mml:math display="inline" id="im232"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.07</td>
<td valign="middle" align="center">86.88<inline-formula>
<mml:math display="inline" id="im233"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</td>
<td valign="middle" align="center">86.59<inline-formula>
<mml:math display="inline" id="im234"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</td>
<td valign="middle" align="center">88.60<inline-formula>
<mml:math display="inline" id="im235"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
</tr>
<tr>
<td valign="middle" align="center"><bold>Ours</bold></td>
<td valign="middle" align="center"><bold>81.87<inline-formula>
<mml:math display="inline" id="im236"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</bold></td>
<td valign="middle" align="center"><bold>80.92<inline-formula>
<mml:math display="inline" id="im237"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</bold></td>
<td valign="middle" align="center"><bold>80.75<inline-formula>
<mml:math display="inline" id="im238"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</bold></td>
<td valign="middle" align="center"><bold>86.55<inline-formula>
<mml:math display="inline" id="im239"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</bold></td>
<td valign="middle" align="center"><bold>88.31<inline-formula>
<mml:math display="inline" id="im240"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.07</bold></td>
<td valign="middle" align="center"><bold>87.63<inline-formula>
<mml:math display="inline" id="im241"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</bold></td>
<td valign="middle" align="center"><bold>87.88<inline-formula>
<mml:math display="inline" id="im242"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.07</bold></td>
<td valign="middle" align="center"><bold>89.42<inline-formula>
<mml:math display="inline" id="im243"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The values in bold refer to our method.</p></fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Performance benchmarking of our approach against leading techniques on our model across INbreast and TCGA-BRCA datasets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Model</th>
<th valign="middle" colspan="4" align="center">INbreast</th>
<th valign="middle" colspan="4" align="center">TCGA-BRCA dataset</th>
</tr>
<tr>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">F1 score</th>
<th valign="middle" align="center">AUC</th>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">F1 score</th>
<th valign="middle" align="center">AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Without latent dynamics modeling</td>
<td valign="middle" align="center">91.62<inline-formula>
<mml:math display="inline" id="im244"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">90.01<inline-formula>
<mml:math display="inline" id="im245"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">90.33<inline-formula>
<mml:math display="inline" id="im246"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">93.12<inline-formula>
<mml:math display="inline" id="im247"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">75.29<inline-formula>
<mml:math display="inline" id="im248"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">73.55<inline-formula>
<mml:math display="inline" id="im249"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">74.12<inline-formula>
<mml:math display="inline" id="im250"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">77.01<inline-formula>
<mml:math display="inline" id="im251"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
</tr>
<tr>
<td valign="middle" align="center">Without semantic treatment embedding</td>
<td valign="middle" align="center">92.47<inline-formula>
<mml:math display="inline" id="im252"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">91.60<inline-formula>
<mml:math display="inline" id="im253"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">91.18<inline-formula>
<mml:math display="inline" id="im254"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">94.08<inline-formula>
<mml:math display="inline" id="im255"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">76.23<inline-formula>
<mml:math display="inline" id="im256"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
<td valign="middle" align="center">74.91<inline-formula>
<mml:math display="inline" id="im257"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.12</td>
<td valign="middle" align="center">75.66<inline-formula>
<mml:math display="inline" id="im258"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.11</td>
<td valign="middle" align="center">78.12<inline-formula>
<mml:math display="inline" id="im259"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.10</td>
</tr>
<tr>
<td valign="middle" align="center">Without latent space anchoring</td>
<td valign="middle" align="center">93.04<inline-formula>
<mml:math display="inline" id="im260"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</td>
<td valign="middle" align="center">92.12<inline-formula>
<mml:math display="inline" id="im261"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">92.30<inline-formula>
<mml:math display="inline" id="im262"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</td>
<td valign="middle" align="center">95.02<inline-formula>
<mml:math display="inline" id="im263"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</td>
<td valign="middle" align="center">77.12<inline-formula>
<mml:math display="inline" id="im264"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">75.82<inline-formula>
<mml:math display="inline" id="im265"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</td>
<td valign="middle" align="center">76.42<inline-formula>
<mml:math display="inline" id="im266"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
<td valign="middle" align="center">78.66<inline-formula>
<mml:math display="inline" id="im267"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</td>
</tr>
<tr>
<td valign="middle" align="center"><bold>Ours</bold></td>
<td valign="middle" align="center"><bold>94.89<inline-formula>
<mml:math display="inline" id="im268"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.07</bold></td>
<td valign="middle" align="center"><bold>93.75<inline-formula>
<mml:math display="inline" id="im269"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</bold></td>
<td valign="middle" align="center"><bold>94.11<inline-formula>
<mml:math display="inline" id="im270"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</bold></td>
<td valign="middle" align="center"><bold>96.21<inline-formula>
<mml:math display="inline" id="im271"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</bold></td>
<td valign="middle" align="center"><bold>77.92<inline-formula>
<mml:math display="inline" id="im272"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</bold></td>
<td valign="middle" align="center"><bold>76.60<inline-formula>
<mml:math display="inline" id="im273"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</bold></td>
<td valign="middle" align="center"><bold>76.98<inline-formula>
<mml:math display="inline" id="im274"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.08</bold></td>
<td valign="middle" align="center"><bold>79.04<inline-formula>
<mml:math display="inline" id="im275"><mml:mrow><mml:mo>&#xa0;</mml:mo><mml:mo>&#xb1;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:math></inline-formula>0.09</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The values in bold refer to our method.</p></fn>
</table-wrap-foot>
</table-wrap>
<p>For fine-grained datasets such as INbreast and TCGA-BRCA, the effect of each module becomes even more pronounced. On Oxford 102, removal of the latent dynamics modeling module drops the accuracy by 3.27%, demonstrating how critical this component is for capturing subtle inter-class differences inherent in flower categories. Similarly, the semantic treatment embedding plays a pivotal role by improving the global-local balance in floral structures, while the latent space anchoring strategy enhances robustness against pose and color variation. On the TCGA-BRCA dataset, which requires recognition of abstract texture patterns, each module provides clear benefits. The latent dynamics modeling module provides a pseudo-semantic backbone that boosts precision and AUC, while semantic treatment embedding supports local pattern decoding, and latent space anchoring enables gradual abstraction&#x2014;essential for perceptual-level recognition. In conclusion, the full ResponseNet model exhibits a holistic improvement over all ablations, and the clear performance drops across all variants underline the necessity of each core module. These results demonstrate that our architectural components are not only additive but also interact synergistically, enabling the model to generalize well across diverse and complex datasets.</p>
<p>To assess generalizability in practical clinical contexts, two real-world oncology datasets were incorporated for extended evaluation. The METABRIC dataset provides gene expression and clinical data for 1980 breast cancer patients, while the CAMELYON16 dataset contains high-resolution histopathology slides for tumor metastasis detection in lymph nodes. ResponseNet was adapted to process structured data in METABRIC and image tiles in CAMELYON16, with model variants incorporating lightweight encoders and symbolic treatment mappings. In both cases, predictive accuracy and interpretability were compared against standard multimodal baselines, including early fusion (feature concatenation), late fusion (modality-specific encoders with shared attention), and gradient-boosted decision trees with imputed features. <xref ref-type="table" rid="T5"><bold>Table&#xa0;5</bold></xref> summarizes the results. The results show that ResponseNet outperforms baseline methods across both datasets in AUROC and F1-score, while uniquely offering interpretability through attention maps and symbolic reasoning modules. Its design enables integration of heterogeneous data types and maintains stability under modality dropout, which was tested by randomly masking clinical or genomic inputs during validation. Less than 5% performance degradation was observed at 20% masking, confirming robustness under incomplete observation&#x2014;a common scenario in oncology practice.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Comparison of predictive performance and interpretability on two real-world multimodal oncology datasets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">Dataset</th>
<th valign="middle" align="center">AUROC</th>
<th valign="middle" align="center">F1 score</th>
<th valign="middle" align="center">Interpretability</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Early fusion MLP</td>
<td valign="middle" align="center">METABRIC</td>
<td valign="middle" align="center">0.772</td>
<td valign="middle" align="center">0.706</td>
<td valign="middle" align="center"><inline-formula>
<mml:math display="inline" id="im276"><mml:mo>&#xd7;</mml:mo></mml:math></inline-formula></td>
</tr>
<tr>
<td valign="middle" align="left">Late fusion transformer</td>
<td valign="middle" align="center">METABRIC</td>
<td valign="middle" align="center">0.793</td>
<td valign="middle" align="center">0.721</td>
<td valign="middle" align="center"><inline-formula>
<mml:math display="inline" id="im277"><mml:mo>&#xd7;</mml:mo></mml:math></inline-formula></td>
</tr>
<tr>
<td valign="middle" align="left">GBDT + imputation</td>
<td valign="middle" align="center">METABRIC</td>
<td valign="middle" align="center">0.781</td>
<td valign="middle" align="center">0.715</td>
<td valign="middle" align="center"><inline-formula>
<mml:math display="inline" id="im278"><mml:mo>&#xd7;</mml:mo></mml:math></inline-formula></td>
</tr>
<tr>
<td valign="middle" align="left"><bold>ResponseNet</bold></td>
<td valign="middle" align="center">METABRIC</td>
<td valign="middle" align="center"><bold>0.831</bold></td>
<td valign="middle" align="center"><bold>0.745</bold></td>
<td valign="middle" align="center">&#x2713;</td>
</tr>
<tr>
<td valign="middle" align="left">Early fusion MLP</td>
<td valign="middle" align="center">CAMELYON16</td>
<td valign="middle" align="center">0.748</td>
<td valign="middle" align="center">0.684</td>
<td valign="middle" align="center"><inline-formula>
<mml:math display="inline" id="im279"><mml:mo>&#xd7;</mml:mo></mml:math></inline-formula></td>
</tr>
<tr>
<td valign="middle" align="left">Late fusion transformer</td>
<td valign="middle" align="center">CAMELYON16</td>
<td valign="middle" align="center">0.765</td>
<td valign="middle" align="center">0.699</td>
<td valign="middle" align="center"><inline-formula>
<mml:math display="inline" id="im280"><mml:mo>&#xd7;</mml:mo></mml:math></inline-formula></td>
</tr>
<tr>
<td valign="middle" align="left">GBDT + imputation</td>
<td valign="middle" align="center">CAMELYON16</td>
<td valign="middle" align="center">0.753</td>
<td valign="middle" align="center">0.691</td>
<td valign="middle" align="center"><inline-formula>
<mml:math display="inline" id="im281"><mml:mo>&#xd7;</mml:mo></mml:math></inline-formula></td>
</tr>
<tr>
<td valign="middle" align="left"><bold>ResponseNet</bold></td>
<td valign="middle" align="center">CAMELYON16</td>
<td valign="middle" align="center"><bold>0.812</bold></td>
<td valign="middle" align="center"><bold>0.724</bold></td>
<td valign="middle" align="center">&#x2713;</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The values in bold refer to our method.</p></fn>
</table-wrap-foot>
</table-wrap>
<p>To provide a concrete demonstration of interpretability in a clinical context, a simulated case study is presented based on a breast cancer patient undergoing neoadjuvant chemotherapy. The model predicts response to standard HER2-targeted therapy and simulates a counterfactual scenario under combination therapy. As shown in <xref ref-type="fig" rid="f6"><bold>Figure&#xa0;6</bold></xref>, the left panel presents a histological attention map from the original slide, along with a predicted probability of response (0.82) and its evolution over time. The right panel illustrates the counterfactual simulation, in which the model estimates a higher disease-free survival probability (0.75) under combination therapy compared to 0.65 under the standard regimen. Additionally, attention-based interpretability highlights tumor regions most relevant to the model&#x2019;s prediction. These outputs demonstrate how model-driven counterfactual reasoning and spatial attention can support clinicians in exploring multiple treatment options and understanding underlying factors influencing predictions. Such visual and quantitative aids can be integrated into multidisciplinary workflows to enhance transparency and trust in AI-assisted decision-making.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Simulated decision support scenario for a breast cancer patient. Left: attention map and predicted response probability under factual treatment. Right: counterfactual simulation comparing disease-free survival probabilities under different therapies, with spatial attribution and projected trends.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1619994-g006.tif">
<alt-text content-type="machine-generated">Case Study 1 displays a heatmap for neoadjuvant chemotherapy response prediction with a probability of 0.82. A line graph shows response probability increasing over time with breast radiotherapy. The Counterfactual scenario shows a similar heatmap with an attention map and disease-free survival probabilities of 0.65 for standard therapy and 0.75 for combination therapy. A line graph indicates improved disease-free probability with combination therapy compared to standard HER2-targeted therapy over 12 months.</alt-text>
</graphic></fig>
<p>To enhance interpretability in clinically actionable formats, the model&#x2019;s outputs are further contextualized using visualization strategies tailored for medical professionals. Attention mechanisms are rendered not as standalone saliency maps, but as spatial overlays directly superimposed on histopathological images. These overlays highlight morphologically relevant tumor regions that contribute most significantly to model predictions, making them accessible to pathologists and oncologists accustomed to traditional slide examination. By preserving spatial continuity with native tissue structures, this form of visualization facilitates more intuitive interpretation than abstract heatmaps. Temporal interpretability is achieved through stratified response curves that simulate predicted outcomes over time under varying therapeutic scenarios&#x2014;for example, in the presented case study, the model generates survival-like trajectories under both standard HER2-targeted therapy and an alternative combination regimen. These trajectory curves not only illustrate predicted differences in disease-free progression but also resemble conventional survival plots used in clinical oncology. This enables clinicians to visually compare risk profiles across treatment paths, supporting informed discussions about therapeutic trade-offs. These interpretability enhancements together shift the focus from model-centric explanation to clinician-facing insight. By embedding attention and prediction in domain-familiar representations&#x2014;namely, slide overlays and longitudinal outcome charts&#x2014;the framework enables practical decision support in oncology settings, bridging technical AI outputs with real-world clinical understanding.</p>
<p>The experimental evaluation focuses on two main aspects: the predictive performance of the model across multiple clinical datasets and its ability to provide interpretable insights into treatment outcomes. Predictive accuracy is measured by comparing forecasted clinical responses&#x2014;such as tumor progression or biomarker levels&#x2014;against ground truth values. Interpretability is assessed by examining visualizations such as attention maps, which highlight influential features or treatment time points that drive model predictions. The framework also supports counterfactual reasoning, enabling simulation of hypothetical outcomes under unobserved treatment scenarios. This capability is particularly relevant for exploring alternative therapeutic strategies and assessing individualized treatment effects. Results are reported on several benchmark datasets and compared against existing baseline models. The method demonstrates superior predictive performance while maintaining interpretability. Attention-based visual outputs and counterfactual predictions provide meaningful explanations, which may support informed decision-making in real-world clinical contexts.</p>
</sec>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusions and future work</title>
<p>In this study, we aimed to address a pivotal challenge in precision oncology: predicting breast cancer treatment response and long-term prognosis using AI. Traditional models often fail to handle the temporal complexity and multimodal nature of clinical data. To overcome this, we proposed an innovative, dynamics-aware deep learning framework centered around a novel architecture, ResponseNet. This model captures both short- and long-term patient response dynamics through multi-level sequence encoding and latent stochastic inference. Complementing this, we introduced two key components: a symbolic treatment abstraction mechanism to ensure pharmacological consistency and an adaptive knowledge infusion (AKI) strategy to integrate clinical expertise via ontologies and treatment guidelines. Experiments conducted on real-world breast cancer datasets confirmed our model&#x2019;s superiority over existing baselines in predicting treatment outcomes and stratifying survival risks. Notably, our approach balances predictive power with clinical interpretability&#x2014;an essential criterion for deployment in healthcare settings.</p>
<p>Despite promising results, two main limitations remain. A model&#x2019;s performance could be influenced by the quality and completeness of clinical data, especially in institutions with less structured electronic health records. Addressing this will require incorporating advanced imputation or semi-supervised techniques to better manage missing values. While AKI allows integration of domain knowledge, its current implementation may underutilize evolving, real-time clinical evidence and patient-specific nuance. Future work should explore dynamic knowledge graphs and continual learning mechanisms to enhance adaptability and relevance in fast-changing clinical environments. Overall, our study lays a foundation for intelligent, interpretable systems that support clinicians in personalizing breast cancer care.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material. Further inquiries can be directed to the corresponding author.</p></sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>BW: Writing &#x2013; original draft, Writing &#x2013; review &amp; editing, Methodology, Supervision, Project administration, Validation, Resources, Visualization. SC: Writing &#x2013; original draft, Writing &#x2013; review &amp; editing, Data curation, Conceptualization, Funding acquisition, Software. WL: Writing &#x2013; original draft, Writing &#x2013; review &amp; editing.</p></sec>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The author declares that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p></sec>
<sec id="s10" sec-type="correction-statement">
<title>Correction note</title>
<p>A correction has been made to this article. Details can be found at: <ext-link xlink:href="https://doi.org/10.3389/fonc.2025.1741682" ext-link-type="uri">10.3389/fonc.2025.1741682</ext-link>.</p></sec>
<sec id="s11" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p></sec>
<sec id="s12" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors&#xa0;and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p></sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Hong</surname> <given-names>D</given-names></name>
<name><surname>Gao</surname> <given-names>L</given-names></name>
<name><surname>Yao</surname> <given-names>J</given-names></name>
<name><surname>Zhang</surname> <given-names>B</given-names></name>
<name><surname>Plaza</surname> <given-names>A</given-names></name>
<name><surname>Chanussot</surname> <given-names>J</given-names></name>
</person-group>. 
<article-title>Graph convolutional networks for hyperspectral image classification</article-title>. <source>IEEE Trans Geosci Remote Sens</source>. (<year>2020</year>) <volume>59</volume>:<page-range>5966&#x2013;78</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TGRS.2020.3015157</pub-id>
</mixed-citation>
</ref>
<ref id="B2">
<label>2</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Yang</surname> <given-names>J</given-names></name>
<name><surname>Shi</surname> <given-names>R</given-names></name>
<name><surname>Wei</surname> <given-names>D</given-names></name>
<name><surname>Liu</surname> <given-names>Z</given-names></name>
<name><surname>Zhao</surname> <given-names>L</given-names></name>
<name><surname>Ke</surname> <given-names>B</given-names></name>
<etal/>
</person-group>. 
<article-title>Medmnist v2 - a large-scale lightweight benchmark for 2d and 3d biomedical image classification</article-title>. <source>Sci Data</source>. (<year>2021</year>) <volume>2</volume>., PMID: <pub-id pub-id-type="pmid">36658144</pub-id>
</mixed-citation>
</ref>
<ref id="B3">
<label>3</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Sun</surname> <given-names>L</given-names></name>
<name><surname>Zhao</surname> <given-names>G</given-names></name>
<name><surname>Zheng</surname> <given-names>Y</given-names></name>
<name><surname>Wu</surname> <given-names>Z</given-names></name>
</person-group>. 
<article-title>Spectral&#x2013;spatial feature tokenization transformer for hyperspectral image classification</article-title>. <source>IEEE Trans Geosci Remote Sens</source>. (<year>2022</year>) <volume>60</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TGRS.2022.3144158</pub-id>
</mixed-citation>
</ref>
<ref id="B4">
<label>4</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Chen</surname> <given-names>C-F</given-names></name>
<name><surname>Fan</surname> <given-names>Q</given-names></name>
<name><surname>Panda</surname> <given-names>R</given-names></name>
</person-group>. 
<article-title>Crossvit: Cross-attention multi-scale vision transformer for image classification</article-title>. <source>IEEE Int Conf Comput Vision</source>. (<year>2021</year>) <page-range>357&#x2013;66</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICCV48922.2021.00041</pub-id>
</mixed-citation>
</ref>
<ref id="B5">
<label>5</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Maur&#xed;cio</surname> <given-names>J</given-names></name>
<name><surname>Domingues</surname> <given-names>I</given-names></name>
<name><surname>Bernardino</surname> <given-names>J</given-names></name>
</person-group>. 
<article-title>Comparing vision transformers and convolutional neural networks for image classification: A literature review</article-title>. <source>Appl Sci</source>. (<year>2023</year>) <volume>13</volume>:<page-range>5521</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/app13095521</pub-id>
</mixed-citation>
</ref>
<ref id="B6">
<label>6</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Rao</surname> <given-names>Y</given-names></name>
<name><surname>Zhao</surname> <given-names>W</given-names></name>
<name><surname>Zhu</surname> <given-names>Z</given-names></name>
<name><surname>Lu</surname> <given-names>J</given-names></name>
<name><surname>Zhou</surname> <given-names>J</given-names></name>
</person-group>. 
<article-title>Global filter networks for image classification</article-title>. <source>Neural Inf Process Syst</source>. (<year>2021</year>).
</mixed-citation>
</ref>
<ref id="B7">
<label>7</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Hong</surname> <given-names>D</given-names></name>
<name><surname>Han</surname> <given-names>Z</given-names></name>
<name><surname>Yao</surname> <given-names>J</given-names></name>
<name><surname>Gao</surname> <given-names>L</given-names></name>
<name><surname>Zhang</surname> <given-names>B</given-names></name>
<name><surname>Plaza</surname> <given-names>A</given-names></name>
<etal/>
</person-group>. 
<article-title>Spectralformer: Rethinking hyperspectral image classification with transformers</article-title>. <source>IEEE Trans Geosci Remote Sens</source>. (<year>2021</year>) <volume>60</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TGRS.2021.3130716</pub-id>
</mixed-citation>
</ref>
<ref id="B8">
<label>8</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Touvron</surname> <given-names>H</given-names></name>
<name><surname>Bojanowski</surname> <given-names>P</given-names></name>
<name><surname>Caron</surname> <given-names>M</given-names></name>
<name><surname>Cord</surname> <given-names>M</given-names></name>
<name><surname>El-Nouby</surname> <given-names>A</given-names></name>
<name><surname>Grave</surname> <given-names>E</given-names></name>
<etal/>
</person-group>. 
<article-title>Resmlp: Feedforward networks for image classification with data-efficient training</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. (<year>2021</year>) <volume>45</volume>:<page-range>5314&#x2013;21</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.2022.3206148</pub-id>, PMID: <pub-id pub-id-type="pmid">36094972</pub-id>
</mixed-citation>
</ref>
<ref id="B9">
<label>9</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Mai</surname> <given-names>Z</given-names></name>
<name><surname>Li</surname> <given-names>R</given-names></name>
<name><surname>Jeong</surname> <given-names>J</given-names></name>
<name><surname>Quispe</surname> <given-names>D</given-names></name>
<name><surname>Kim</surname> <given-names>HJ</given-names></name>
<name><surname>Sanner</surname> <given-names>S</given-names></name>
</person-group>. 
<article-title>Online continual learning in image classification: An empirical survey</article-title>. <source>Neurocomputing</source>. (<year>2021</year>) <volume>469</volume>:<page-range>28:51</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.neucom.2021.10.021</pub-id>
</mixed-citation>
</ref>
<ref id="B10">
<label>10</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Wang</surname> <given-names>X</given-names></name>
<name><surname>Yang</surname> <given-names>S</given-names></name>
<name><surname>Zhang</surname> <given-names>J</given-names></name>
<name><surname>Wang</surname> <given-names>M</given-names></name>
<name><surname>Zhang</surname> <given-names>J</given-names></name>
<name><surname>Yang</surname> <given-names>W</given-names></name>
<etal/>
</person-group>. 
<article-title>Transformer-based unsupervised contrastive learning for histopathological image classification</article-title>. <source>Med Image Anal</source>. (<year>2022</year>) <volume>81</volume>:<page-range>102559</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.media.2022.102559</pub-id>, PMID: <pub-id pub-id-type="pmid">35952419</pub-id>
</mixed-citation>
</ref>
<ref id="B11">
<label>11</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Tian</surname> <given-names>Y</given-names></name>
<name><surname>Wang</surname> <given-names>Y</given-names></name>
<name><surname>Krishnan</surname> <given-names>D</given-names></name>
<name><surname>Tenenbaum</surname> <given-names>J</given-names></name>
<name><surname>Isola</surname> <given-names>P</given-names></name>
</person-group>. 
<article-title>Rethinking few-shot image classification: a good embedding is all you need</article-title>? <source>Eur Conf Comput Vision</source>. (<year>2020</year>) <volume>12359</volume>:<page-range>266&#x2013;82</page-range>.
</mixed-citation>
</ref>
<ref id="B12">
<label>12</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Azizi</surname> <given-names>S</given-names></name>
<name><surname>Mustafa</surname> <given-names>B</given-names></name>
<name><surname>Ryan</surname> <given-names>F</given-names></name>
<name><surname>Beaver</surname> <given-names>Z</given-names></name>
<name><surname>Freyberg</surname> <given-names>J</given-names></name>
<name><surname>Deaton</surname> <given-names>J</given-names></name>
<etal/>
</person-group>. 
<article-title>Big self-supervised models advance medical image classification</article-title>. <source>IEEE Int Conf Comput Vision</source>. (<year>2021</year>) <page-range>3478&#x2013;88</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICCV48922.2021.00346</pub-id>
</mixed-citation>
</ref>
<ref id="B13">
<label>13</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Li</surname> <given-names>B</given-names></name>
<name><surname>Li</surname> <given-names>Y</given-names></name>
<name><surname>Eliceiri</surname> <given-names>K</given-names></name>
</person-group>. 
<article-title>Dual-stream multiple instance learning network for whole slide image classification with self-supervised contrastive learning</article-title>. <source>Comput Vision Pattern Recognition</source>. (<year>2020</year>) <page-range>14318&#x2013;28</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR46437.2021.01409</pub-id>, PMID: <pub-id pub-id-type="pmid">35047230</pub-id>
</mixed-citation>
</ref>
<ref id="B14">
<label>14</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Radenkovic</surname> <given-names>S</given-names></name>
<name><surname>Konjevic</surname> <given-names>G</given-names></name>
<name><surname>Jurisic</surname> <given-names>V</given-names></name>
<name><surname>Karadzic</surname> <given-names>K</given-names></name>
<name><surname>Nikitovic</surname> <given-names>M</given-names></name>
<name><surname>Gopcevic</surname> <given-names>K</given-names></name>
</person-group>. 
<article-title>Values of mmp-2 and mmp-9 in tumor tissue of basal-like breast cancer patients</article-title>. <source>Cell Biochem biophysics</source>. (<year>2014</year>) <volume>68</volume>:<page-range>143&#x2013;52</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s12013-013-9701-x</pub-id>, PMID: <pub-id pub-id-type="pmid">23812723</pub-id>
</mixed-citation>
</ref>
<ref id="B15">
<label>15</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Radenkovic</surname> <given-names>S</given-names></name>
<name><surname>Milosevic</surname> <given-names>Z</given-names></name>
<name><surname>Konjevic</surname> <given-names>G</given-names></name>
<name><surname>Karadzic</surname> <given-names>K</given-names></name>
<name><surname>Rovcanin</surname> <given-names>B</given-names></name>
<name><surname>Buta</surname> <given-names>M</given-names></name>
<etal/>
</person-group>. 
<article-title>Lactate dehydrogenase, catalase, and superoxide dismutase in tumor tissue of breast cancer patients in respect to mammographic findings</article-title>. <source>Cell Biochem biophysics</source>. (<year>2013</year>) <volume>66</volume>:<page-range>287&#x2013;95</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s12013-012-9482-7</pub-id>, PMID: <pub-id pub-id-type="pmid">23197387</pub-id>
</mixed-citation>
</ref>
<ref id="B16">
<label>16</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Jurisic</surname> <given-names>V</given-names></name>
<name><surname>Radenkovic</surname> <given-names>S</given-names></name>
<name><surname>Konjevic</surname> <given-names>G</given-names></name>
</person-group>. 
<article-title>The actual role of ldh as tumor marker, biochemical and clinical aspects</article-title>. <source>Adv Cancer biomarkers: Biochem to clinic Crit revision</source>. (<year>2015</year>) <volume>115&#x2013;124</volume>., PMID: <pub-id pub-id-type="pmid">26530363</pub-id>
</mixed-citation>
</ref>
<ref id="B17">
<label>17</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Radenkovic</surname> <given-names>S</given-names></name>
<name><surname>Konjevic</surname> <given-names>G</given-names></name>
<name><surname>Isakovic</surname> <given-names>A</given-names></name>
<name><surname>Stevanovic</surname> <given-names>P</given-names></name>
<name><surname>Gopcevic</surname> <given-names>K</given-names></name>
<name><surname>Jurisic</surname> <given-names>V</given-names></name>
</person-group>. 
<article-title>Her2-positive breast cancer patients: correlation between mammographic and pathological findings</article-title>. <source>Radiat Prot dosimetry</source>. (<year>2014</year>) <volume>162</volume>:<page-range>125&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/rpd/ncu243</pub-id>, PMID: <pub-id pub-id-type="pmid">25063784</pub-id>
</mixed-citation>
</ref>
<ref id="B18">
<label>18</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Wang</surname> <given-names>Q</given-names></name>
<name><surname>Zhao</surname> <given-names>L</given-names></name>
<name><surname>Sun</surname> <given-names>J</given-names></name>
<name><surname>Xu</surname> <given-names>R</given-names></name>
</person-group>. 
<article-title>Emerging molecular mechanisms of resistance to targeted therapy in lung cancer</article-title>. <source>Front Oncol</source>. (<year>2025</year>) <volume>15</volume>:<elocation-id>1540195</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fonc.2025.1540195</pub-id>, PMID: <pub-id pub-id-type="pmid">40352592</pub-id>
</mixed-citation>
</ref>
<ref id="B19">
<label>19</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Zhou</surname> <given-names>L</given-names></name>
<name><surname>Tang</surname> <given-names>M</given-names></name>
<name><surname>Huang</surname> <given-names>Y</given-names></name>
<name><surname>Liu</surname> <given-names>Q</given-names></name>
</person-group>. 
<article-title>Immunometabolic remodeling in colorectal cancer: Progress and therapeutic perspectives</article-title>. <source>Front Oncol</source>. (<year>2025</year>) <volume>15</volume>:<elocation-id>1555369</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fonc.2025.1555369</pub-id>, PMID: <pub-id pub-id-type="pmid">40342817</pub-id>
</mixed-citation>
</ref>
<ref id="B20">
<label>20</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Coudray</surname> <given-names>N</given-names></name>
<name><surname>Ocampo</surname> <given-names>PS</given-names></name>
<name><surname>Sakellaropoulos</surname> <given-names>T</given-names></name>
<name><surname>Narula</surname> <given-names>N</given-names></name>
<name><surname>Snuderl</surname> <given-names>M</given-names></name>
<name><surname>Feny&#xf6;</surname> <given-names>D</given-names></name>
<etal/>
</person-group>. 
<article-title>Classification and mutation prediction from non&#x2013;small cell lung cancer histopathology images using deep learning</article-title>. <source>Nat. Med.</source> (<year>2018</year>) <volume>24</volume>:<page-range>1559&#x2013;67</page-range>., PMID: <pub-id pub-id-type="pmid">30224757</pub-id>
</mixed-citation>
</ref>
<ref id="B21">
<label>21</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Bhojanapalli</surname> <given-names>S</given-names></name>
<name><surname>Chakrabarti</surname> <given-names>A</given-names></name>
<name><surname>Glasner</surname> <given-names>D</given-names></name>
<name><surname>Li</surname> <given-names>D</given-names></name>
<name><surname>Unterthiner</surname> <given-names>T</given-names></name>
<name><surname>Veit</surname> <given-names>A</given-names></name>
</person-group>. 
<article-title>Understanding robustness of transformers for image classification</article-title>. <source>IEEE Int Conf Comput Vision</source>. (<year>2021</year>) <page-range>10231&#x2013;41</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICCV48922.2021.01007</pub-id>
</mixed-citation>
</ref>
<ref id="B22">
<label>22</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Kim</surname> <given-names>HE</given-names></name>
<name><surname>Cosa-Linan</surname> <given-names>A</given-names></name>
<name><surname>Santhanam</surname> <given-names>N</given-names></name>
<name><surname>Jannesari</surname> <given-names>M</given-names></name>
<name><surname>Maros</surname> <given-names>M</given-names></name>
<name><surname>Ganslandt</surname> <given-names>T</given-names></name>
</person-group>. 
<article-title>Transfer learning for medical image classification: a literature review</article-title>. <source>BMC Med Imaging</source>. (<year>2022</year>) <volume>22</volume>.
</mixed-citation>
</ref>
<ref id="B23">
<label>23</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Zhang</surname> <given-names>C</given-names></name>
<name><surname>Cai</surname> <given-names>Y</given-names></name>
<name><surname>Lin</surname> <given-names>G</given-names></name>
<name><surname>Shen</surname> <given-names>C</given-names></name>
</person-group>. 
<article-title>Deepemd: Few-shot image classification with differentiable earth mover&#x2019;s distance and structured classifiers</article-title>. <source>Comput Vision Pattern Recognition</source>. (<year>2020</year>) <page-range>12203&#x2013;13</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR42600.2020</pub-id>
</mixed-citation>
</ref>
<ref id="B24">
<label>24</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Roy</surname> <given-names>SK</given-names></name>
<name><surname>Deria</surname> <given-names>A</given-names></name>
<name><surname>Hong</surname> <given-names>D</given-names></name>
<name><surname>Rasti</surname> <given-names>B</given-names></name>
<name><surname>Plaza</surname> <given-names>A</given-names></name>
<name><surname>Chanussot</surname> <given-names>J</given-names></name>
</person-group>. 
<article-title>Multimodal fusion transformer for remote sensing image classification</article-title>. <source>IEEE Trans Geosci Remote Sens</source>. (<year>2022</year>) <volume>61</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TGRS.2023.3286826</pub-id>
</mixed-citation>
</ref>
<ref id="B25">
<label>25</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Zhu</surname> <given-names>Y</given-names></name>
<name><surname>Zhuang</surname> <given-names>F</given-names></name>
<name><surname>Wang</surname> <given-names>J</given-names></name>
<name><surname>Ke</surname> <given-names>G</given-names></name>
<name><surname>Chen</surname> <given-names>J</given-names></name>
<name><surname>Bian</surname> <given-names>J</given-names></name>
<etal/>
</person-group>. 
<article-title>Deep subdomain adaptation network for image classification</article-title>. In: <source><italic>IEEE Transactions on Neural Networks and Learning Systems</italic></source> (<year>2020</year>) <volume>32</volume>:<page-range>1713&#x2013;22</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TNNLS.2020.2988928</pub-id>, PMID: <pub-id pub-id-type="pmid">32365037</pub-id>
</mixed-citation>
</ref>
<ref id="B26">
<label>26</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Shehata</surname> <given-names>M</given-names></name>
<name><surname>Abouelkheir</surname> <given-names>RT</given-names></name>
<name><surname>Gayhart</surname> <given-names>M</given-names></name>
<name><surname>Van Bogaert</surname> <given-names>E</given-names></name>
<name><surname>Abou El-Ghar</surname> <given-names>M</given-names></name>
<name><surname>Dwyer</surname> <given-names>AC</given-names></name>
<etal/>
</person-group>. 
<article-title>Role of ai and radiomic markers in early diagnosis of renal cancer and clinical outcome prediction: a brief review</article-title>. <source>Cancers</source>. (<year>2023</year>) <volume>15</volume>:<fpage>2835</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/cancers15102835</pub-id>, PMID: <pub-id pub-id-type="pmid">37345172</pub-id>
</mixed-citation>
</ref>
<ref id="B27">
<label>27</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Li</surname> <given-names>Y</given-names></name>
<name><surname>Zhang</surname> <given-names>W</given-names></name>
<name><surname>Chen</surname> <given-names>H</given-names></name>
<name><surname>Liu</surname> <given-names>M</given-names></name>
</person-group>. 
<article-title>Hepatocellular carcinoma: Novel insights into tumor microenvironment and therapeutic targets</article-title>. <source>Front Oncol</source>. (<year>2025</year>) <volume>15</volume>: doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fonc.2025</pub-id>
</mixed-citation>
</ref>
<ref id="B28">
<label>28</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Chen</surname> <given-names>L</given-names></name>
<name><surname>Li</surname> <given-names>S</given-names></name>
<name><surname>Bai</surname> <given-names>Q</given-names></name>
<name><surname>Yang</surname> <given-names>J</given-names></name>
<name><surname>Jiang</surname> <given-names>S</given-names></name>
<name><surname>Miao</surname> <given-names>Y</given-names></name>
</person-group>. 
<article-title>Review of image classification algorithms based on convolutional neural networks</article-title>. <source>Remote Sens</source>. (<year>2021</year>) <volume>13</volume>:<page-range>4712</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs13224712</pub-id>
</mixed-citation>
</ref>
<ref id="B29">
<label>29</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Ashtiani</surname> <given-names>F</given-names></name>
<name><surname>Geers</surname> <given-names>AJ</given-names></name>
<name><surname>Aflatouni</surname> <given-names>F</given-names></name>
</person-group>. 
<article-title>An on-chip photonic deep neural network for image classification</article-title>. <source>Nature</source>. (<year>2021</year>) <volume>606</volume>:<page-range>501&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41586-022-04714-0</pub-id>, PMID: <pub-id pub-id-type="pmid">35650432</pub-id>
</mixed-citation>
</ref>
<ref id="B30">
<label>30</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Masana</surname> <given-names>M</given-names></name>
<name><surname>Liu</surname> <given-names>X</given-names></name>
<name><surname>Twardowski</surname> <given-names>B</given-names></name>
<name><surname>Menta</surname> <given-names>M</given-names></name>
<name><surname>Bagdanov</surname> <given-names>AD</given-names></name>
<name><surname>van de Weijer</surname> <given-names>J</given-names></name>
</person-group>. 
<article-title>Class-incremental learning: Survey and performance evaluation on image classification</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. (<year>2020</year>) <volume>45</volume>:<page-range>5513&#x2013;33</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.2022.3213473</pub-id>., PMID: <pub-id pub-id-type="pmid">36215375</pub-id>
</mixed-citation>
</ref>
<ref id="B31">
<label>31</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Dai</surname> <given-names>Y</given-names></name>
<name><surname>Gao</surname> <given-names>Y</given-names></name>
</person-group>. 
<article-title>Transmed: Transformers advance multi-modal medical image classification</article-title>. <source>Diagnostics</source>. (<year>2021</year>) <volume>11</volume>:<page-range>1384</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/diagnostics11081384</pub-id>, PMID: <pub-id pub-id-type="pmid">34441318</pub-id>
</mixed-citation>
</ref>
<ref id="B32">
<label>32</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Sheykhmousa</surname> <given-names>M</given-names></name>
<name><surname>Mahdianpari</surname> <given-names>M</given-names></name>
<name><surname>Ghanbari</surname> <given-names>H</given-names></name>
<name><surname>Mohammadimanesh</surname> <given-names>F</given-names></name>
<name><surname>Ghamisi</surname> <given-names>P</given-names></name>
<name><surname>Homayouni</surname> <given-names>S</given-names></name>
</person-group>. 
<article-title>Support vector machine versus random forest for remote sensing image classification: A meta-analysis and systematic review</article-title>. <source>IEEE J Selected Topics Appl Earth Observations Remote Sens</source>. (<year>2020</year>) <volume>13</volume>:<page-range>6308&#x2013;25</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/JSTARS.2020.3026724</pub-id>
</mixed-citation>
</ref>
<ref id="B33">
<label>33</label>
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name><surname>Mascarenhas</surname> <given-names>S</given-names></name>
<name><surname>l Agarwal</surname> <given-names>M</given-names></name>
</person-group>. 
<article-title>A comparison between vgg16, vgg19 and resnet50 architecture frameworks for image classification</article-title>, in: <conf-name>2021 International Conference on Disruptive Technologies for Multi-Disciplinary Research and Applications (CENTCON)</conf-name>: 
<publisher-name>IEEE</publisher-name> (<year>2021</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CENTCON52345.2021.9687944,.</pub-id>
</mixed-citation>
</ref>
<ref id="B34">
<label>34</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Shehata</surname> <given-names>M</given-names></name>
<name><surname>Elhosseini</surname> <given-names>M</given-names></name>
</person-group>. <source>Charting new frontiers: Insights and future directions in ml and dl for image processing</source>. (<publisher-loc>Switzerland</publisher-loc>: 
<publisher-name>MPDI</publisher-name>) (<year>2024</year>) <volume>13</volume>:<page-range>1345</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/electronics13071345</pub-id>
</mixed-citation>
</ref>
<ref id="B35">
<label>35</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Shehata</surname> <given-names>M</given-names></name>
<name><surname>Abouelkheir</surname> <given-names>RT</given-names></name>
<name><surname>Elhosseini</surname> <given-names>M</given-names></name>
</person-group>. <source>New diagnostic perspectives in urogenital radiology. Sec. Nephrology</source> (<year>2023</year>) <volume>10</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fmed.2023.1280300</pub-id>, PMID: <pub-id pub-id-type="pmid">38053616</pub-id>
</mixed-citation>
</ref>
<ref id="B36">
<label>36</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Zhang</surname> <given-names>Y</given-names></name>
<name><surname>Li</surname> <given-names>W</given-names></name>
<name><surname>Sun</surname> <given-names>W</given-names></name>
<name><surname>Tao</surname> <given-names>R</given-names></name>
<name><surname>Du</surname> <given-names>Q</given-names></name>
</person-group>. 
<article-title>Single-source domain expansion network for cross-scene hyperspectral image classification</article-title>. <source>IEEE Trans Image Process</source>. (<year>2022</year>) <volume>32</volume>:<page-range>1498&#x2013;1512</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIP.2023.3243853</pub-id>, PMID: <pub-id pub-id-type="pmid">37027628</pub-id>
</mixed-citation>
</ref>
<ref id="B37">
<label>37</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Taori</surname> <given-names>R</given-names></name>
<name><surname>Dave</surname> <given-names>A</given-names></name>
<name><surname>Shankar</surname> <given-names>V</given-names></name>
<name><surname>Carlini</surname> <given-names>N</given-names></name>
<name><surname>Recht</surname> <given-names>B</given-names></name>
<name><surname>Schmidt</surname> <given-names>L</given-names></name>
</person-group>. 
<article-title>Measuring robustness to natural distribution shifts in image classification</article-title>. <source>Neural Inf Process Syst</source>. (<year>2020</year>).
</mixed-citation>
</ref>
<ref id="B38">
<label>38</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Dong</surname> <given-names>H</given-names></name>
<name><surname>Zhang</surname> <given-names>L</given-names></name>
<name><surname>Zou</surname> <given-names>B</given-names></name>
</person-group>. 
<article-title>Exploring vision transformers for polarimetric sar image classification</article-title>. <source>IEEE Trans Geosci Remote Sens</source>. (<year>2022</year>) <volume>30</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TGRS.2021.3137383</pub-id>
</mixed-citation>
</ref>
<ref id="B39">
<label>39</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Peng</surname> <given-names>J</given-names></name>
<name><surname>Huang</surname> <given-names>Y</given-names></name>
<name><surname>SUN</surname> <given-names>W</given-names></name>
<name><surname>Chen</surname> <given-names>N</given-names></name>
<name><surname>Ning</surname> <given-names>Y</given-names></name>
<name><surname>Du</surname> <given-names>Q</given-names></name>
</person-group>. 
<article-title>Domain adaptation in remote sensing image classification: A survey</article-title>. <source>IEEE J Selected Topics Appl Earth Observations Remote Sens</source>. (<year>2022</year>) <volume>15</volume>:<page-range>9842&#x2013;59</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/JSTARS.2022.3220875</pub-id>
</mixed-citation>
</ref>
<ref id="B40">
<label>40</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Bazi</surname> <given-names>Y</given-names></name>
<name><surname>Bashmal</surname> <given-names>L</given-names></name>
<name><surname>Rahhal</surname> <given-names>MMA</given-names></name>
<name><surname>Dayil</surname> <given-names>RA</given-names></name>
<name><surname>Ajlan</surname> <given-names>NA</given-names></name>
</person-group>. 
<article-title>Vision transformers for remote sensing image classification</article-title>. <source>Remote Sens</source>. (<year>2021</year>) <volume>13</volume>:<page-range>516</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs13030516</pub-id>
</mixed-citation>
</ref>
<ref id="B41">
<label>41</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Zheng</surname> <given-names>X</given-names></name>
<name><surname>Sun</surname> <given-names>H</given-names></name>
<name><surname>Lu</surname> <given-names>X</given-names></name>
<name><surname>Xie</surname> <given-names>W</given-names></name>
</person-group>. 
<article-title>Rotation-invariant attention network for hyperspectral image classification</article-title>. <source>IEEE Trans Image Process</source>. (<year>2022</year>) <volume>31</volume>:<page-range>4251&#x2013;65</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIP.2022.3177322</pub-id>, PMID: <pub-id pub-id-type="pmid">35635815</pub-id>
</mixed-citation>
</ref>
<ref id="B42">
<label>42</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Seo</surname> <given-names>H</given-names></name>
<name><surname>Brand</surname> <given-names>L</given-names></name>
<name><surname>Barco</surname> <given-names>LS</given-names></name>
<name><surname>Wang</surname> <given-names>H</given-names></name>
</person-group>. 
<article-title>Scaling multi-instance support vector machine to breast cancer detection on the breakhis dataset</article-title>. <source>Bioinformatics</source>. (<year>2022</year>) <volume>38</volume>:<fpage>i92</fpage>&#x2013;<lpage>i100</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btac267</pub-id>, PMID: <pub-id pub-id-type="pmid">35758811</pub-id>
</mixed-citation>
</ref>
<ref id="B43">
<label>43</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Lee</surname> <given-names>RS</given-names></name>
<name><surname>Dunnmon</surname> <given-names>JA</given-names></name>
<name><surname>He</surname> <given-names>A</given-names></name>
<name><surname>Tang</surname> <given-names>S</given-names></name>
<name><surname>Re</surname> <given-names>C</given-names></name>
<name><surname>Rubin</surname> <given-names>DL</given-names></name>
</person-group>. 
<article-title>Comparison of segmentation-free and segmentation-dependent computer-aided diagnosis of breast masses on a public mammography dataset</article-title>. <source>J Biomed Inf</source>. (<year>2021</year>) <volume>113</volume>:<fpage>103656</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jbi.2020.103656</pub-id>, PMID: <pub-id pub-id-type="pmid">33309994</pub-id>
</mixed-citation>
</ref>
<ref id="B44">
<label>44</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Rezaei</surname> <given-names>Z</given-names></name>
</person-group>. 
<article-title>A review on image-based approaches for breast cancer detection, segmentation, and classification</article-title>. <source>Expert Syst Appl</source>. (<year>2021</year>) <volume>182</volume>:<fpage>115204</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eswa.2021.115204</pub-id>
</mixed-citation>
</ref>
<ref id="B45">
<label>45</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Villareal</surname> <given-names>RJT</given-names></name>
<name><surname>Abu</surname> <given-names>PAR</given-names></name>
</person-group>. 
<article-title>Patch-based convolutional neural networks for tcga-brca breast cancer classification</article-title>. In: <source>Advances in visual computing: 16th international symposium, ISVC 2021, virtual event, october 4-6, 2021, proceedings, part II</source>. <publisher-loc>Cham</publisher-loc>: 
<publisher-name>Springer</publisher-name> (<year>2021</year>). p. <fpage>29</fpage>&#x2013;<lpage>40</lpage>.
</mixed-citation>
</ref>
<ref id="B46">
<label>46</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Elpeltagy</surname> <given-names>M</given-names></name>
<name><surname>Sallam</surname> <given-names>H</given-names></name>
</person-group>. 
<article-title>Automatic prediction of covid- 19 from chest images using modified resnet50</article-title>. <source>Multimedia Tools Appl</source>. (<year>2021</year>) <volume>80</volume>:<page-range>26451&#x2013;63</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11042-021-10783-6</pub-id>, PMID: <pub-id pub-id-type="pmid">33967592</pub-id>
</mixed-citation>
</ref>
<ref id="B47">
<label>47</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Hong</surname> <given-names>S</given-names></name>
<name><surname>Wu</surname> <given-names>J</given-names></name>
<name><surname>Zhu</surname> <given-names>L</given-names></name>
</person-group>. 
<article-title>A brain tumor classification algorithm based on vit-b/16</article-title>. In: <source>2024 36th chinese control and decision conference (CCDC)</source>. <publisher-loc>China</publisher-loc>: 
<publisher-name>IEEE</publisher-name> (<year>2024</year>). p. <page-range>3154&#x2013;9</page-range>.
</mixed-citation>
</ref>
<ref id="B48">
<label>48</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Preetha</surname> <given-names>R</given-names></name>
<name><surname>Priyadarsini</surname> <given-names>MJP</given-names></name>
<name><surname>Nisha</surname> <given-names>J</given-names></name>
</person-group>. 
<article-title>Automated brain tumor detection from magnetic resonance images using fine-tuned efficientnet-b4 convolutional neural network</article-title>. <source>IEEE Access</source>. (<year>2024</year>) <volume>12</volume>:<page-range>112181&#x2013;95</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2024.3442979</pub-id>
</mixed-citation>
</ref>
<ref id="B49">
<label>49</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Yu</surname> <given-names>W</given-names></name>
<name><surname>Zhou</surname> <given-names>P</given-names></name>
<name><surname>Yan</surname> <given-names>S</given-names></name>
<name><surname>Wang</surname> <given-names>X</given-names></name>
</person-group>. 
<article-title>Inceptionnext: When inception meets convnext</article-title>. In: <source>Proceedings of the IEEE/cvf conference on computer vision and pattern recognition</source> (<year>2024</year>) <page-range>5672&#x2013;83</page-range>.
</mixed-citation>
</ref>
<ref id="B50">
<label>50</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Mohandass</surname> <given-names>G</given-names></name>
<name><surname>Krishnan</surname> <given-names>GH</given-names></name>
<name><surname>Selvaraj</surname> <given-names>D</given-names></name>
<name><surname>Sridhathan</surname> <given-names>C</given-names></name>
</person-group>. 
<article-title>Lung cancer classification using optimized attention-based convolutional neural network with densenet-201 transfer learning model on ct image</article-title>. <source>Biomed Signal Process Control</source>. (<year>2024</year>) <volume>95</volume>:<fpage>106330</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.bspc.2024.106330</pub-id>
</mixed-citation>
</ref>
<ref id="B51">
<label>51</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Pandey</surname> <given-names>S</given-names></name>
<name><surname>Sindhuja</surname> <given-names>B</given-names></name>
<name><surname>Nagamanjularani</surname> <given-names>C</given-names></name>
<name><surname>Nagarajan</surname> <given-names>S</given-names></name>
</person-group>. 
<article-title>Exploring transfer learning techniques for flower recognition using cnn</article-title>. In: <source>Data science and security: proceedings of IDSCS 2022</source>. 
<publisher-name>Springer</publisher-name> (<year>2022</year>). p. <fpage>393</fpage>&#x2013;<lpage>401</lpage>.
</mixed-citation>
</ref>
</ref-list>
<fn-group>
<fn id="n1" fn-type="custom" custom-type="edited-by">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/770972">Mohamed Shehata</ext-link>, Midway College, United States</p></fn>
<fn id="n2" fn-type="custom" custom-type="reviewed-by">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/152842">Rashid Ibrahim Mehmood</ext-link>, Islamic University of Madinah, Saudi Arabia</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/286068">Vladimir Jurisic</ext-link>, University of Kragujevac, Serbia</p></fn>
</fn-group>
</back>
</article>