<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" article-type="research-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Med.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Medicine</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Med.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2296-858X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmed.2026.1751768</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Automated tumor regression grade assessment and survival prediction in esophageal cancer via weakly supervised multiple instance learning</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes"><name><surname>Liu</surname> <given-names>Zhengjin</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="author-notes" rid="fn0001"><sup>&#x2020;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3290047"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x0026; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
</contrib>
<contrib contrib-type="author" equal-contrib="yes"><name><surname>Zhao</surname> <given-names>Lin</given-names></name><xref ref-type="aff" rid="aff2"><sup>2</sup></xref><xref ref-type="author-notes" rid="fn0001"><sup>&#x2020;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
</contrib>
<contrib contrib-type="author" equal-contrib="yes"><name><surname>Zhao</surname> <given-names>Ziqing</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="author-notes" rid="fn0001"><sup>&#x2020;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2873344"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Funding acquisition" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author" equal-contrib="yes"><name><surname>Qin</surname> <given-names>Wenjuan</given-names></name><xref ref-type="aff" rid="aff3"><sup>3</sup></xref><xref ref-type="author-notes" rid="fn0001"><sup>&#x2020;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Funding acquisition" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
</contrib>
<contrib contrib-type="author" equal-contrib="yes"><name><surname>Zhong</surname> <given-names>Jun</given-names></name><xref ref-type="aff" rid="aff4"><sup>4</sup></xref><xref ref-type="author-notes" rid="fn0001"><sup>&#x2020;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
</contrib>
<contrib contrib-type="author"><name><surname>Lin</surname> <given-names>Fenglian</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
</contrib>
<contrib contrib-type="author"><name><surname>Lu</surname> <given-names>Meizhen</given-names></name><xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
</contrib>
<contrib contrib-type="author"><name><surname>Guo</surname> <given-names>Ruixiang</given-names></name><xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
</contrib>
<contrib contrib-type="author"><name><surname>Guo</surname> <given-names>Qunhuang</given-names></name><xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
</contrib>
<contrib contrib-type="author"><name><surname>Xu</surname> <given-names>Hui</given-names></name><xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
</contrib>
<contrib contrib-type="author" corresp="yes"><name><surname>Li</surname> <given-names>Shouguo</given-names></name><xref ref-type="aff" rid="aff3"><sup>3</sup></xref><xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
</contrib>
<contrib contrib-type="author" corresp="yes"><name><surname>Zheng</surname> <given-names>Hao</given-names></name><xref ref-type="aff" rid="aff6"><sup>6</sup></xref><xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author" corresp="yes"><name><surname>Lu</surname> <given-names>Haijie</given-names></name><xref ref-type="aff" rid="aff3"><sup>3</sup></xref><xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3257245"/>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label><institution>Department of Pathology, Zhongshan Hospital of Xiamen University, School of Medicine, Xiamen University</institution>, <city>Xiamen</city>, <country country="cn">China</country></aff>
<aff id="aff2"><label>2</label><institution>Department of Computer Science at School of Informatics, Xiamen University</institution>, <city>Xiamen</city>, <country country="cn">China</country></aff>
<aff id="aff3"><label>3</label><institution>Department of Radiation Oncology, Zhongshan Hospital of Xiamen University, School of Medicine, Xiamen University</institution>, <city>Xiamen</city>, <country country="cn">China</country></aff>
<aff id="aff4"><label>4</label><institution>Department of Radiology, Zhongshan Hospital of Xiamen University, School of Medicine, Xiamen University</institution>, <city>Xiamen</city>, <country country="cn">China</country></aff>
<aff id="aff5"><label>5</label><institution>Department of Health Medicine, Zhongshan Hospital of Xiamen University, School of Medicine, Xiamen University</institution>, <city>Xiamen</city>, <country country="cn">China</country></aff>
<aff id="aff6"><label>6</label><institution>Department of Thoracic Surgery, The First Affiliated Hospital of Anhui Medical University</institution>, <city>Hefei</city>, <country country="cn">China</country></aff>
<author-notes>
<corresp id="c001"><label>&#x002A;</label>Correspondence: Shouguo Li, <email xlink:href="mailto:xmliyx@126.com">xmliyx@126.com</email>; Hao Zheng, <email xlink:href="mailto:pojunayfy@gmail.com">pojunayfy@gmail.com</email>; Haijie Lu, <email xlink:href="mailto:luhaijie@xmu.edu.cn">luhaijie@xmu.edu.cn</email></corresp>
<fn fn-type="equal" id="fn0001">
<label>&#x2020;</label>
<p>These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2026-02-02">
<day>02</day>
<month>02</month>
<year>2026</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2026</year>
</pub-date>
<volume>13</volume>
<elocation-id>1751768</elocation-id>
<history>
<date date-type="received">
<day>22</day>
<month>11</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>07</day>
<month>01</month>
<year>2026</year>
</date>
<date date-type="accepted">
<day>13</day>
<month>01</month>
<year>2026</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2026 Liu, Zhao, Zhao, Qin, Zhong, Lin, Lu, Guo, Guo, Xu, Li, Zheng and Lu.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder>Liu, Zhao, Zhao, Qin, Zhong, Lin, Lu, Guo, Guo, Xu, Li, Zheng and Lu</copyright-holder>
<license>
<ali:license_ref start_date="2026-02-02">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Esophageal cancer remains a major global health burden and is among the leading causes of cancer-related deaths. Accurate evaluation of tumor regression grade (TRG) after neoadjuvant therapy is essential for assessing treatment response and guiding postoperative management. However, conventional TRG assessment relies heavily on subjective histopathological assessments, leading to considerable inter-observer variability and limited reproducibility. We aimed to develop an objective and automated TRG assessment framework using artificial intelligence for digital pathology.</p>
</sec>
<sec>
<title>Methods</title>
<p>A retrospective analysis was conducted on 157 patients with esophageal cancer and 1,298 hematoxylin and eosin-stained whole-slide images. Three slide-level pathology foundation models and three multiple instance learning methods were evaluated within a patient-level multiple instance learning framework, enabling weakly supervised TRG prediction based solely on patient-level labels.</p>
</sec>
<sec>
<title>Results</title>
<p>The proposed framework achieved a classification accuracy of 82.7% and demonstrated strong agreement with manual pathologist grading. Notably, artificial intelligence-derived TRG score provided superior prognostic stratification compared with conventional assessments, showing significant associations with progression-free and overall survival.</p>
</sec>
<sec>
<title>Discussion</title>
<p>This study presents a foundation-model-driven, patient-level multiple instance learning framework for automated evaluation of TRG in esophageal cancer. This approach offers a standardized, reproducible, and clinically interpretable solution that reduces the workload of pathologists and improves prognostic precision. These findings highlight the potential of weakly supervised AI pathology in advancing personalized treatment assessment and decision-making in digital oncology.</p>
</sec>
</abstract>
<kwd-group>
<kwd>artificial intelligence</kwd>
<kwd>multiple instance learning</kwd>
<kwd>neoadjuvant therapy</kwd>
<kwd>pathology foundation model</kwd>
<kwd>tumor regression grade</kwd>
</kwd-group>
<funding-group>
<award-group id="gs1">
<funding-source id="sp1">
<institution-wrap>
<institution>Natural Science Foundation of Fujian Province of China</institution>
</institution-wrap>
</funding-source>
<award-id rid="sp1">2022J0113440</award-id>
<award-id rid="sp1">2021J011332</award-id>
</award-group>
<award-group id="gs2">
<funding-source id="sp2">
<institution-wrap>
<institution>Fujian Provincial Health Technology Project</institution>
</institution-wrap>
</funding-source>
<award-id rid="sp2">2021CXB023</award-id>
</award-group>
<funding-statement>The author(s) declared that financial support was received for this work and/or its publication. This study was sponsored by Fujian Provincial Health Technology Project (No. 2021CXB023 to HL), Natural Science Foundation of Fujian Province of China (Nos. 2021J011332 to WQ and 2022J0113440 to ZZ).</funding-statement>
</funding-group>
<counts>
<fig-count count="5"/>
<table-count count="3"/>
<equation-count count="0"/>
<ref-count count="29"/>
<page-count count="12"/>
<word-count count="7862"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Pathology</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>Esophageal cancer (EC) is a prevalent malignant tumor of the digestive tract, ranking seventh in incidence and sixth in mortality globally (<xref ref-type="bibr" rid="ref1">1</xref>). In clinical practice, most patients with EC are diagnosed at locally advanced stages. Because of its poor prognosis, multimodal treatment strategies combining neoadjuvant therapy (NAT) with surgical resection have recently become the standard approach, replacing surgical treatment alone (<xref ref-type="bibr" rid="ref2 ref3 ref4 ref5">2&#x2013;5</xref>). The primary objective of NAT is to improve the complete resection rate of tumors and eliminate potential micrometastases through preoperative downstaging, significantly improving patient prognosis (<xref ref-type="bibr" rid="ref6">6</xref>, <xref ref-type="bibr" rid="ref7">7</xref>). Accurate assessment of pathological response after NAT is essential for evaluating therapeutic efficacy and determining optimal postoperative management strategies. Tumor regression grade (TRG) is a crucial histopathological indicator of NAT efficacy and for informing clinical decision-making (<xref ref-type="bibr" rid="ref8">8</xref>, <xref ref-type="bibr" rid="ref9">9</xref>). However, traditional TRG assessments are subjective and rely on visual estimates of the proportion of residual tumor cells relative to fibrosis, or the percentage of residual tumor area within whole-slide images (WSIs) (<xref ref-type="bibr" rid="ref10">10</xref>). Such manual evaluations are associated with significant inter-observer variability and limited reproducibility. Moreover, pathologists typically need to carefully review dozens, or even hundreds, of histopathological slides to complete the TRG assessment of a single patient with EC, which is extremely time-consuming and imposes a heavy burden on increasingly strained pathologic resources. Therefore, there is an urgent need to develop an objective, accurate, efficient, and reproducible automated tool for evaluating TRG in EC.</p>
<p>The integration of whole-slide imaging with artificial intelligence (AI), particularly deep learning, has recently shown notable potential in various tumor diagnostic and prognostic tasks (<xref ref-type="bibr" rid="ref11 ref12 ref13 ref14 ref15 ref16">11&#x2013;16</xref>). In the field of TRG, several studies have explored the feasibility of AI-assisted assessment following NAT. Tolkach et al. (<xref ref-type="bibr" rid="ref17">17</xref>) developed an AI system that can be used to automatically detect tumor regions and grade histological regression in esophageal adenocarcinoma, achieving a strong correlation with clinical outcomes and performance comparable to expert pathologists&#x2019; assessments. Wang et al. (<xref ref-type="bibr" rid="ref18">18</xref>) further proposed a semi-supervised knowledge distillation framework to evaluate the pathological response of esophageal squamous cell carcinoma after NAT, demonstrating a strong agreement with pathologists&#x2019; assessments of residual tumor percentage. However, their approach required manually labeled image patches and primarily focused on continuous residual tumor quantification rather than direct TRG score prediction.</p>
<p>To address these limitations, we developed an accurate and efficient weakly supervised AI framework for the automated assessment of TRG score after NAT in EC. Compared with the fully supervised methods adopted in previous studies, weakly supervised approaches, such as multiple instance learning (MIL), have emerged as powerful alternatives in computational pathology. As highlighted in a recent comprehensive survey by Waqas et al. (<xref ref-type="bibr" rid="ref19">19</xref>), MIL has gained significant traction in medical image analysis by drastically reducing the annotation burden while maintaining robust performance across diverse diagnostic tasks. This paradigm enables efficient model training without detailed pixel- or region-level annotations (<xref ref-type="bibr" rid="ref20 ref21 ref22">20&#x2013;22</xref>). Our framework relies solely on patient-level labels provided by pathologists, eliminating the need for precise tumor annotations or manual slide-level grading. Specifically, we first incorporated pathology foundation models into our workflow and systematically compared several state-of-the-art slide-level encoders, including CHIEF (<xref ref-type="bibr" rid="ref23">23</xref>), Prov-Gigapath (<xref ref-type="bibr" rid="ref24">24</xref>), and TITAN (<xref ref-type="bibr" rid="ref25">25</xref>) to identify the optimal feature extractor for WSI representation. Subsequently, we implemented an MIL-based architecture, in which each patient was treated as a &#x201C;bag&#x201D; comprising multiple WSIs from the same patient as &#x201C;instances.&#x201D; Next, several advanced attention-based MIL algorithms were evaluated to construct a robust patient-level TRG assessment model.</p>
<p>In this study, we demonstrated that AI-predicted TRG score at the patient level were highly consistent with manual pathologist grading across cross-validation folds. Notably, the AI-derived grade exhibited superior prognostic power compared with manual TRG score, indicating the model&#x2019;s capacity to capture clinically meaningful histological patterns beyond human perception. This framework effectively mitigates the challenges of subjectivity, limited reproducibility, and time-intensive evaluations inherent in traditional TRG assessments. Moreover, by providing enhanced prognostic stratification, it holds strong potential to inform personalized adjuvant treatment decisions for patients with EC, underscoring its clinical applicability and translational significance in the era of pathological AI.</p>
</sec>
<sec sec-type="materials|methods" id="sec2">
<label>2</label>
<title>Materials and methods</title>
<sec id="sec3">
<label>2.1</label>
<title>Study participants</title>
<p>Eligible patients were aged 18&#x2013;75&#x202F;years of either sex, with an Eastern Cooperative Oncology Group performance status of 0&#x2013;1. All cases were pathologically confirmed as esophageal squamous cell carcinoma, and the patients had radiologically and clinically resectable locally advanced disease that was treated with NAT. Only patients with complete clinicopathological and follow-up data were included in the analyses.</p>
<p>Patients were excluded if they had non-squamous histological subtypes of EC, severe systemic comorbidities that contraindicated standard treatment, or synchronous malignancies in other organs.</p>
</sec>
<sec id="sec4">
<label>2.2</label>
<title>Ethical approval</title>
<p>This study was conducted in accordance with the principles of the Declaration of Helsinki. Ethical approval and a waiver of informed consent were obtained from the Ethics Committee of Zhongshan Hospital Affiliated to Xiamen University.</p>
</sec>
<sec id="sec5">
<label>2.3</label>
<title>Observation indicators</title>
<p>Histopathological examination was conducted by two experienced pathologists who analyzed the hematoxylin and eosin (H&#x0026;E)-stained slides to evaluate the resection margins and lymph nodes. Tumor staging was based on the 8th edition of the American Joint Committee on Cancer staging system (<xref ref-type="bibr" rid="ref26">26</xref>). The tumor response to NAT was assessed using the College of American Pathologists (CAP) grading system (modified Ryan scheme) (<xref ref-type="bibr" rid="ref8">8</xref>). The specific category-to-criteria mapping is defined as follows: TRG 0 (no viable cancer cells); TRG 1 [single cells or rare small groups of cancer cells, &#x003C;10% viable residual tumor cells (VRTCs)]; TRG 2 (residual cancer cells outgrown by fibrosis, 10&#x2013;50% VRTCs); and TRG 3 (minimal or no tumor regression, &#x003E;50% VRTCs). Patients were categorized as good (TRG 0&#x2013;1) or poor (TRG 2&#x2013;3) responders. Overall survival (OS) was defined as the time from diagnosis to death, and progression-free survival (PFS) was defined as the time to tumor progression or death from any cause.</p>
</sec>
<sec id="sec6">
<label>2.4</label>
<title>Histopathology data collection and scoring</title>
<p>All hematoxylin and eosin (H&#x0026;E)-stained sections were digitally scanned and subjected to quality control. The TRG score of the patients was determined based on postoperative pathological assessment criteria and served as the primary source of the labels.</p>
<p>This study included 157 patients with EC who underwent NAT followed by surgical resection. H&#x0026;E-stained WSIs were retrospectively collected, yielding 1,298 digital pathology slides. Patient-level TRG labels were obtained from pathology reports, and the final TRG score was determined by integrating multiple slides per patient. Given that TRG is inherently a case-level construct and patient-level labels may not represent the histology of individual slides (e.g., a slide with no tumor cells from a TRG 2 patient), direct inheritance of labels is imprecise. Therefore, to enable the evaluation of slide-level feature extraction capabilities, an independent dataset of 120 WSIs was randomly sampled. Two experienced pathologists manually assigned a &#x201C;slide-level TRG&#x201D; to each WSI solely based on the proportion of residual tumor area observed on that specific slide, following the CAP criteria. These slide-specific annotations served as the ground truth for the technical benchmarking of foundation models.</p>
</sec>
<sec id="sec7">
<label>2.5</label>
<title>Foundation models for slide-level feature extraction</title>
<p>To extract discriminative features from the WSIs, we systematically evaluated three recently developed pathology foundation models: Prov-Gigapath, TITAN, and CHIEF, using 120 WSIs with slide-level TRG annotations.</p>
<p>Prov-Gigapath is a large-scale model trained on 170,000 WSIs. It integrates a tile encoder with a LongNet-based slide encoder to process gigapixel images. While it offers robustness through scale, its architecture is primarily patch-centric. TITAN, in contrast, adopts a multimodal framework (ConCH v1.5 encoder) designed to integrate images and language. Trained on WSIs paired with captions, it excels in cross-modal alignment and zero-shot predictions, focusing on semantic generalization rather than pure histological hierarchy.</p>
<p>CHIEF, however, introduces a hierarchical tokenization mechanism that allows the model to simultaneously capture fine-grained cellular morphology and broader tissue architecture. This architectural characteristic formed our primary <italic>a priori</italic> rationale for selecting CHIEF as the optimal backbone. Accurate TRG grading requires the simultaneous recognition of residual tumor cells (local features) and the surrounding fibrotic stroma (global context). CHIEF&#x2019;s ability to integrate these multi-scale features makes it theoretically more suitable for TRG assessment compared to the patch-centric approach of Prov-Gigapath or the language-aligned focus of TITAN.</p>
<p>Consequently, CHIEF was selected as the designated feature extractor, with the empirical benchmarking serving primarily to validate this theoretical hypothesis. The pre-trained backbone generates a 768-dimensional representation for each slide. For all experiments, features were extracted using these foundation models at 20&#x00D7; resolution and subsequently subjected to downstream analysis.</p>
<p>For each WSI, features were extracted using all three foundation models at 20&#x00D7; resolution and subsequently underwent further analysis.</p>
</sec>
<sec id="sec8">
<label>2.6</label>
<title>Feature visualization</title>
<p>To evaluate whether the learned embeddings reflected TRG-specific differences, we used nonlinear dimensionality reduction methods, namely t-distributed stochastic neighbor embedding and uniform manifold approximation and projection. These techniques project high-dimensional embeddings onto a two-dimensional space, enabling the qualitative visualization of class separation while preserving local neighborhood relationships and global data structures.</p>
</sec>
<sec id="sec9">
<label>2.7</label>
<title>Slide-level classification</title>
<p>A quantitative evaluation of slide-level classification was performed using support vector machines trained on embeddings from each foundation model. The support vector machine classifier was chosen for its proven effectiveness in high-dimensional feature spaces and robustness against overfitting. A radial basis function kernel was used to capture the nonlinear class boundaries, and the hyperparameters were optimized within each training fold. To ensure robust estimation of model generalization, five-fold cross-validation was applied, with training and testing sets partitioned at the slide level. The performance metrics included overall accuracy, area under the receiver operating characteristic curve (AUC), precision, recall, <italic>F</italic><sub>1</sub>-score, and confusion matrices.</p>
</sec>
<sec id="sec10">
<label>2.8</label>
<title>Patient-level TRG prediction</title>
<p>Following the comparative evaluation of the foundation models at the slide level, the best-performing foundation model was incorporated into a patient-level predictive framework. Because TRG labels were available only at the case level, we adopted an MIL strategy that treats each patient as a bag and the associated WSIs as instances within that bag. The bag inherits the patient-level TRG label, whereas the labels of the individual instances remain unknown. This approach is particularly suitable for weakly supervised pathology tasks, in which detailed annotations are difficult to obtain.</p>
<p>Three MIL variants are implemented and compared. First, attention-based MIL aggregates slide-level embeddings using an attention mechanism that assigns variable importance weights to each instance, highlighting the most informative slides for prediction. Second, adaptive cross-instance MIL (ACMIL) extends this framework by modeling dependencies among slides and dynamically adjusting the weighting scheme to account for inter-instance relationships. Third, the clustering-constrained attention MIL (CLAM) imposes an additional constraint that partitions instances within a bag into clusters representing positive and negative evidence, thereby enhancing the interpretability and robustness of the aggregated representation. In all three cases, CHIEF was used as the feature extraction backbone because it demonstrated the strongest discriminative power in the preliminary experiments.</p>
<p>Patient-level MIL models were trained and evaluated using five-fold cross-validation, with data split at the patient level to avoid information leakage across the folds. The evaluation metrics included overall accuracy, macro-<italic>F</italic><sub>1</sub> score, and AUC, with confusion matrices providing detailed insights into class-specific performance.</p>
</sec>
<sec id="sec11">
<label>2.9</label>
<title>Implementation details</title>
<p>For each patient, feature embeddings extracted from all available WSIs (ranging from 2 to 10 slides) were aggregated to form a single patient-level bag, enabling the MIL framework to capture intra-patient histological heterogeneity across different tissue sections. No padding or truncation was applied; all slides belonging to a patient were included during bag aggregation. Five-fold cross-validation was performed with patient-level splits to strictly prevent information leakage between training and test sets.</p>
<p>To address class imbalance between favorable (TRG 0&#x2013;1) and poor (TRG 2&#x2013;3) response groups, class-balanced resampling was applied at the patient (bag) level during training. Model optimization was conducted using the AdamW optimizer with an initial learning rate of 2&#x202F;&#x00D7;&#x202F;10<sup>&#x2212;4</sup> and a weight decay of 4&#x202F;&#x00D7;&#x202F;10<sup>&#x2212;4</sup>. The batch size was set to 2 patients per iteration, and all models were trained for 100 epochs.</p>
<p>A cosine annealing learning rate scheduler with warm-up was employed. The warm-up phase lasted for six epochs with an initial learning rate of 1&#x202F;&#x00D7;&#x202F;10<sup>&#x2212;6</sup>. The cosine schedule was initialized with a cycle length of two epochs, with the cycle length doubling after each restart up to a maximum of 20 epochs. The minimum learning rate was set to 1&#x202F;&#x00D7;&#x202F;10<sup>&#x2212;7</sup>. To ensure stable optimization under weak supervision, each epoch consisted of a fixed number of 200 training iterations.</p>
<p>The training objective was defined as a label-smoothed cross-entropy loss with a smoothing factor of 0.2, which helped mitigate overconfidence and improve generalization in the presence of noisy patient-level supervision. All hyperparameters were kept identical across cross-validation folds, resulting in negligible variation and ensuring reproducible model training.</p>
</sec>
<sec id="sec12">
<label>2.10</label>
<title>Statistical analysis</title>
<p>Statistical analyses were performed using MATLAB (MathWorks, Natick, MA, United States). Model performance was assessed using accuracy, precision, recall, <italic>F</italic><sub>1</sub>-score, and AUC with 95% confidence intervals (CIs). Group-wise comparisons were performed using paired <italic>t</italic>-tests or Wilcoxon signed-rank tests, with significance set at <italic>p</italic>&#x202F;&#x003C;&#x202F;0.05. Uniform manifold approximation and projection were performed for feature visualization and implemented using MATLAB.</p>
<p>Deep learning experiments, including feature extraction from foundation models and MIL frameworks, were conducted in Python (version 3.10) using an NVIDIA GTX 3090 GPU.</p>
</sec>
</sec>
<sec sec-type="results" id="sec13">
<label>3</label>
<title>Results</title>
<sec id="sec14">
<label>3.1</label>
<title>Baseline characteristics of study participants</title>
<p>We analyzed 157 patients with esophageal squamous cell carcinoma who received NAT treatment. Among them, 126 (80.3%) were male, and 100 (63.7%) were aged &#x2265;60&#x202F;years. Postoperative pathological outcomes showed that 41 and 116 patients achieved TRG 0&#x2013;1 and TRG 2&#x2013;3, respectively. The baseline characteristics of the patients are summarized in <xref ref-type="table" rid="tab1">Table 1</xref>.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Baseline characteristics of patients with esophageal squamous cell carcinoma receiving neoadjuvant therapy.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Characteristic</th>
<th align="center" valign="top">
<italic>N</italic>
</th>
<th align="center" valign="top">(%)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top" char="." colspan="3">Age (years)</td>
</tr>
<tr>
<td align="left" valign="top">&#x003C;60</td>
<td align="center" valign="top">57</td>
<td align="char" valign="top" char=".">36.3%</td>
</tr>
<tr>
<td align="left" valign="top">&#x2265;60</td>
<td align="center" valign="top">100</td>
<td align="char" valign="top" char=".">63.7%</td>
</tr>
<tr>
<td align="left" valign="top" char="." colspan="3">Sex</td>
</tr>
<tr>
<td align="left" valign="top">Male</td>
<td align="center" valign="top">126</td>
<td align="char" valign="top" char=".">80.3%</td>
</tr>
<tr>
<td align="left" valign="top">Female</td>
<td align="center" valign="top">31</td>
<td align="char" valign="top" char=".">19.7%</td>
</tr>
<tr>
<td align="left" valign="top" char="." colspan="3">BMI</td>
</tr>
<tr>
<td align="left" valign="top">&#x003C;18</td>
<td align="center" valign="top">31</td>
<td align="char" valign="top" char=".">19.8%</td>
</tr>
<tr>
<td align="left" valign="top">18&#x2013;24</td>
<td align="center" valign="top">105</td>
<td align="char" valign="top" char=".">66.8%</td>
</tr>
<tr>
<td align="left" valign="top">&#x003E;24</td>
<td align="center" valign="top">21</td>
<td align="char" valign="top" char=".">13.4%</td>
</tr>
<tr>
<td align="left" valign="top" char="." colspan="3">Tumor location</td>
</tr>
<tr>
<td align="left" valign="top">Upper</td>
<td align="center" valign="top">14</td>
<td align="char" valign="top" char=".">9.0%</td>
</tr>
<tr>
<td align="left" valign="top">Middle</td>
<td align="center" valign="top">93</td>
<td align="char" valign="top" char=".">59.2%</td>
</tr>
<tr>
<td align="left" valign="top">Lower</td>
<td align="center" valign="top">50</td>
<td align="char" valign="top" char=".">31.8%</td>
</tr>
<tr>
<td align="left" valign="top" char="." colspan="3">Clinical stage</td>
</tr>
<tr>
<td align="left" valign="top">II</td>
<td align="center" valign="top">39</td>
<td align="char" valign="top" char=".">24.8%</td>
</tr>
<tr>
<td align="left" valign="top">III</td>
<td align="center" valign="top">102</td>
<td align="char" valign="top" char=".">65.0%</td>
</tr>
<tr>
<td align="left" valign="top">IVa</td>
<td align="center" valign="top">16</td>
<td align="char" valign="top" char=".">10.2%</td>
</tr>
<tr>
<td align="left" valign="top" char="." colspan="3">Neoadjuvant therapy cycle</td>
</tr>
<tr>
<td align="left" valign="top">&#x2264;2</td>
<td align="center" valign="top">121</td>
<td align="char" valign="top" char=".">77.1%</td>
</tr>
<tr>
<td align="left" valign="top">&#x003E;2</td>
<td align="center" valign="top">36</td>
<td align="char" valign="top" char=".">22.9%</td>
</tr>
<tr>
<td align="left" valign="top" char="." colspan="3">ypT stage</td>
</tr>
<tr>
<td align="left" valign="top">T0&#x2013;T2</td>
<td align="center" valign="top">73</td>
<td align="char" valign="top" char=".">46.5%</td>
</tr>
<tr>
<td align="left" valign="top">T3&#x2013;T4</td>
<td align="center" valign="top">84</td>
<td align="char" valign="top" char=".">53.5%</td>
</tr>
<tr>
<td align="left" valign="top" char="." colspan="3">ypN stage</td>
</tr>
<tr>
<td align="left" valign="top">N0</td>
<td align="center" valign="top">91</td>
<td align="char" valign="top" char=".">58.0%</td>
</tr>
<tr>
<td align="left" valign="top">N1&#x2013;N3</td>
<td align="center" valign="top">66</td>
<td align="char" valign="top" char=".">42.0%</td>
</tr>
<tr>
<td align="left" valign="top" char="." colspan="3">TRG</td>
</tr>
<tr>
<td align="left" valign="top">TRG 0&#x2013;1</td>
<td align="center" valign="top">41</td>
<td align="char" valign="top" char=".">26.1%</td>
</tr>
<tr>
<td align="left" valign="top">TRG 2&#x2013;3</td>
<td align="center" valign="top">116</td>
<td align="char" valign="top" char=".">73.9%</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>TRG, tumor regression grade; BMI, body mass index.</p>
</table-wrap-foot>
</table-wrap>
<p>We designed a two-stage analytical workflow to systematically evaluate the performance of the pathology foundation models and establish an interpretable AI framework for TRG score prediction (<xref ref-type="fig" rid="fig1">Figure 1</xref>). At the slide level, three large-scale pathology foundation models were evaluated using 120 annotated WSIs to extract high-dimensional histopathological embeddings. The best-performing model was integrated into a patient-level prediction framework based on MIL, in which slide-level embeddings served as instances, and patient-level TRG score served as weak supervisory labels.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Overall analytical workflow for tumor regression grade prediction in esophageal cancer. Schematic illustration of the two-stage pipeline. In stage 1, slide-level embeddings were extracted from H&#x0026;E WSIs using three pathology foundation models (CHIEF, Prov-Gigapath, and TITAN) and benchmarked using a five-fold cross-validated support vector machine (SVM) classification. In stage 2, the best-performing model (CHIEF) provided slide-level representations for patient-level MIL frameworks (ABMIL, ACMIL, and CLAM), enabling weakly supervised prediction of TRG. Quantitative evaluation and interpretability analyses were performed at the slide and patient levels. H&#x0026;E, hematoxylin and eosin; WSIs, whole-slide images; MIL, multiple instance learning.</p>
</caption>
<graphic xlink:href="fmed-13-1751768-g001.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Flowchart illustrating a process for TRG scoring and patient classification using WSIs. It begins with 120 WSIs with slide-level TRG scores. Foundation models, Prov-Gigapath, TITAN, and CHIEF, extract features. CHIEF is selected as the best model. Slide-level classification and feature visualization through UMAP are shown. It includes patient-level TRG scoring from 1,298 WSIs of 157 patients, employing multiple instance learning models: abmil, acmil, and clam. The process culminates in automated scoring, patient classification, and survival analysis.</alt-text>
</graphic>
</fig>
<p>Furthermore, we projected high-dimensional slide-level embeddings onto 2D using uniform manifold approximation and projection to assess the discriminative capacity of slide-level features extracted with different pathology foundation models (<xref ref-type="fig" rid="fig2">Figure 2</xref>). The resulting visualization revealed that the embeddings generated by CHIEF and TITAN demonstrated a clearer separation between the TRG categories, particularly for the extreme classes, such as TRG 0 and TRG 3. In contrast, the features derived from Prov-Gigapath showed substantial overlap across categories, indicating limited discriminability for this specific task. These observations suggest that models pretrained with broader histological diversity (CHIEF) or multimodal objectives (TITAN) were more effective in capturing regression-related morphological variations than the patch-centric Prov-Gigapath model.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Two-dimensional visualization of slide-level embeddings generated using different pathology foundation models. The plots illustrate the feature space distributions extracted by three distinct backbones: <bold>(A)</bold> CHIEF, <bold>(B)</bold> Prov-Gigapath, and <bold>(C)</bold> TITAN. Method: High-dimensional slide representations (e.g., 768 dimensions for CHIEF) were projected onto a 2D plane using uniform manifold approximation and projection (UMAP). Data representation: Each individual point represents a single whole slide image (WSI) (slide-level), color-coded by its ground-truth TRG category (red: TRG 3; blue: TRG 0; etc.). Takeaway: The visualization demonstrates that the CHIEF model produces the most distinct separation between the extreme classes (TRG 0 vs. TRG 3) with minimal overlap compared to Prov-Gigapath and TITAN, visually validating its selection as the optimal feature extractor for the subsequent patient-level MIL framework.</p>
</caption>
<graphic xlink:href="fmed-13-1751768-g002.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Three scatter plots labeled A, B, and C showcase UMAP visualizations of features for datasets titled CHIEF, Prov-Gigapath, and TITAN. Each plot contains colored dots representing clusters with labels 0, 1, 2, and 3. The axes are labeled UMAP 1 and UMAP 2.</alt-text>
</graphic>
</fig>
<p>The quantitative classification experiments further confirmed these qualitative trends (<xref ref-type="table" rid="tab2">Table 2</xref>). In five-fold cross-validation, CHIEF consistently outperformed the other two models when using support vector machine classifiers trained on slide-level features. It achieved balanced precision, recall, and <italic>F</italic><sub>1</sub>-scores across most TRG categories, with particularly high performance for TRG 3 (<italic>F</italic><sub>1</sub>&#x202F;=&#x202F;0.933) and acceptable results for intermediate categories such as TRG 1 and TRG 2 (<italic>F</italic><sub>1</sub> ranging from 0.556 to 0.746). TITAN achieved competitive performance in TRG 2 and 3 but was less effective in the lower categories. Prov-Gigapath performed poorly overall, with most <italic>F</italic><sub>1</sub>-scores of 0.40. These findings indicate that CHIEF is the most suitable backbone for downstream patient-level modeling (see <xref ref-type="table" rid="tab3">Table 3</xref>).</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Slide-level classification performance of pathology foundation models.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">TRG score</th>
<th align="center" valign="top">Precision</th>
<th align="center" valign="top">Recall</th>
<th align="center" valign="top"><italic>F</italic><sub>1</sub>-score</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" colspan="4">CHIEF</td>
</tr>
<tr>
<td align="left" valign="middle">0</td>
<td align="char" valign="middle" char=".">0.686</td>
<td align="char" valign="middle" char=".">0.801</td>
<td align="char" valign="middle" char=".">0.738</td>
</tr>
<tr>
<td align="left" valign="middle">1</td>
<td align="char" valign="middle" char=".">0.625</td>
<td align="char" valign="middle" char=".">0.500</td>
<td align="char" valign="middle" char=".">0.556</td>
</tr>
<tr>
<td align="left" valign="middle">2</td>
<td align="char" valign="middle" char=".">0.733</td>
<td align="char" valign="middle" char=".">0.759</td>
<td align="char" valign="middle" char=".">0.746</td>
</tr>
<tr>
<td align="left" valign="middle">3</td>
<td align="char" valign="middle" char=".">0.933</td>
<td align="char" valign="middle" char=".">0.933</td>
<td align="char" valign="middle" char=".">0.933</td>
</tr>
<tr>
<td align="left" valign="middle" colspan="4">Prov-Gigapath</td>
</tr>
<tr>
<td align="left" valign="middle">0</td>
<td align="char" valign="middle" char=".">0.286</td>
<td align="char" valign="middle" char=".">0.333</td>
<td align="char" valign="middle" char=".">0.308</td>
</tr>
<tr>
<td align="left" valign="middle">1</td>
<td align="char" valign="middle" char=".">0.133</td>
<td align="char" valign="middle" char=".">0.067</td>
<td align="char" valign="middle" char=".">0.089</td>
</tr>
<tr>
<td align="left" valign="middle">2</td>
<td align="char" valign="middle" char=".">0.242</td>
<td align="char" valign="middle" char=".">0.267</td>
<td align="char" valign="middle" char=".">0.254</td>
</tr>
<tr>
<td align="left" valign="middle">3</td>
<td align="char" valign="middle" char=".">0.361</td>
<td align="char" valign="middle" char=".">0.448</td>
<td align="char" valign="middle" char=".">0.400</td>
</tr>
<tr>
<td align="left" valign="middle" colspan="4">TITAN</td>
</tr>
<tr>
<td align="left" valign="middle">0</td>
<td align="char" valign="middle" char=".">0.565</td>
<td align="char" valign="middle" char=".">0.433</td>
<td align="char" valign="middle" char=".">0.491</td>
</tr>
<tr>
<td align="left" valign="middle">1</td>
<td align="char" valign="middle" char=".">0.444</td>
<td align="char" valign="middle" char=".">0.533</td>
<td align="char" valign="middle" char=".">0.485</td>
</tr>
<tr>
<td align="left" valign="middle">2</td>
<td align="char" valign="middle" char=".">0.645</td>
<td align="char" valign="middle" char=".">0.667</td>
<td align="char" valign="middle" char=".">0.656</td>
</tr>
<tr>
<td align="left" valign="middle">3</td>
<td align="char" valign="middle" char=".">0.931</td>
<td align="char" valign="middle" char=".">0.931</td>
<td align="char" valign="middle" char=".">0.931</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Patient-level TRG prediction performance of multiple instance learning (MIL) models.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th/>
<th align="center" valign="top" colspan="3">ABMIL</th>
<th align="center" valign="top" colspan="3">ACMIL</th>
<th align="center" valign="top" colspan="3">CLAM</th>
</tr>
<tr>
<th align="left" valign="top">Class</th>
<th align="center" valign="top">AUC</th>
<th align="center" valign="top">95% CI lower</th>
<th align="center" valign="top">95% CI upper</th>
<th align="center" valign="top">AUC</th>
<th align="center" valign="top">95% CI lower</th>
<th align="center" valign="top">95% CI upper</th>
<th align="center" valign="top">AUC</th>
<th align="center" valign="top">95% CI lower</th>
<th align="center" valign="top">95% CI upper</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="bottom" colspan="10">Fold 1</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 0</td>
<td align="char" valign="bottom" char=".">0.918</td>
<td align="char" valign="bottom" char=".">0.817</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">0.866</td>
<td align="char" valign="bottom" char=".">0.695</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">0.970</td>
<td align="char" valign="bottom" char=".">0.903</td>
<td align="char" valign="bottom" char=".">1.000</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 1</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">0.903</td>
<td align="char" valign="bottom" char=".">0.795</td>
<td align="char" valign="bottom" char=".">0.994</td>
<td align="char" valign="bottom" char=".">0.968</td>
<td align="char" valign="bottom" char=".">0.898</td>
<td align="char" valign="bottom" char=".">1.000</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 2</td>
<td align="char" valign="bottom" char=".">0.689</td>
<td align="char" valign="bottom" char=".">0.332</td>
<td align="char" valign="bottom" char=".">0.971</td>
<td align="char" valign="bottom" char=".">0.748</td>
<td align="char" valign="bottom" char=".">0.414</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">0.696</td>
<td align="char" valign="bottom" char=".">0.425</td>
<td align="char" valign="bottom" char=".">0.964</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 3</td>
<td align="char" valign="bottom" char=".">0.914</td>
<td align="char" valign="bottom" char=".">0.800</td>
<td align="char" valign="bottom" char=".">0.988</td>
<td align="char" valign="bottom" char=".">0.886</td>
<td align="char" valign="bottom" char=".">0.765</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">0.894</td>
<td align="char" valign="bottom" char=".">0.758</td>
<td align="char" valign="bottom" char=".">0.980</td>
</tr>
<tr>
<td align="left" valign="bottom">Macro-average</td>
<td align="char" valign="middle" char=".">0.880</td>
<td align="char" valign="middle" char=".">0.699</td>
<td align="char" valign="middle" char=".">0.964</td>
<td align="char" valign="middle" char=".">0.851</td>
<td align="char" valign="middle" char=".">0.719</td>
<td align="char" valign="middle" char=".">0.976</td>
<td align="char" valign="middle" char=".">0.882</td>
<td align="char" valign="middle" char=".">0.755</td>
<td align="char" valign="middle" char=".">0.987</td>
</tr>
<tr>
<td align="left" valign="bottom">Micro-average</td>
<td align="char" valign="middle" char=".">0.899</td>
<td align="char" valign="middle" char=".">0.790</td>
<td align="char" valign="middle" char=".">0.963</td>
<td align="char" valign="middle" char=".">0.868</td>
<td align="char" valign="middle" char=".">0.772</td>
<td align="char" valign="middle" char=".">0.973</td>
<td align="char" valign="middle" char=".">0.904</td>
<td align="char" valign="middle" char=".">0.804</td>
<td align="char" valign="middle" char=".">0.967</td>
</tr>
<tr>
<td align="left" valign="middle" colspan="10">Fold 2</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 0</td>
<td align="char" valign="middle" char=".">0.964</td>
<td align="char" valign="middle" char=".">0.839</td>
<td align="char" valign="middle" char=".">1.000</td>
<td align="char" valign="middle" char=".">0.973</td>
<td align="char" valign="middle" char=".">0.875</td>
<td align="char" valign="middle" char=".">1.000</td>
<td align="char" valign="middle" char=".">0.973</td>
<td align="char" valign="middle" char=".">0.903</td>
<td align="char" valign="middle" char=".">1.000</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 1</td>
<td align="char" valign="middle" char=".">0.800</td>
<td align="char" valign="middle" char=".">0.578</td>
<td align="char" valign="middle" char=".">0.971</td>
<td align="char" valign="middle" char=".">0.967</td>
<td align="char" valign="middle" char=".">0.894</td>
<td align="char" valign="middle" char=".">1.000</td>
<td align="char" valign="middle" char=".">0.750</td>
<td align="char" valign="middle" char=".">0.405</td>
<td align="char" valign="middle" char=".">1.000</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 2</td>
<td align="char" valign="middle" char=".">0.698</td>
<td align="char" valign="middle" char=".">0.397</td>
<td align="char" valign="middle" char=".">0.937</td>
<td align="char" valign="middle" char=".">0.620</td>
<td align="char" valign="middle" char=".">0.410</td>
<td align="char" valign="middle" char=".">0.889</td>
<td align="char" valign="middle" char=".">0.604</td>
<td align="char" valign="middle" char=".">0.398</td>
<td align="char" valign="middle" char=".">0.784</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 3</td>
<td align="char" valign="middle" char=".">0.750</td>
<td align="char" valign="middle" char=".">0.599</td>
<td align="char" valign="middle" char=".">0.910</td>
<td align="char" valign="middle" char=".">0.782</td>
<td align="char" valign="middle" char=".">0.619</td>
<td align="char" valign="middle" char=".">0.947</td>
<td align="char" valign="middle" char=".">0.762</td>
<td align="char" valign="middle" char=".">0.587</td>
<td align="char" valign="middle" char=".">0.931</td>
</tr>
<tr>
<td align="left" valign="bottom">Macro-average</td>
<td align="char" valign="middle" char=".">0.803</td>
<td align="char" valign="middle" char=".">0.680</td>
<td align="char" valign="middle" char=".">0.937</td>
<td align="char" valign="middle" char=".">0.835</td>
<td align="char" valign="middle" char=".">0.695</td>
<td align="char" valign="middle" char=".">0.921</td>
<td align="char" valign="middle" char=".">0.772</td>
<td align="char" valign="middle" char=".">0.626</td>
<td align="char" valign="middle" char=".">0.899</td>
</tr>
<tr>
<td align="left" valign="bottom">Micro-average</td>
<td align="char" valign="middle" char=".">0.828</td>
<td align="char" valign="middle" char=".">0.732</td>
<td align="char" valign="middle" char=".">0.909</td>
<td align="char" valign="middle" char=".">0.846</td>
<td align="char" valign="middle" char=".">0.736</td>
<td align="char" valign="middle" char=".">0.914</td>
<td align="char" valign="middle" char=".">0.826</td>
<td align="char" valign="middle" char=".">0.707</td>
<td align="char" valign="middle" char=".">0.907</td>
</tr>
<tr>
<td align="left" valign="middle" colspan="10">Fold 3</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 0</td>
<td align="char" valign="bottom" char=".">0.962</td>
<td align="char" valign="bottom" char=".">0.862</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">0.969</td>
<td align="char" valign="bottom" char=".">0.889</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">0.969</td>
<td align="char" valign="bottom" char=".">0.877</td>
<td align="char" valign="bottom" char=".">1.000</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 1</td>
<td align="char" valign="bottom" char=".">0.714</td>
<td align="char" valign="bottom" char=".">0.345</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">0.690</td>
<td align="char" valign="bottom" char=".">0.378</td>
<td align="char" valign="bottom" char=".">0.970</td>
<td align="char" valign="bottom" char=".">0.869</td>
<td align="char" valign="bottom" char=".">0.716</td>
<td align="char" valign="bottom" char=".">1.000</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 2</td>
<td align="char" valign="bottom" char=".">0.613</td>
<td align="char" valign="bottom" char=".">0.259</td>
<td align="char" valign="bottom" char=".">0.920</td>
<td align="char" valign="bottom" char=".">0.727</td>
<td align="char" valign="bottom" char=".">0.286</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">0.667</td>
<td align="char" valign="bottom" char=".">0.375</td>
<td align="char" valign="bottom" char=".">0.981</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 3</td>
<td align="char" valign="bottom" char=".">0.857</td>
<td align="char" valign="bottom" char=".">0.671</td>
<td align="char" valign="bottom" char=".">0.974</td>
<td align="char" valign="bottom" char=".">0.882</td>
<td align="char" valign="bottom" char=".">0.744</td>
<td align="char" valign="bottom" char=".">0.991</td>
<td align="char" valign="bottom" char=".">0.887</td>
<td align="char" valign="bottom" char=".">0.759</td>
<td align="char" valign="bottom" char=".">0.973</td>
</tr>
<tr>
<td align="left" valign="bottom">Macro-average</td>
<td align="char" valign="bottom" char=".">0.787</td>
<td align="char" valign="bottom" char=".">0.639</td>
<td align="char" valign="bottom" char=".">0.911</td>
<td align="char" valign="bottom" char=".">0.817</td>
<td align="char" valign="bottom" char=".">0.708</td>
<td align="char" valign="bottom" char=".">0.955</td>
<td align="char" valign="bottom" char=".">0.848</td>
<td align="char" valign="bottom" char=".">0.747</td>
<td align="char" valign="bottom" char=".">0.931</td>
</tr>
<tr>
<td align="left" valign="bottom">Micro-average</td>
<td align="char" valign="bottom" char=".">0.862</td>
<td align="char" valign="bottom" char=".">0.755</td>
<td align="char" valign="bottom" char=".">0.938</td>
<td align="char" valign="bottom" char=".">0.895</td>
<td align="char" valign="bottom" char=".">0.813</td>
<td align="char" valign="bottom" char=".">0.984</td>
<td align="char" valign="bottom" char=".">0.888</td>
<td align="char" valign="bottom" char=".">0.768</td>
<td align="char" valign="bottom" char=".">0.964</td>
</tr>
<tr>
<td align="left" valign="middle" colspan="10">Fold 4</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 0</td>
<td align="char" valign="bottom" char=".">0.931</td>
<td align="char" valign="bottom" char=".">0.778</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">0.977</td>
<td align="char" valign="bottom" char=".">0.907</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">1.000</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 1</td>
<td align="char" valign="bottom" char=".">0.828</td>
<td align="char" valign="bottom" char=".">0.633</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">0.914</td>
<td align="char" valign="bottom" char=".">0.782</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">0.966</td>
<td align="char" valign="bottom" char=".">0.915</td>
<td align="char" valign="bottom" char=".">1.000</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 2</td>
<td align="char" valign="bottom" char=".">0.815</td>
<td align="char" valign="bottom" char=".">0.631</td>
<td align="char" valign="bottom" char=".">0.965</td>
<td align="char" valign="bottom" char=".">0.755</td>
<td align="char" valign="bottom" char=".">0.565</td>
<td align="char" valign="bottom" char=".">0.985</td>
<td align="char" valign="bottom" char=".">0.772</td>
<td align="char" valign="bottom" char=".">0.519</td>
<td align="char" valign="bottom" char=".">1.000</td>
</tr>
<tr>
<td align="left" valign="bottom">TRG 3</td>
<td align="char" valign="bottom" char=".">0.946</td>
<td align="char" valign="bottom" char=".">0.853</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">0.913</td>
<td align="char" valign="bottom" char=".">0.774</td>
<td align="char" valign="bottom" char=".">1.000</td>
<td align="char" valign="bottom" char=".">0.900</td>
<td align="char" valign="bottom" char=".">0.724</td>
<td align="char" valign="bottom" char=".">1.000</td>
</tr>
<tr>
<td align="left" valign="bottom">Macro-average</td>
<td align="char" valign="bottom" char=".">0.880</td>
<td align="char" valign="bottom" char=".">0.774</td>
<td align="char" valign="bottom" char=".">0.977</td>
<td align="char" valign="bottom" char=".">0.890</td>
<td align="char" valign="bottom" char=".">0.770</td>
<td align="char" valign="bottom" char=".">0.981</td>
<td align="char" valign="bottom" char=".">0.909</td>
<td align="char" valign="bottom" char=".">0.804</td>
<td align="char" valign="bottom" char=".">1.000</td>
</tr>
<tr>
<td align="left" valign="bottom">Micro-average</td>
<td align="char" valign="bottom" char=".">0.891</td>
<td align="char" valign="bottom" char=".">0.793</td>
<td align="char" valign="bottom" char=".">0.964</td>
<td align="char" valign="bottom" char=".">0.916</td>
<td align="char" valign="bottom" char=".">0.844</td>
<td align="char" valign="bottom" char=".">0.977</td>
<td align="char" valign="bottom" char=".">0.940</td>
<td align="char" valign="top" char=".">0.867</td>
<td align="char" valign="top" char=".">0.992</td>
</tr>
<tr>
<td align="left" valign="top" colspan="10">Fold 5</td>
</tr>
<tr>
<td align="left" valign="top">TRG 0</td>
<td align="char" valign="top" char=".">0.905</td>
<td align="char" valign="top" char=".">0.813</td>
<td align="char" valign="top" char=".">1.000</td>
<td align="char" valign="top" char=".">0.917</td>
<td align="char" valign="top" char=".">0.773</td>
<td align="char" valign="top" char=".">1.000</td>
<td align="char" valign="top" char=".">0.964</td>
<td align="char" valign="top" char=".">0.869</td>
<td align="char" valign="top" char=".">1.000</td>
</tr>
<tr>
<td align="left" valign="top">TRG 1</td>
<td align="char" valign="top" char=".">0.454</td>
<td align="char" valign="top" char=".">0.138</td>
<td align="char" valign="top" char=".">0.762</td>
<td align="char" valign="top" char=".">0.585</td>
<td align="char" valign="top" char=".">0.267</td>
<td align="char" valign="top" char=".">0.893</td>
<td align="char" valign="top" char=".">0.592</td>
<td align="char" valign="top" char=".">0.267</td>
<td align="char" valign="top" char=".">0.907</td>
</tr>
<tr>
<td align="left" valign="top">TRG 2</td>
<td align="char" valign="top" char=".">0.877</td>
<td align="char" valign="top" char=".">0.759</td>
<td align="char" valign="top" char=".">0.995</td>
<td align="char" valign="top" char=".">0.709</td>
<td align="char" valign="top" char=".">0.500</td>
<td align="char" valign="top" char=".">0.877</td>
<td align="char" valign="top" char=".">0.727</td>
<td align="char" valign="top" char=".">0.535</td>
<td align="char" valign="top" char=".">0.889</td>
</tr>
<tr>
<td align="left" valign="top">TRG 3</td>
<td align="char" valign="top" char=".">0.969</td>
<td align="char" valign="top" char=".">0.889</td>
<td align="char" valign="top" char=".">1.000</td>
<td align="char" valign="top" char=".">0.939</td>
<td align="char" valign="top" char=".">0.846</td>
<td align="char" valign="top" char=".">1.000</td>
<td align="char" valign="top" char=".">0.939</td>
<td align="char" valign="top" char=".">0.833</td>
<td align="char" valign="top" char=".">1.000</td>
</tr>
<tr>
<td align="left" valign="top">Macro-average</td>
<td align="char" valign="top" char=".">0.801</td>
<td align="char" valign="top" char=".">0.707</td>
<td align="char" valign="top" char=".">0.875</td>
<td align="char" valign="top" char=".">0.787</td>
<td align="char" valign="top" char=".">0.681</td>
<td align="char" valign="top" char=".">0.883</td>
<td align="char" valign="top" char=".">0.806</td>
<td align="char" valign="top" char=".">0.657</td>
<td align="char" valign="top" char=".">0.932</td>
</tr>
<tr>
<td align="left" valign="top">Micro-average</td>
<td align="char" valign="top" char=".">0.833</td>
<td align="char" valign="top" char=".">0.720</td>
<td align="char" valign="top" char=".">0.960</td>
<td align="char" valign="top" char=".">0.830</td>
<td align="char" valign="top" char=".">0.732</td>
<td align="char" valign="top" char=".">0.905</td>
<td align="char" valign="top" char=".">0.812</td>
<td align="char" valign="top" char=".">0.691</td>
<td align="char" valign="top" char=".">0.902</td>
</tr>
<tr>
<td align="left" valign="top" colspan="10">Overall</td>
</tr>
<tr>
<td align="left" valign="top">TRG 0</td>
<td align="char" valign="top" char=".">0.897</td>
<td align="char" valign="top" char=".">0.824</td>
<td align="char" valign="top" char=".">0.957</td>
<td align="char" valign="top" char=".">0.899</td>
<td align="char" valign="top" char=".">0.814</td>
<td align="char" valign="top" char=".">0.964</td>
<td align="char" valign="top" char=".">0.955</td>
<td align="char" valign="top" char=".">0.899</td>
<td align="char" valign="top" char=".">0.992</td>
</tr>
<tr>
<td align="left" valign="top">TRG 1</td>
<td align="char" valign="top" char=".">0.707</td>
<td align="char" valign="top" char=".">0.516</td>
<td align="char" valign="top" char=".">0.847</td>
<td align="char" valign="top" char=".">0.735</td>
<td align="char" valign="top" char=".">0.567</td>
<td align="char" valign="top" char=".">0.868</td>
<td align="char" valign="top" char=".">0.771</td>
<td align="char" valign="top" char=".">0.628</td>
<td align="char" valign="top" char=".">0.879</td>
</tr>
<tr>
<td align="left" valign="top">TRG 2</td>
<td align="char" valign="top" char=".">0.749</td>
<td align="char" valign="top" char=".">0.627</td>
<td align="char" valign="top" char=".">0.856</td>
<td align="char" valign="top" char=".">0.737</td>
<td align="char" valign="top" char=".">0.627</td>
<td align="char" valign="top" char=".">0.839</td>
<td align="char" valign="top" char=".">0.693</td>
<td align="char" valign="top" char=".">0.584</td>
<td align="char" valign="top" char=".">0.776</td>
</tr>
<tr>
<td align="left" valign="top">TRG 3</td>
<td align="char" valign="top" char=".">0.877</td>
<td align="char" valign="top" char=".">0.814</td>
<td align="char" valign="top" char=".">0.939</td>
<td align="char" valign="top" char=".">0.870</td>
<td align="char" valign="top" char=".">0.799</td>
<td align="char" valign="top" char=".">0.919</td>
<td align="char" valign="top" char=".">0.859</td>
<td align="char" valign="top" char=".">0.804</td>
<td align="char" valign="top" char=".">0.913</td>
</tr>
<tr>
<td align="left" valign="top">Macro-average</td>
<td align="char" valign="top" char=".">0.807</td>
<td align="char" valign="top" char=".">0.746</td>
<td align="char" valign="top" char=".">0.865</td>
<td align="char" valign="top" char=".">0.810</td>
<td align="char" valign="top" char=".">0.745</td>
<td align="char" valign="top" char=".">0.868</td>
<td align="char" valign="top" char=".">0.820</td>
<td align="char" valign="top" char=".">0.771</td>
<td align="char" valign="top" char=".">0.880</td>
</tr>
<tr>
<td align="left" valign="top">Micro-average</td>
<td align="char" valign="top" char=".">0.860</td>
<td align="char" valign="top" char=".">0.817</td>
<td align="char" valign="top" char=".">0.913</td>
<td align="char" valign="top" char=".">0.873</td>
<td align="char" valign="top" char=".">0.828</td>
<td align="char" valign="top" char=".">0.911</td>
<td align="char" valign="top" char=".">0.870</td>
<td align="char" valign="top" char=".">0.829</td>
<td align="char" valign="top" char=".">0.904</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>TRG, tumor regression grade; AUC, area under the receiver operating characteristic curve.</p>
</table-wrap-foot>
</table-wrap>
<p>Based on the slide-level results, we constructed a patient-level TRG prediction framework using MIL. The dataset comprised 157 patients, each contributing 2&#x2013;10 WSIs. Because only patient-level TRG labels were available, each patient was treated as a <italic>bag</italic> containing multiple <italic>instances</italic> (WSIs), with the CHIEF-extracted features of each slide serving as instance representations.</p>
<p>We implemented and compared three representative MIL approaches, namely attention-based MIL (ABMIL), ACMIL, and CLAM, to aggregate slide-level features into patient-level predictions. The model was trained using a five-fold cross-validation setting to ensure patient-level independence between the folds.</p>
<p>Across the five-fold cross-validation, the models demonstrated robust stability. We report the performance as the mean metrics across the folds. The MIL variants achieved a mean macro-average AUC of 0.810 (95% CI: 0.745&#x2013;0.865) for ACMIL and 0.820 (95% CI: 0.771&#x2013;0.880) for CLAM. The mean micro-average AUCs consistently exceeded 0.86. Specifically, the CLAM model achieved the highest mean macro-average AUC, while ACMIL demonstrated the most balanced performance with a mean overall accuracy of 82.7%.</p>
<p>The overall receiver operating characteristic curves (<xref ref-type="fig" rid="fig3">Figure 3</xref>) illustrated clear separability across the TRG categories, with TRG 0 and 3 achieving the highest discriminative power (AUCs of 0.955 and 0.859 for CLAM, respectively). Intermediate grades (TRG 1&#x2013;2) showed moderate but consistent classification performance (AUCs&#x202F;=&#x202F;0.693&#x2013;0.771).</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Receiver operating characteristic (ROC) curve analysis of patient-level multiple instance learning models for tumor regression grade prediction. Evaluation setting: The curves represent the patient-level classification performance evaluated using a 5-fold cross-validation scheme. Feature embeddings were extracted using the CHIEF foundation model. Metrics: The plots display the macro-average (blue) and micro-average (orange) ROC curves for the multi-class prediction (TRG 0&#x2013;3). Model comparison: <bold>(A)</bold> ABMIL, <bold>(B)</bold> ACMIL, and <bold>(C)</bold> CLAM. The shaded areas (if applicable) or curves represent the mean performance across folds, demonstrating the robust discriminative ability of the MIL frameworks, with CLAM and ACMIL showing superior area under the curve (AUC) values.</p>
</caption>
<graphic xlink:href="fmed-13-1751768-g003.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Three panels, labeled A, B, and C, display multiclass ROC curves for models ABMIL, ACMIL, and CLAM respectively. Each graph plots true positive rate against false positive rate, using lines for different targets and averages. Panel A shows AUC values for targets, macro, and micro averages as 0.925, 0.649, 0.720, 0.887, 0.807, and 0.860. Panel B shows 0.932, 0.691, 0.709, 0.886, 0.810, and 0.873. Panel C shows 0.944, 0.702, 0.676, 0.884, 0.820, and 0.870.</alt-text>
</graphic>
</fig>
<p>The patient-level confusion matrices (<xref ref-type="fig" rid="fig4">Figure 4</xref>) revealed that most classification errors occurred between adjacent TRG categories, reflecting the inherent ambiguity in histological response assessment. Across all three MIL models, predictions of TRG 0 and TRG 3 showed high agreement with the ground truth (accuracy &#x003E;0.85), whereas TRG 1 and TRG 2 were more prone to misclassification owing to overlapping morphological features such as mixed residual tumor and stromal fibrosis.</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Confusion matrices of patient-level tumor regression grade classification. Experimental setup: The matrices visualize the classification results on the test sets accumulated across 5-fold cross-validation. Input features were extracted using the CHIEF backbone. Axes definition: The <italic>Y</italic>-axis represents the manual TRG score (ground truth) assigned by pathologists, while the <italic>X</italic>-axis represents the predicted TRG score generated by the AI models. Values: Numbers inside the cells indicate the raw count of patients. Darker colors represent higher density/agreement. Comparison: <bold>(A)</bold> ABMIL, <bold>(B)</bold> ACMIL, and <bold>(C)</bold> CLAM. Notably, the ACMIL model <bold>(B)</bold> demonstrates the strongest diagonal alignment (highest agreement with human graders) and the most balanced classification accuracy (82.7%) compared to other variants.</p>
</caption>
<graphic xlink:href="fmed-13-1751768-g004.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Three confusion matrices compare predicted versus manual TRG scores. (A) ABMIL with higher accuracy at predicted score 3. (B) ACMIL showing moderate prediction accuracy, highest at score 3. (C) CLAM with strong accuracy at score 3, followed by scores 0 and 2.</alt-text>
</graphic>
</fig>
<p>Notably, the ACMIL model demonstrated the most balanced performance, correctly classifying 84.6% of the TRG 0/1 cases and 80.3% of the TRG 2/3 cases, yielding an overall accuracy of 82.7%. The confusion matrices further revealed that the MIL models captured the ordinal structure of TRG grading, as most mispredictions deviated by only one grade, underscoring the biological continuity between adjacent TRG levels rather than the random errors.</p>
<p>To assess the prognostic relevance of TRG in EC, we examined whether pathologist-assigned or AI-predicted TRG score could stratify patients by long-term survival (<xref ref-type="fig" rid="fig5">Figure 5</xref>). Given the clinical convention that treatment responders exhibit lower TRG score, the patients were classified into two groups: TRG 0&#x2013;1 (favorable response) and TRG 2&#x2013;3 (poor response). Kaplan&#x2013;Meier survival curves were generated for PFS and OS, and survival differences were evaluated using the log-rank test. Hazard ratios (HRs) and 95% CIs were estimated using Cox proportional hazard regression.</p>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>Prognostic significance of pathologist-assigned and artificial intelligence-predicted tumor grade. Kaplan&#x2013;Meier survival curves for overall survival (OS) and progression-free survival (PFS) stratified by TRG 0&#x2013;1 (favorable response) versus TRG 2&#x2013;3 (poor response). <bold>(A,E)</bold> Pathologist-assigned TRG score for PFS and OS. <bold>(B,F)</bold> TRG predictions using the ABMIL model. <bold>(C,G)</bold> TRG predictions using the ACMIL model. <bold>(D,H)</bold> TRG predictions using the CLAM model. Among all comparisons, the ACMIL model <bold>(C,G)</bold> provides the most statistically significant prognostic stratification (PFS: <italic>p</italic>&#x202F;=&#x202F;0.0196; OS: <italic>p</italic>&#x202F;=&#x202F;0.0259), surpassing both manual pathologist grading (OS: <italic>p</italic>&#x202F;=&#x202F;0.0683) and other MIL approaches.</p>
</caption>
<graphic xlink:href="fmed-13-1751768-g005.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Eight Kaplan-Meier survival plots compare progression-free survival (PFS) and overall survival (OS) for two groups: TRG 0/1 (good prognosis) in blue and TRG 2/3 (poor prognosis) in red. Each panel (A-H) represents different methods: Pathologist, ABMIL, ACMIL, and CLAM. The x-axis shows time in months, the y-axis shows survival probability. p-values and hazard ratios (HR) with confidence intervals are included. Panel (A) shows a significant difference in survival between groups, while others vary in significance.</alt-text>
</graphic>
</fig>
<p>When pathologist-assigned TRG score was used, the difference in OS between the TRG 0&#x2013;1 and TRG 2&#x2013;3 groups was not statistically significant (log-rank <italic>p</italic>&#x202F;=&#x202F;0.068; HR&#x202F;=&#x202F;0.428, 95% CI: 0.205&#x2013;0.894).</p>
<p>In contrast, the AI-predicted TRG score, particularly those generated using the ACMIL model, demonstrated an improved ability to stratify patient outcomes. For PFS, the predicted TRG 0/1 group showed a sustained advantage over the predicted TRG 2/3 group throughout the follow-up period [HR&#x202F;=&#x202F;0.312, 95% confidence interval (CI), 0.151&#x2013;0.642; <italic>p</italic>&#x202F;=&#x202F;0.0196]. A consistent pattern was observed in OS. Patients with predicted TRG 0/1 achieved longer OS than those with predicted TRG 2/3 (HR&#x202F;=&#x202F;0.327, 95% CI: 0.157&#x2013;0.682, <italic>p</italic>&#x202F;=&#x202F;0.0259).</p>
<p>These results suggest that AI-derived TRG score captures the histopathological correlates of treatment response more effectively than conventional grading, providing superior discrimination of long-term outcomes. Specifically, ACMIL-predicted TRG has emerged as a robust surrogate biomarker for patient prognosis after NAT, offering potential value for response assessment and clinical trial endpoint refinement.</p>
</sec>
</sec>
<sec sec-type="discussion" id="sec15">
<label>4</label>
<title>Discussion</title>
<p>In this study, we developed and validated a weakly supervised AI framework for the automated assessment of TRG in EC following NAT. Through combining large-scale pathology foundation models with MIL, the proposed approach achieved accurate, efficient, and reproducible patient-level TRG score prediction without the need for manual region annotations. The results demonstrated high consistency between AI-predicted and pathologist-assigned TRG score. Notably, the AI-derived TRG score exhibited superior prognostic value for PFS and OS. These findings highlight the potential of weakly supervised pathological AI systems in enhancing the objectivity and clinical utility of histopathological response assessments.</p>
<p>Conventional TRG evaluation remains the standard method for assessing histopathological responses after NAT; however, it is inherently limited by subjectivity, inter-observer variability, and workload intensity. The emergence of digital pathology and AI has provided new opportunities to overcome these challenges. Previous studies, such as those by Tolkach et al. (<xref ref-type="bibr" rid="ref17">17</xref>) and Wang et al. (<xref ref-type="bibr" rid="ref18">18</xref>), have demonstrated the feasibility of AI-assisted TRG evaluation in esophageal tumors. However, these approaches rely on manual patch-level labeling or continuous tumor percentage estimation, restricting scalability and clinical adoption.</p>
<p>Our framework overcomes these limitations through a fully weakly supervised design in which each patient is modeled as a &#x201C;bag&#x201D; containing multiple WSIs as instances. The model was trained directly on patient-level TRG labels that are readily available in clinical pathology reports, eliminating the need for manual slide- or pixel-level annotations. This strategy significantly reduces data preparation costs and enhances the feasibility of applying AI models to large-scale real-world cohorts.</p>
<p>The adoption of pathology foundation models played a central role in our framework. By evaluating three recent slide-level pretrained models, we identified CHIEF as the most discriminative and robust backbone for feature extraction at the slide level. The hierarchical vision transformer structure of CHIEF enables the simultaneous encoding of fine-grained cellular features and global tissue architecture, which are crucial for accurately characterizing tumor regression after NAT. These findings emphasize that model selection in computational pathology should be task-specific. Although Prov-Gigapath offers scalability through patch-level robustness, it lacks contextual integration. In contrast, CHIEF captures tissue-level organization more effectively, which is critical for TRG differentiation. This systematic comparison also contributes to a broader understanding of how foundation models can be leveraged and optimized for downstream clinical applications.</p>
<p>Building on CHIEF embeddings, we implemented and compared three representative MIL algorithms to integrate multiple WSIs per patient into a unified predictive model. Although the field is rapidly evolving with newer histology-focused MIL approaches, such as the method described in reference (<xref ref-type="bibr" rid="ref27">27</xref>), which further optimize feature aggregation, we focused on established methods, including CLAM and ACMIL, to ensure robust reproducibility. In our comparative analysis, ACMIL achieved the best overall balance between accuracy and stability, reaching an accuracy of 82.7% and a macro-average AUC of 0.81 under five-fold cross-validation. Furthermore, confusion matrix analysis revealed that most classification errors occurred between adjacent TRG categories, reflecting the biological continuity of tumor regression rather than random misclassification.</p>
<p>Notably, AI-derived TRG score demonstrated stronger prognostic power than manual grading did. AI-predicted TRG (0&#x2013;1 vs. 2&#x2013;3) was used to effectively stratify patients according to PFS and OS, whereas the pathologist-assigned grades were not statistically significant. This superior prognostic stratification suggests that the AI model captures underlying histological complexities that extend beyond human visual perception, a conclusion validated within the emerging quantitative pathology paradigm. Li et al. (<xref ref-type="bibr" rid="ref28">28</xref>) demonstrated that quantitative descriptors of tissue complexity, such as structural entropy, can provide biologically meaningful information about tumor structure by establishing an entropy-based comprehensive measurement framework in digital pathology. Li (<xref ref-type="bibr" rid="ref29">29</xref>) improved interpretability in weakly supervised settings by decoding the potential spatial relationships between cells from pathological images through extracting spatial descriptors. Although manual TRG grading mainly quantifies the percentage of residual tumor burden, spatial tissue organization and tissue complexity are also key determinants of prognosis. The AI model can sensitively capture this information for prognostic prediction. Additionally, the MIL framework proposed in this study integrates the global information from multiple whole-slide images of the same patient, effectively capturing the spatial heterogeneity of individual patients. This heterogeneity, although known to have a significant impact on prognosis, is underestimated in manual TRG score. Clinically, AI-assisted TRG score could achieve standardized, objective and reproducible response assessments. Moreover, AI can capture the continuous biological spectrum of pathological responses. The TRG score predicted by AI are not a strictly ordered classification but a continuous severity grade, which can provide a more refined and biologically accurate reflection of treatment effects. This advantage helps to reduce inter-observer variability and facilitate multicenter clinical trials. Furthermore, the AI framework can provide more accurate survival stratification, offering information for individualized postoperative management decisions, including the need for adjuvant therapy or intensified surveillance in high-risk patients.</p>
<p>This study has several methodological and translational strengths. First, it introduces a fully weakly supervised paradigm for TRG assessment using only patient-level labels, eliminating the dependence on detailed annotations. Second, it leverages the representational capacity of pathology foundation models, allowing generalization to diverse histopathological patterns. Third, the model was prognostically validated, establishing its potential as a diagnostic tool and biomarker of clinical outcomes. Collectively, these advances provide a scalable and explainable framework suitable for real-world clinical deployment.</p>
<p>Despite its promising performance, this study has some limitations that highlight the directions for future studies. The model was developed and validated using data from a single medical center, which may limit its external generalizability owing to variations in slide preparation, staining, and scanning protocols across institutions. Although the foundation model (CHIEF) utilized in this study&#x2014;pretrained on over 1 million diverse slides&#x2014;provides inherent feature robustness against such variations, domain shifts caused by different scanners and staining intensities remain a potential challenge. Future studies should not only involve large-scale, multicenter validations but also incorporate rigorous stain normalization and domain adaptation techniques to ensure consistent performance across diverse clinical settings. The lack of multi-factor adjustment is another limitation. With larger sample sizes, subsequent research should assess the independent AI-TRG prognostic value while adjusting for key covariates such as ypT, ypN, and staging. Finally, the relatively small cohort of 157 patients, although cross-validated, underscores the need for larger datasets to further improve stability and representational capacity.</p>
<p>We proposed a weakly supervised AI framework for the automated assessment of TRG in post-NAT for EC. This framework performs well in classification accuracy and exhibits superior prognostic stratification compared with conventional manual assessment. By integrating the pathology foundation model with advanced MIL paradigms, our study provides a practical solution to the current challenges of labor-intensive annotations and subjective variability in TRG assessments. In addition to its direct clinical applicability for the precise diagnosis and treatment of EC, our study offers a methodological reference for similar pathological image analysis tasks in other cancer types that rely on patient-level weak labels.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec16">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec sec-type="ethics-statement" id="sec17">
<title>Ethics statement</title>
<p>This study was conducted in accordance with the principles of the Declaration of Helsinki. Ethical approval and a waiver of informed consent were obtained from the Ethics Committee of Zhongshan Hospital Affiliated to Xiamen University. Written informed consent was not obtained from the individual(s) for the publication of any potentially identifiable images or data included in this article because identifiable personal information, such as names of the enrolled patients and name of hospital, was removed. This study was approved by the Ethics Committees of Zhongshan Hospital of Xiamen University.</p>
</sec>
<sec sec-type="author-contributions" id="sec18">
<title>Author contributions</title>
<p>ZL: Investigation, Writing &#x2013; review &#x0026; editing, Conceptualization, Writing &#x2013; original draft, Visualization, Formal analysis, Resources, Validation. LZ: Data curation, Writing &#x2013; original draft, Methodology, Formal analysis, Investigation. ZZ: Validation, Investigation, Resources, Funding acquisition, Writing &#x2013; original draft. WQ: Resources, Investigation, Writing &#x2013; original draft, Funding acquisition. JZ: Writing &#x2013; original draft, Resources, Investigation. FL: Writing &#x2013; original draft, Resources, Investigation. ML: Investigation, Writing &#x2013; original draft, Resources. RG: Resources, Writing &#x2013; original draft, Investigation. QG: Investigation, Writing &#x2013; original draft, Resources. HX: Writing &#x2013; original draft, Validation, Investigation, Resources. SL: Supervision, Writing &#x2013; original draft, Investigation, Conceptualization, Project administration. HZ: Validation, Conceptualization, Supervision, Writing &#x2013; original draft. HL: Conceptualization, Investigation, Funding acquisition, Supervision, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>The authors thank Editage (<ext-link xlink:href="http://www.editage.cn" ext-link-type="uri">www.editage.cn</ext-link>) for providing English language editing services.</p>
</ack>
<sec sec-type="COI-statement" id="sec19">
<title>Conflict of interest</title>
<p>The author(s) declared that this work was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec20">
<title>Generative AI statement</title>
<p>The author(s) declared that Generative AI was not used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="sec21">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><label>1.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Obermannov&#x00E1;</surname> <given-names>R</given-names></name> <name><surname>Alsina</surname> <given-names>M</given-names></name> <name><surname>Cervantes</surname> <given-names>A</given-names></name> <name><surname>Leong</surname> <given-names>T</given-names></name> <name><surname>Lordick</surname> <given-names>F</given-names></name> <name><surname>Nilsson</surname> <given-names>M</given-names></name> <etal/></person-group>. <article-title>Oesophageal cancer: ESMO clinical practice guideline for diagnosis, treatment and follow-up</article-title>. <source>Ann Oncol</source>. (<year>2022</year>) <volume>33</volume>:<fpage>992</fpage>&#x2013;<lpage>1004</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.annonc.2022.07.003</pub-id>, <pub-id pub-id-type="pmid">35914638</pub-id></mixed-citation></ref>
<ref id="ref2"><label>2.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>G</given-names></name> <name><surname>Su</surname> <given-names>X</given-names></name> <name><surname>Yang</surname> <given-names>H</given-names></name> <name><surname>Luo</surname> <given-names>G</given-names></name> <name><surname>Gao</surname> <given-names>C</given-names></name> <name><surname>Zheng</surname> <given-names>Y</given-names></name> <etal/></person-group>. <article-title>Neoadjuvant programmed death-1 blockade plus chemotherapy in locally advanced esophageal squamous cell carcinoma</article-title>. <source>Ann Transl Med</source>. (<year>2021</year>) <volume>9</volume>:<fpage>1254</fpage>. doi: <pub-id pub-id-type="doi">10.21037/atm-21-3352</pub-id>, <pub-id pub-id-type="pmid">34532391</pub-id></mixed-citation></ref>
<ref id="ref3"><label>3.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>Z</given-names></name> <name><surname>Zheng</surname> <given-names>Q</given-names></name> <name><surname>Chen</surname> <given-names>H</given-names></name> <name><surname>Xiang</surname> <given-names>J</given-names></name> <name><surname>Hu</surname> <given-names>H</given-names></name> <name><surname>Li</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>Efficacy and safety of neoadjuvant chemotherapy and immunotherapy in locally resectable advanced esophageal squamous cell carcinoma</article-title>. <source>J Thorac Dis</source>. (<year>2021</year>) <volume>13</volume>:<fpage>3518</fpage>&#x2013;<lpage>28</lpage>. doi: <pub-id pub-id-type="doi">10.21037/jtd-21-340</pub-id>, <pub-id pub-id-type="pmid">34277047</pub-id></mixed-citation></ref>
<ref id="ref4"><label>4.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Muro</surname> <given-names>K</given-names></name> <name><surname>Kojima</surname> <given-names>T</given-names></name> <name><surname>Moriwaki</surname> <given-names>T</given-names></name> <name><surname>Kato</surname> <given-names>K</given-names></name> <name><surname>Nagashima</surname> <given-names>F</given-names></name> <name><surname>Kawakami</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>Second-line pembrolizumab versus chemotherapy in Japanese patients with advanced esophageal cancer: subgroup analysis from KEYNOTE-181</article-title>. <source>Esophagus</source>. (<year>2022</year>) <volume>19</volume>:<fpage>137</fpage>&#x2013;<lpage>45</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10388-021-00877-3</pub-id>, <pub-id pub-id-type="pmid">34591237</pub-id></mixed-citation></ref>
<ref id="ref5"><label>5.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bolger</surname> <given-names>JC</given-names></name> <name><surname>Donohoe</surname> <given-names>CL</given-names></name> <name><surname>Lowery</surname> <given-names>M</given-names></name> <name><surname>Reynolds</surname> <given-names>JV</given-names></name></person-group>. <article-title>Advances in the curative management of oesophageal cancer</article-title>. <source>Br J Cancer</source>. (<year>2022</year>) <volume>126</volume>:<fpage>706</fpage>&#x2013;<lpage>17</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41416-021-01485-9</pub-id>, <pub-id pub-id-type="pmid">34675397</pub-id></mixed-citation></ref>
<ref id="ref6"><label>6.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Qin</surname> <given-names>J</given-names></name> <name><surname>Xue</surname> <given-names>L</given-names></name> <name><surname>Hao</surname> <given-names>A</given-names></name> <name><surname>Guo</surname> <given-names>X</given-names></name> <name><surname>Jiang</surname> <given-names>T</given-names></name> <name><surname>Ni</surname> <given-names>Y</given-names></name> <etal/></person-group>. <article-title>Neoadjuvant chemotherapy with or without camrelizumab in resectable esophageal squamous cell carcinoma: the randomized phase 3 escort-NEO/NCCES01 trial</article-title>. <source>Nat Med</source>. (<year>2024</year>) <volume>30</volume>:<fpage>2549</fpage>&#x2013;<lpage>57</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41591-024-03064-w</pub-id>, <pub-id pub-id-type="pmid">38956195</pub-id></mixed-citation></ref>
<ref id="ref7"><label>7.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>van Hagen</surname> <given-names>P</given-names></name> <name><surname>Hulshof</surname> <given-names>MC</given-names></name> <name><surname>van Lanschot</surname> <given-names>JJB</given-names></name> <name><surname>Steyerberg</surname> <given-names>EW</given-names></name> <name><surname>van Berge Henegouwen</surname> <given-names>MI</given-names></name> <name><surname>Wijnhoven</surname> <given-names>BP</given-names></name> <etal/></person-group>. <article-title>Preoperative chemoradiotherapy for esophageal or junctional cancer</article-title>. <source>N Engl J Med</source>. (<year>2012</year>) <volume>366</volume>:<fpage>2074</fpage>&#x2013;<lpage>84</lpage>. doi: <pub-id pub-id-type="doi">10.1056/NEJMoa1112088</pub-id></mixed-citation></ref>
<ref id="ref8"><label>8.</label><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Burgart</surname> <given-names>LJ</given-names></name> <name><surname>Chopp</surname> <given-names>WV</given-names></name> <name><surname>Jain</surname> <given-names>D</given-names></name></person-group>. (<year>2022</year>). <article-title>Protocol for the examination of specimens from patients with carcinoma of the esophagus</article-title>. Available online at: <ext-link xlink:href="https://www.scribd.com/document/733338659/Esophagus-4-2-0-1-REL-CAPCP" ext-link-type="uri">https://www.scribd.com/document/733338659/Esophagus-4-2-0-1-REL-CAPCP</ext-link>. (Accessed March 20, 2023)</mixed-citation></ref>
<ref id="ref9"><label>9.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lerttanatum</surname> <given-names>N</given-names></name> <name><surname>Tharavej</surname> <given-names>C</given-names></name> <name><surname>Chongpison</surname> <given-names>Y</given-names></name> <name><surname>Sanpavat</surname> <given-names>A</given-names></name></person-group>. <article-title>Comparison of tumor regression grading system in locally advanced esophageal squamous cell carcinoma after preoperative radio-chemotherapy to determine the most accurate system predicting prognosis</article-title>. <source>J Gastrointest Oncol</source>. (<year>2019</year>) <volume>10</volume>:<fpage>276</fpage>&#x2013;<lpage>82</lpage>. doi: <pub-id pub-id-type="doi">10.21037/jgo.2018.12.01</pub-id>, <pub-id pub-id-type="pmid">31032095</pub-id></mixed-citation></ref>
<ref id="ref10"><label>10.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>L</given-names></name> <name><surname>Wei</surname> <given-names>XF</given-names></name> <name><surname>Li</surname> <given-names>CJ</given-names></name> <name><surname>Yang</surname> <given-names>ZY</given-names></name> <name><surname>Yu</surname> <given-names>YK</given-names></name> <name><surname>Li</surname> <given-names>HM</given-names></name> <etal/></person-group>. <article-title>Pathologic responses and surgical outcomes after neoadjuvant immunochemotherapy versus neoadjuvant chemoradiotherapy in patients with locally advanced esophageal squamous cell carcinoma</article-title>. <source>Front Immunol</source>. (<year>2022</year>) <volume>13</volume>:<fpage>1052542</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fimmu.2022.1052542</pub-id>, <pub-id pub-id-type="pmid">36466925</pub-id></mixed-citation></ref>
<ref id="ref11"><label>11.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kleppe</surname> <given-names>A</given-names></name> <name><surname>Skrede</surname> <given-names>OJ</given-names></name> <name><surname>De Raedt</surname> <given-names>S</given-names></name> <name><surname>Liest&#x00F8;l</surname> <given-names>K</given-names></name> <name><surname>Kerr</surname> <given-names>DJ</given-names></name> <name><surname>Danielsen</surname> <given-names>HE</given-names></name></person-group>. <article-title>Designing deep learning studies in cancer diagnostics</article-title>. <source>Nat Rev Cancer</source>. (<year>2021</year>) <volume>21</volume>:<fpage>199</fpage>&#x2013;<lpage>211</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41568-020-00327-9</pub-id>, <pub-id pub-id-type="pmid">33514930</pub-id></mixed-citation></ref>
<ref id="ref12"><label>12.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>CL</given-names></name> <name><surname>Chen</surname> <given-names>CC</given-names></name> <name><surname>Yu</surname> <given-names>WH</given-names></name> <name><surname>Chen</surname> <given-names>SH</given-names></name> <name><surname>Chang</surname> <given-names>YC</given-names></name> <name><surname>Hsu</surname> <given-names>TI</given-names></name> <etal/></person-group>. <article-title>An annotation-free whole-slide training approach to pathological classification of lung cancer types using deep learning</article-title>. <source>Nat Commun</source>. (<year>2021</year>) <volume>12</volume>:<fpage>1193</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41467-021-21467-y</pub-id>, <pub-id pub-id-type="pmid">33608558</pub-id></mixed-citation></ref>
<ref id="ref13"><label>13.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>L</given-names></name> <name><surname>Pan</surname> <given-names>L</given-names></name> <name><surname>Wang</surname> <given-names>H</given-names></name> <name><surname>Liu</surname> <given-names>M</given-names></name> <name><surname>Feng</surname> <given-names>Z</given-names></name> <name><surname>Rong</surname> <given-names>P</given-names></name> <etal/></person-group>. <article-title>DHUnet: dual-branch hierarchical global&#x2013;local fusion network for whole slide image segmentation</article-title>. <source>Biomed Signal Process Control</source>. (<year>2023</year>) <volume>85</volume>:<fpage>104976</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.bspc.2023.104976</pub-id></mixed-citation></ref>
<ref id="ref14"><label>14.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Theocharopoulos</surname> <given-names>C</given-names></name> <name><surname>Davakis</surname> <given-names>S</given-names></name> <name><surname>Ziogas</surname> <given-names>DC</given-names></name> <name><surname>Theocharopoulos</surname> <given-names>A</given-names></name> <name><surname>Foteinou</surname> <given-names>D</given-names></name> <name><surname>Mylonakis</surname> <given-names>A</given-names></name> <etal/></person-group>. <article-title>Deep learning for image analysis in the diagnosis and management of esophageal cancer</article-title>. <source>Cancer</source>. (<year>2024</year>) <volume>16</volume>:<fpage>3285</fpage>. doi: <pub-id pub-id-type="doi">10.3390/cancers16193285</pub-id>, <pub-id pub-id-type="pmid">39409906</pub-id></mixed-citation></ref>
<ref id="ref15"><label>15.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Diao</surname> <given-names>S</given-names></name> <name><surname>Luo</surname> <given-names>W</given-names></name> <name><surname>Hou</surname> <given-names>J</given-names></name> <name><surname>Lambo</surname> <given-names>R</given-names></name> <name><surname>Al-Kuhali</surname> <given-names>HA</given-names></name> <name><surname>Zhao</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>Deep multi-magnification similarity learning for histopathological image classification</article-title>. <source>IEEE J Biomed Health Inform</source>. (<year>2023</year>) <volume>27</volume>:<fpage>1535</fpage>&#x2013;<lpage>45</lpage>. doi: <pub-id pub-id-type="doi">10.1109/JBHI.2023.3237137</pub-id>, <pub-id pub-id-type="pmid">37021898</pub-id></mixed-citation></ref>
<ref id="ref16"><label>16.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Abboretti</surname> <given-names>F</given-names></name> <name><surname>Mantziari</surname> <given-names>S</given-names></name> <name><surname>Didisheim</surname> <given-names>L</given-names></name> <name><surname>Sch&#x00E4;fer</surname> <given-names>M</given-names></name> <name><surname>Teixeira Farinha</surname> <given-names>H</given-names></name></person-group>. <article-title>Prognostic value of tumor regression grade (TRG) after oncological gastrectomy for gastric cancer</article-title>. <source>Langenbecks Arch Surg</source>. (<year>2024</year>) <volume>409</volume>:<fpage>199</fpage>. doi: <pub-id pub-id-type="doi">10.1007/s00423-024-03388-8</pub-id>, <pub-id pub-id-type="pmid">38935163</pub-id></mixed-citation></ref>
<ref id="ref17"><label>17.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tolkach</surname> <given-names>Y</given-names></name> <name><surname>Wolgast</surname> <given-names>LM</given-names></name> <name><surname>Damanakis</surname> <given-names>A</given-names></name> <name><surname>Pryalukhin</surname> <given-names>A</given-names></name> <name><surname>Schallenberg</surname> <given-names>S</given-names></name> <name><surname>Hulla</surname> <given-names>W</given-names></name> <etal/></person-group>. <article-title>Artificial intelligence for tumour tissue detection and histological regression grading in oesophageal adenocarcinomas: a retrospective algorithm development and validation study</article-title>. <source>Lancet Digit Health</source>. (<year>2023</year>) <volume>5</volume>:<fpage>e265</fpage>&#x2013;<lpage>75</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S2589-7500(23)00027-4</pub-id>, <pub-id pub-id-type="pmid">37100542</pub-id></mixed-citation></ref>
<ref id="ref18"><label>18.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y</given-names></name> <name><surname>Zhang</surname> <given-names>W</given-names></name> <name><surname>Chen</surname> <given-names>L</given-names></name> <name><surname>Xie</surname> <given-names>J</given-names></name> <name><surname>Zheng</surname> <given-names>X</given-names></name> <name><surname>Jin</surname> <given-names>Y</given-names></name> <etal/></person-group>. <article-title>Development of an interpretable deep learning model for pathological tumor response assessment after neoadjuvant therapy</article-title>. <source>Biol Proced Online</source>. (<year>2024</year>) <volume>26</volume>:<fpage>10</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12575-024-00234-5</pub-id>, <pub-id pub-id-type="pmid">38632527</pub-id></mixed-citation></ref>
<ref id="ref19"><label>19.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Waqas</surname> <given-names>M</given-names></name> <name><surname>Ahmed</surname> <given-names>SU</given-names></name> <name><surname>Tahir</surname> <given-names>MA</given-names></name> <name><surname>Wu</surname> <given-names>J</given-names></name> <name><surname>Qureshi</surname> <given-names>R</given-names></name></person-group>. <article-title>Exploring multiple instance learning (MIL): a brief survey</article-title>. <source>Expert Syst Appl</source>. (<year>2024</year>) <volume>250</volume>:<fpage>123893</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.eswa.2024.123893</pub-id></mixed-citation></ref>
<ref id="ref20"><label>20.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lu</surname> <given-names>MY</given-names></name> <name><surname>Williamson</surname> <given-names>DFK</given-names></name> <name><surname>Chen</surname> <given-names>TY</given-names></name> <name><surname>Chen</surname> <given-names>RJ</given-names></name> <name><surname>Barbieri</surname> <given-names>M</given-names></name> <name><surname>Mahmood</surname> <given-names>F</given-names></name></person-group>. <article-title>Data-efficient and weakly supervised computational pathology on whole-slide images</article-title>. <source>Nat Biomed Eng</source>. (<year>2021</year>) <volume>5</volume>:<fpage>555</fpage>&#x2013;<lpage>70</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41551-020-00682-w</pub-id>, <pub-id pub-id-type="pmid">33649564</pub-id></mixed-citation></ref>
<ref id="ref21"><label>21.</label><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Y</given-names></name> <name><surname>Li</surname> <given-names>H</given-names></name> <name><surname>Sun</surname> <given-names>Y</given-names></name> <name><surname>Zheng</surname> <given-names>S</given-names></name> <name><surname>Zhu</surname> <given-names>C</given-names></name> <name><surname>Yang</surname> <given-names>L</given-names></name></person-group>. <article-title>Attention-challenging multiple instance learning for whole slide image classification</article-title> In: <person-group person-group-type="editor"><name><surname>Leonardis</surname> <given-names>A</given-names></name> <name><surname>Ricci</surname> <given-names>E</given-names></name> <name><surname>Roth</surname> <given-names>S</given-names></name> <name><surname>Russakovsky</surname> <given-names>O</given-names></name> <name><surname>Sattler</surname> <given-names>T</given-names></name> <name><surname>Varol</surname> <given-names>G</given-names></name></person-group>, editors. <source>Computer vision&#x2014;ECCV 2024. Lecture notes in computer science</source>. <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2025</year>). <fpage>125</fpage>&#x2013;<lpage>43</lpage>.</mixed-citation></ref>
<ref id="ref22"><label>22.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>C</given-names></name> <name><surname>Sun</surname> <given-names>Q</given-names></name> <name><surname>Zhu</surname> <given-names>W</given-names></name> <name><surname>Zhang</surname> <given-names>L</given-names></name> <name><surname>Zhang</surname> <given-names>J</given-names></name> <name><surname>Liu</surname> <given-names>B</given-names></name></person-group>. <article-title>Transformer based multiple instance learning for WSI breast cancer classification</article-title>. <source>Biomed Signal Process Control</source>. (<year>2024</year>) <volume>89</volume>:<fpage>105755</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.bspc.2023.105755</pub-id></mixed-citation></ref>
<ref id="ref23"><label>23.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>X</given-names></name> <name><surname>Zhao</surname> <given-names>J</given-names></name> <name><surname>Marostica</surname> <given-names>E</given-names></name> <name><surname>Yuan</surname> <given-names>W</given-names></name> <name><surname>Jin</surname> <given-names>J</given-names></name> <name><surname>Zhang</surname> <given-names>J</given-names></name> <etal/></person-group>. <article-title>A pathology foundation model for cancer diagnosis and prognosis prediction</article-title>. <source>Nature</source>. (<year>2024</year>) <volume>634</volume>:<fpage>970</fpage>&#x2013;<lpage>8</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41586-024-07894-z</pub-id>, <pub-id pub-id-type="pmid">39232164</pub-id></mixed-citation></ref>
<ref id="ref24"><label>24.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>H</given-names></name> <name><surname>Usuyama</surname> <given-names>N</given-names></name> <name><surname>Bagga</surname> <given-names>J</given-names></name> <name><surname>Zhang</surname> <given-names>S</given-names></name> <name><surname>Rao</surname> <given-names>R</given-names></name> <name><surname>Naumann</surname> <given-names>T</given-names></name> <etal/></person-group>. <article-title>A whole-slide foundation model for digital pathology from real-world data</article-title>. <source>Nature</source>. (<year>2024</year>) <volume>630</volume>:<fpage>181</fpage>&#x2013;<lpage>8</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41586-024-07441-w</pub-id>, <pub-id pub-id-type="pmid">38778098</pub-id></mixed-citation></ref>
<ref id="ref25"><label>25.</label><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>T</given-names></name> <name><surname>Wagner</surname> <given-names>SJ</given-names></name> <name><surname>Song</surname> <given-names>AH</given-names></name> <name><surname>Chen</surname> <given-names>RJ</given-names></name> <name><surname>Lu</surname> <given-names>MY</given-names></name> <name><surname>Zhang</surname> <given-names>A</given-names></name> <etal/></person-group>. (<year>2024</year>) <article-title>Multimodal whole slide foundation model for pathology</article-title>. <italic>arXiv</italic>. Available online at: <ext-link xlink:href="https://doi.org/10.48550/arXiv.2411.19666" ext-link-type="uri">https://doi.org/10.48550/arXiv.2411.19666</ext-link>. [Epub ahead of preprint]</mixed-citation></ref>
<ref id="ref26"><label>26.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sudo</surname> <given-names>N</given-names></name> <name><surname>Ichikawa</surname> <given-names>H</given-names></name> <name><surname>Muneoka</surname> <given-names>Y</given-names></name> <name><surname>Hanyu</surname> <given-names>T</given-names></name> <name><surname>Kano</surname> <given-names>Y</given-names></name> <name><surname>Ishikawa</surname> <given-names>T</given-names></name> <etal/></person-group>. <article-title>Clinical utility of ypTNM stage grouping in the 8th edition of the American Joint Committee on Cancer TNM staging system for esophageal squamous cell carcinoma</article-title>. <source>Ann Surg Oncol</source>. (<year>2021</year>) <volume>28</volume>:<fpage>650</fpage>&#x2013;<lpage>60</lpage>. doi: <pub-id pub-id-type="doi">10.1245/s10434-020-09181-3</pub-id>, <pub-id pub-id-type="pmid">33025354</pub-id></mixed-citation></ref>
<ref id="ref27"><label>27.</label><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Waqas</surname> <given-names>M</given-names></name> <name><surname>Bandyopadhyay</surname> <given-names>R</given-names></name> <name><surname>Showkatian</surname> <given-names>E</given-names></name> <name><surname>Muneer</surname> <given-names>A</given-names></name> <name><surname>Zafar</surname> <given-names>A</given-names></name> <name><surname>Alvarez</surname> <given-names>FR</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>The next layer: augmenting foundation models with structure-preserving and attention-guided learning for local patches to global context awareness in computational pathology</article-title>. <italic>arXiv</italic>. Available online at: <ext-link xlink:href="https://doi.org/10.48550/arXiv.2508.19914" ext-link-type="uri">https://doi.org/10.48550/arXiv.2508.19914</ext-link>. [Epub ahead of preprint]</mixed-citation></ref>
<ref id="ref28"><label>28.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>X</given-names></name> <name><surname>Ren</surname> <given-names>X</given-names></name> <name><surname>Venugopal</surname> <given-names>R</given-names></name></person-group>. <article-title>Entropy measures for quantifying complexity in digital pathology and spatial omics</article-title>. <source>iScience</source>. (<year>2025</year>) <volume>28</volume>:<fpage>112765</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.isci.2025.112765</pub-id>, <pub-id pub-id-type="pmid">40546955</pub-id></mixed-citation></ref>
<ref id="ref29"><label>29.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>X</given-names></name></person-group>. <article-title>Deciphering cell to cell spatial relationship for pathology images using SpatialQPFs</article-title>. <source>Sci Rep</source>. (<year>2024</year>) <volume>14</volume>:<fpage>29585</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-024-81383-1</pub-id>, <pub-id pub-id-type="pmid">39609630</pub-id></mixed-citation></ref>
</ref-list>
<fn-group>
<fn fn-type="custom" custom-type="edited-by" id="fn0002">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/618108/overview">Shuoyu Xu</ext-link>, Bio-totem Pte Ltd., China</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by" id="fn0003">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2986200/overview">Xiao Li</ext-link>, Roche Diagnostics, United States</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3306204/overview">Muhammad Waqas</ext-link>, The University of Texas MD Anderson Cancer Center, United States</p>
</fn>
</fn-group>
<fn-group>
<fn fn-type="abbr" id="abbrev1">
<label>Abbreviations:</label>
<p>EC, Esophageal cancer; NAT, Neoadjuvant therapy; TRG, Tumor regression grade; WSI, Whole-slide images; MIL, Multiple instance learning; VRTCs, Viable residual tumor cells; OS, Overall survival; PFS, Progression-free survival; H&#x0026;E, Hematoxylin and eosin; AUC, Area under the receiver operating characteristic curve; ABMIL, Attention-based multiple instance learning; ACMIL, Adaptive cross-instance multiple instance learning; CLAM, Clustering-constrained attention multiple instance learning; CI, Confidence interval; HR, Hazard ratio.</p>
</fn>
</fn-group>
</back>
</article>