<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article article-type="methods-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1381851</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2024.1381851</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>SAFE-MIL: a statistically interpretable framework for screening potential targeted therapy patients based on risk estimation</article-title>
<alt-title alt-title-type="left-running-head">Guan et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2024.1381851">10.3389/fgene.2024.1381851</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Guan</surname>
<given-names>Yanfang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2801889/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Xue</surname>
<given-names>Zhengfa</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2810723/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Wang</surname>
<given-names>Jiayin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/615156/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Ai</surname>
<given-names>Xinghao</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Rongrong</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/723766/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yi</surname>
<given-names>Xin</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/894505/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Lu</surname>
<given-names>Shun</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1338362/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Liu</surname>
<given-names>Yuqian</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1855422/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Computer Science and Technology</institution>, <institution>Xi&#x2019;an Jiaotong University</institution>, <addr-line>Xi&#x2019;an</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Shaanxi Engineering Research Center of Medical and Health Big Data</institution>, <institution>Xi&#x2019;an Jiaotong University</institution>, <addr-line>Xi&#x2019;an</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Geneplus Beijing Institute</institution>, <addr-line>Beijing</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Shanghai Chest Hospital</institution>, <institution>Shanghai Jiao Tong University School of Medicine</institution>, <addr-line>Shanghai</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1512658/overview">Xiao Li</ext-link>, Shandong Provincial Qianfoshan Hospital, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/293547/overview">Zhongheng Zhang</ext-link>, Sir Run Run Shaw Hospital, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2613306/overview">Zhangyang Ai</ext-link>, Hunan University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Yuqian Liu, <email>yuqian.liu@xjtu.edu.cn</email>; Shun Lu, <email>shunlu@sjtu.edu.cn</email>
</corresp>
<fn fn-type="equal" id="fn001">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work and share first authorship</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>15</day>
<month>08</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1381851</elocation-id>
<history>
<date date-type="received">
<day>04</day>
<month>02</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>31</day>
<month>07</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Guan, Xue, Wang, Ai, Chen, Yi, Lu and Liu.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Guan, Xue, Wang, Ai, Chen, Yi, Lu and Liu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Patients with the target gene mutation frequently derive significant clinical benefits from target therapy. However, differences in the abundance level of mutations among patients resulted in varying survival benefits, even among patients with the same target gene mutations. Currently, there is a lack of rational and interpretable models to assess the risk of treatment failure<bold>.</bold> In this study, we investigated the underlying coupled factors contributing to variations in medication sensitivity and established a statistically interpretable framework, named SAFE-MIL, for risk estimation. We first constructed an effectiveness label for each patient from the perspective of exploring the optimal grouping of patients&#x2019; positive judgment values and sampled patients into 600 and 1,000 groups, respectively, based on multi-instance learning (MIL). A novel and interpretable loss function was further designed based on the Hosmer-Lemeshow test for this framework. By integrating multi-instance learning with the Hosmer-Lemeshow test, SAFE-MIL is capable of accurately estimating the risk of drug treatment failure across diverse patient cohorts and providing the optimal threshold for assessing the risk stratification simultaneously. We conducted a comprehensive case study involving 457 non-small cell lung cancer patients with EGFR mutations treated with EGFR tyrosine kinase inhibitors. Results demonstrate that SAFE-MIL outperforms traditional regression methods with higher accuracy and can accurately assess patients&#x2019; risk stratification. This underscores its ability to accurately capture inter-patient variability in risk while providing statistical interpretability. SAFE-MIL is able to effectively guide clinical decision-making regarding the use of drugs in targeted therapy and provides an interpretable computational framework for other patient stratification problems. The SAFE-MIL framework has proven its effectiveness in capturing inter-patient variability in risk and providing statistical interpretability. It outperforms traditional regression methods and can effectively guide clinical decision-making in the use of drugs for targeted therapy. SAFE-MIL offers a valuable interpretable computational framework that can be applied to other patient stratification problems, enhancing the precision of risk assessment in personalized medicine. The source code for SAFE-MIL is available for further exploration and application at <ext-link ext-link-type="uri" xlink:href="https://github.com/Nevermore233/SAFE-MIL">https://github.com/Nevermore233/SAFE-MIL</ext-link>.</p>
</abstract>
<kwd-group>
<kwd>EGFR</kwd>
<kwd>non-small cell lung cancer</kwd>
<kwd>target therapy</kwd>
<kwd>risk estimation</kwd>
<kwd>Hosmer-Lemeshow test</kwd>
<kwd>multi-instance learning</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Lung cancer holds the highest incidence rate among all types of cancer and boasts the largest selection of approved targeted therapeutic agents. Consequently, targeted therapy for lung cancer has become a prevalent and routine treatment in clinical practice. However, patients with target mutations show heterogeneous and diminished responses to the treatment (<xref ref-type="bibr" rid="B7">Cheng et al., 2020</xref>). Several studies have shown that its effectiveness is related to the gene mutation abundance level (<xref ref-type="bibr" rid="B3">Blakely et al., 2017</xref>; <xref ref-type="bibr" rid="B29">Samstein et al., 2019</xref>). Higher mutation abundance level may be related to better treatment outcomes (<xref ref-type="bibr" rid="B44">Zhou et al., 2011</xref>; <xref ref-type="bibr" rid="B36">Yan et al., 2019</xref>; <xref ref-type="bibr" rid="B35">Wang et al., 2021</xref>; <xref ref-type="bibr" rid="B20">Liu et al., 2022</xref>), such as a longer median survival time (<xref ref-type="bibr" rid="B20">Liu et al., 2022</xref>). In the event of a low abundance level, the therapeutic effect of targeted drugs may be reduced or completely lost (<xref ref-type="bibr" rid="B33">Tang et al., 2021</xref>). However, this issue remains unresolved in clinical practice. It is due to the absence of statistically interpretable tools that can accurately assess the risk of treatment failure in patients receiving targeted therapy. Furthermore, the optimal stratification threshold for targeted therapy based on mutation abundance level is unknown among the patient cohort. Hence, there is an urgent and unmet clinical need to develop a broadly applicable and statistically interpretable method approach capable of effectively assessing the risk associated with medicine utilization and then give the optimal threshold to assist clinical decision-making in target therapy.</p>
<p>Performing drug screening and selecting appropriate personalized treatment based on individual genomic, proteomic, and clinical features is one of the paramount goals of precision medicine (<xref ref-type="bibr" rid="B22">Nemati et al., 2018</xref>; <xref ref-type="bibr" rid="B42">Zhang et al., 2020</xref>; <xref ref-type="bibr" rid="B2">Banerjee et al., 2021</xref>; <xref ref-type="bibr" rid="B32">Sotudian and Paschalidis, 2022</xref>; <xref ref-type="bibr" rid="B11">Diao et al., 2023</xref>; <xref ref-type="bibr" rid="B6">Chen et al., 2024</xref>). Various clinical trials generate high quality results on effectiveness comparison of different drugs for the treatment of the same disease which is crucial for drug development and clinical practice (<xref ref-type="bibr" rid="B27">Rubin and Gilliland, 2012</xref>; <xref ref-type="bibr" rid="B40">Zhang Y. et al., 2021</xref>). Clinically, patients harboring same target gene mutations have different medication risks due to differences in mutation abundance level and drug sensitivity (<xref ref-type="bibr" rid="B26">Robichaux et al., 2021</xref>; <xref ref-type="bibr" rid="B35">Wang et al., 2021</xref>). The use of inappropriate drug regimens in patients with high risk may lead to delays in the patient&#x2019;s condition and additional medical costs. One of the results is that mild patients may suffer severe or even catastrophic lesions (<xref ref-type="bibr" rid="B30">Schnipper et al., 2015</xref>; <xref ref-type="bibr" rid="B9">Daoud et al., 2020</xref>). Therefore, it is very necessary to assess the risk of medication use and utilize it in the clinical.</p>
<p>Nevertheless, this problem is different from the conventional drug effectiveness prediction problem. With the rapid development of artificial intelligence, a multitude of computational approaches to drug effectiveness prediction have been developed (<xref ref-type="bibr" rid="B15">G&#xf6;ttlich et al., 2016</xref>; <xref ref-type="bibr" rid="B5">Chang et al., 2022</xref>; <xref ref-type="bibr" rid="B25">Peng et al., 2022</xref>; <xref ref-type="bibr" rid="B21">&#x141;osi&#x144;ska et al., 2022</xref>; <xref ref-type="bibr" rid="B37">Yang et al., 2023</xref>). Most of these approaches focus on utilizing the powerful feature extraction abilities and learning capabilities of artificial intelligence models to predict the drug responses of patients. Various computational models were employed to obtain potential vector representations of drugs and diseases for drug effectiveness prediction. However, a deep learning model is a black box, posing challenges in elucidating the acquired knowledge and the underlying principles guiding its predictive capabilities (<xref ref-type="bibr" rid="B19">Kuenzi et al., 2020</xref>). The existing drug effectiveness prediction models exhibit a tendency towards excessive complexity and lack relevant statistical interpretations, such as deep neural networks. According to Food and Drug Administration (FDA) policy, the assessment needs to be conducted in a manner that allows for statistical interpretation. Hence, this is a risk likelihood estimation problem rather than a prediction problem. Furthermore, the patient stratification problem is a matter of interest for both the clinical and drug regulatory communities, rather than the drug effectiveness prediction. Therefore, these existing drug effectiveness prediction models have high predictive accuracy; yet, their practical applicability in clinic practice is limited. Exploring the challenge of designing a reasonable model from the perspective of optimal patient stratification to consider the risk of treatment failure holds significant academic value.</p>
<p>The widespread application of multiple instance learning (MIL) in drug-target, drug activity, and drug effectiveness prediction (<xref ref-type="bibr" rid="B14">Fu et al., 2012</xref>; <xref ref-type="bibr" rid="B43">Zhao et al., 2013</xref>; <xref ref-type="bibr" rid="B28">Saberian et al., 2022</xref>; <xref ref-type="bibr" rid="B34">Wang et al., 2022</xref>) has attracted our attention. In clinical practice, the risk of drug failure is typically assessed on a population-wide scale, making it challenging to provide a personalized estimate of treatment failure risk for an individual patient prior to the initiation of medication. In MIL, the dataset is divided into groups of multiple instances, with the group labels known but the instance labels unknown, which has a good correlation with patient grouping in clinical drug practice. For example, suppose there are many microscopic tissue slice images used to detect cancer. Each image may contain multiple small regions, which we refer to as &#x201c;instances.&#x201d; The whole image can be labeled as &#x201c;cancer&#x201d; or &#x201c;no cancer,&#x201d; but it may be uncertain whether each specific small region contains cancer cells. In this case, we can consider each image as a &#x201c;group&#x201d; containing multiple &#x201c;instance regions.&#x201d; The algorithm needs to predict the label of the &#x201c;group&#x201d; based on the known labels of the &#x201c;instances.&#x201d; In the context of drug effectiveness prediction, using MIL to consider and model drug effectiveness may be a feasible idea.</p>
<p>This paper presents a generic novel framework termed SAFE-MIL for screening potential targeted therapy patients based on risk estimation, using epidermal growth factor receptor (EGFR)-mutant non-small cell lung cancer (NSCLC) as an example. The proposed framework integrates multi-instance learning with the Hosmer-Lemeshow test (<xref ref-type="bibr" rid="B16">Hosmer and Lemesbow, 1980</xref>) to estimate the risk of drug use, which provides statistical interpretability and can address the limitations of traditional methods. Experimental results demonstrate that SAFE-MIL provides interpretability and higher accuracy, hence potentially facilitating the process of therapeutic drug selection for patients.</p>
</sec>
<sec sec-type="patients|methods" id="s2">
<title>2 Patients and methods</title>
<p>In this paper, we present a spontaneous formulation of the drug risk estimation problem as an instance of multi-instance learning. We address the problem in three stages, first organizing the given clinical information and drug action information of patients into instances, then identifying the estimated scores of all instances in the same group, and finally combining all the estimated scores as the output. Prior to these stages, data preprocessing was performed.</p>
<sec id="s2-1">
<title>2.1 Patients and samples collection</title>
<p>In this research, we acquired three sets of Non-Small Cell Lung Cancer (NSCLC) sequencing data, exclusively sourced from EGFR-mutated NSCLC patients. Inclusion criteria for patients in this study: Patients must carry EGFR mutations (EGFRm) and be clinically selected for anti-tumor treatment with a tyrosine kinase inhibitor (TKI) based on the detection results. Exclusion criteria is that patients who have not received EGFR-TKI treatment according to the detection results, or patients carrying mutations in other drug target sites. The first batch consists of data from 100 patients with treatment-naive NSCLC (stage III&#x2013;IV), all of whom underwent first-line targeted therapy. This information was gathered from 14 medical centers spanning the period from 23 February 2017, to 31 December 2019. The trial has been officially registered with the identifier NCT03059641. The second and third batches comprises data from 237 patients and 120 patients (stage IV) with EGFR-mutated NSCLC who underwent next-generation sequencing at Geneplus-Beijing (Beijing, China) between 2016 and December 2019 (<xref ref-type="table" rid="T1">Table 1</xref>). Tumor tissue samples were utilized to identify actionable mutations for targeted therapy in these patients. Notably, all these patients received anti-EGFR targeted therapy. It&#x2019;s important to note that all participants in the study provided written informed consent. The research was conducted in accordance with a protocol approved by the Institutional Review Board of Shanghai Chest Hospital. Cohort 1 is the first batch of NSCLC patients were collected. In order to better generate multi-instance datasets, we conducted experiments using cohort 1 as independent data.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Demographic and baseline characteristics of cohort-1,2,3.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th colspan="7" align="left">Participants (n &#x3d; 457)</th>
</tr>
<tr>
<th align="left"/>
<th colspan="2" align="left">Cohort1(n &#x3d; 100)</th>
<th colspan="2" align="left">Cohort2 (n &#x3d; 237)</th>
<th colspan="2" align="left">Cohort3(n &#x3d; 120)</th>
</tr>
<tr>
<th align="left">Gender</th>
<th align="left">NO.</th>
<th align="left">%</th>
<th align="left">NO.</th>
<th align="left">%</th>
<th align="left">NO.</th>
<th align="left">%</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">&#x2003;Female</td>
<td align="left">64</td>
<td align="left">64</td>
<td align="left">142</td>
<td align="left">60</td>
<td align="left">68</td>
<td align="left">57</td>
</tr>
<tr>
<td align="left">&#x2003;Male</td>
<td align="left">36</td>
<td align="left">36</td>
<td align="left">95</td>
<td align="left">40</td>
<td align="left">52</td>
<td align="left">43</td>
</tr>
<tr>
<td align="left">Age (26&#x2013;86)</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left"/>
</tr>
<tr>
<td align="left">&#x2003;&#x3e;60</td>
<td align="left">56</td>
<td align="left">56</td>
<td align="left">111</td>
<td align="left">47</td>
<td align="left">58</td>
<td align="left">48</td>
</tr>
<tr>
<td align="left">&#x2003;&#x3c;&#x3d;60</td>
<td align="left">44</td>
<td align="left">44</td>
<td align="left">126</td>
<td align="left">53</td>
<td align="left">62</td>
<td align="left">52</td>
</tr>
<tr>
<td align="left">Stage</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left"/>
</tr>
<tr>
<td align="left">&#x2003;III</td>
<td align="left">10</td>
<td align="left">10</td>
<td align="left">0</td>
<td align="left">0</td>
<td align="left">0</td>
<td align="left">0</td>
</tr>
<tr>
<td align="left">&#x2003;IV</td>
<td align="left">90</td>
<td align="left">90</td>
<td align="left">237</td>
<td align="left">100</td>
<td align="left">120</td>
<td align="left">100</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Cohort 3 comprised 120 patients with NSCLC accepted EGFR-TKI target therapy (43.3% male, stage IV) (<xref ref-type="sec" rid="s12">Supplementary Table S5</xref>), with ages at diagnosis ranging from 35 to 83&#xa0;years. In the tumor samples, a total of 952 SVs are included (<xref ref-type="sec" rid="s12">Supplementary Table S6</xref>). All patients were found to carry EGFR-sensitive mutations. Cohorts 1 and 2, 3 have recorded slightly different survival information, with the former documenting patients&#x2019; progression-free survival (PFS), while the latter records time to treatment failure (TTF).</p>
<p>DNA extraction, targeted capture, and NGS Genetic analysis for all data were performed as previously described (<xref ref-type="bibr" rid="B23">Nong et al., 2018</xref>; <xref ref-type="bibr" rid="B39">Zhang et al., 2019</xref>). Sequencing libraries were prepared from genomic DNA were prepared using Illumina TruSeq DNA Library Preparation Kits (Illumina) or MGIEasy Universal Library Prep Set (MGI Tech). Libraries were hybridized to custom-designed biotinylated oligonucleotide probes (Integrated DNA Technologies, Inc) targeting 1,021 genes. Prepared libraries were sequenced on a NextSeq CN 500 (Illumina) or MGISEQ-2000 sequencer (MGI Tech, Shenzhen, China). After the entire run was completed, image analyses, error estimation and base calling were performed to generate primary data. We then removed a few unqualified sequences from the primary data using a local dynamic programming algorithm, which included low-quality reads, defined as reads that contained more than 10 percent Ns in the read length, 50% reads with a quality value of less than five and with an average quality of less than 10 and adapter sequences including indexed sequence. The remaining sequences were termed as clean reads for further analysis. The clean reads were aligned to the reference human genome (hg19) with Burrows-Wheeler Aligner (version 0.7.12-r1039). Variants were called with GATK (version 3.4&#x2013;46-gbc02625) and MuTect (version 1.1.4). Contra (v2.0.8) was used to detect copy-number variants (CNVs), and NCsv (in-house software version 0.2.3) was used to detect structural variants (SVs). Targeted capture sequencing required a minimal mean effective depth of coverage of 300 in tumor tissues Variants were filtered to exclude synonymous variants, known germline variants in dbSNP, and variants that occur at a population abundance level of &#x3e;1% in the Exome Sequencing Project.</p>
</sec>
<sec id="s2-2">
<title>2.2 The model and algorithm</title>
<p>The SAFE-MIL framework was used, which utilizes multiple instances learning to predict drug failure risk in patients. The main steps of the SAFE-MIL framework include:<list list-type="simple">
<list-item>
<p>i. Designing drug effectiveness labels based on clinical features of patients.</p>
</list-item>
<list-item>
<p>ii. Building a model using multiple instances learning to predict drug failure risk by aggregating the labels of each patient.</p>
</list-item>
<list-item>
<p>iii. Using a new loss function based on the Hosmer-lemeshow test to capture differences between groups and improve the performance and statistical interpretability of the model.</p>
</list-item>
<list-item>
<p>iv. Learn the relationship between mutation abundance level and drug failure risk, and determine the optimal positive threshold for drug failure.</p>
</list-item>
</list>
</p>
<p>Cohort 1 is the first batch of NSCLC patients we collected. In order to better generate multi-instance datasets, we conducted experiments using cohort 1 as independent data. We randomly divided cohort 1 into a training set and a testing set and conducted independent experiments. Similarly, we subsequently collected more samples (cohort 2 and cohort 3) and conducted similar independent experiments. Due to slight differences in collection time, clinical features, and other aspects among these three cohorts, we did not merge them.</p>
</sec>
<sec id="s2-3">
<title>2.3 The design of drug effectiveness labels for patients</title>
<p>In order to better illustrate the variations in therapeutic efficacy among different patients, we have considered representing the therapeutic efficacy of a drug as a probability value ranging from 0 to 1. Initially, we evaluated the correlation between patients&#x2019; clinical characteristics and the effectiveness of the drug based on existing literature. Given that a patient&#x2019;s drug response is closely linked to their PFS (<xref ref-type="bibr" rid="B24">Paz-Ares et al., 2018</xref>), we utilized the patients&#x2019; PFS to determine the drug&#x2019;s effectiveness for each individual of cohort 1. For cohort 2, we attempted to assess the efficacy of the drug using TTF. To begin, patients with a PFS/TTF exceeding 8&#xa0;months as of the cut-off value were included in the experimental cohort. Subsequently, we employed a min-max scaling technique to map the PFS/TTF values of these patients onto a scale of 0&#x2013;1. This scaled value was then used to represent the probability of the patient&#x2019;s drug efficacy.</p>
</sec>
<sec id="s2-4">
<title>2.4 Data preprocessing and patient grouping</title>
<p>The data preprocessing process is shown in <xref ref-type="fig" rid="F2">Figure 2A</xref>. Given that the efficacy of a medication for an individual patient remains unobservable prior to treatment, we propose an unsupervised clustering approach to categorize patients with analogous characteristics into distinct cohorts. Subsequently, we employ multi-instance learning to estimate the probability of drug effectiveness across these patient groups. We used the K-means unsupervised clustering algorithm to group patients into four clusters (details see <xref ref-type="sec" rid="s12">Supplementary Materials S1, S2</xref>, <xref ref-type="sec" rid="s12">Figure S1</xref>). We then randomly sampled 1&#x2013;10 patients with replacement to form a group, with each patient serving as an instance. We created datasets based on the two aforementioned patient cohorts. For each cohort, we generated two datasets of varying sizes to demonstrate the model&#x2019;s generalization capacity. One dataset comprised 600 bags (cohort 1&#x2013;600 and cohort 2&#x2013;600), with 150 bags randomly generated from each cluster. The other dataset included 1,000 bags (cohort 2&#x2013;1,000 and cohort 2&#x2013;1,000), with 250 bags randomly generated from each cluster. The drug effectiveness of the groups was determined by calculating the average value of the drug effectiveness label assigned to the instances in the group. The raw input data, preprocessed output data, and grouped data can be accessed at GitHub repository.</p>
</sec>
<sec id="s2-5">
<title>2.5 Objective function</title>
<p>Here, risk estimation is formulated as a regression problem. We combine multiple instances learning and Hosmer-lemeshow test to design our model. Suppose that the training set consists of <italic>N</italic> training groups {<italic>G</italic>
<sub>
<italic>1</italic>
</sub>, <italic>G</italic>
<sub>
<italic>2</italic>
</sub>, &#x2026;, <italic>G</italic>
<sub>
<italic>N</italic>
</sub>} and each group contains <italic>M</italic>
<sub>
<italic>i</italic>
</sub> instances {<italic>G</italic>
<sub>
<italic>i1</italic>
</sub>, <italic>G</italic>
<sub>
<italic>i2</italic>
</sub>, &#x2026;, <italic>G</italic>
<sub>
<italic>iMi</italic>
</sub>}, where the label corresponding to each training group <italic>G</italic>
<sub>
<italic>i</italic>
</sub> is marked <italic>L</italic>
<sub>
<italic>i</italic>
</sub>. In addition, each instance in the groups is a <italic>p</italic>-dimensional attribute value vector.</p>
<p>In multi-instance learning, the task is to predict the labels of unseen groups. The learning algorithm can acquire the label of the group, but cannot obtain the label of the instance. Therefore, we define the global error function of the neural network at the group level using the labels of the training groups as<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:msubsup>
</mml:mstyle>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>Where <italic>E</italic>
<sub>
<italic>i</italic>
</sub> is the output error corresponding to the group <italic>G</italic>
<sub>
<italic>i</italic>
</sub> (<xref ref-type="disp-formula" rid="e1">Equation 1</xref>).</p>
<p>To enhance the statistical interpretability of our framework and establish trust between clinicians and our framework, we consider designing loss functions based on statistical tests. We found that the Hosmer-Lemeshow test (<xref ref-type="bibr" rid="B16">Hosmer and Lemesbow, 1980</xref>), which evaluates the goodness-of-fit of grouped data, is closely related to multiple instance learning. The Hosmer-Lemeshow test is widely used in the evaluation of risk models (<xref ref-type="bibr" rid="B18">Kramer and Zimmerman, 2007</xref>; <xref ref-type="bibr" rid="B38">Zhang L. et al., 2021</xref>; <xref ref-type="bibr" rid="B10">Davies et al., 2022</xref>). Given the excellent performance of the Hosmer-Lemeshow test in model verification, we design the loss function according to the calculation formula of Hosmer-Lemeshow (HL) statistic (<italic>HL</italic>
<sub>
<italic>s</italic>
</sub>) (<xref ref-type="disp-formula" rid="e2">Equation 2</xref>). The Hosmer-Lemeshow test is a statistical method used to evaluate the goodness-of-fit for binary logistic regression models. The basic idea is to divide the data into several groups and then compare the actual observed values with the predicted values in each group. In the Hosmer-Lemeshow test, the <italic>p</italic>-value is calculated based on the chi-squared distribution, corresponding to the Hosmer-Lemeshow statistic&#x2019;s cumulative distribution function value. If the <italic>p</italic>-value is very small (less than 0.05), it indicates that the model does not fit well, meaning there is a significant difference between the predicted probabilities and the actual observed values. If the <italic>p</italic>-value is large, it indicates that the model fits well, meaning there is no significant difference between the predicted probabilities and the actual observed values. Suppose that the observed value (<italic>Observed.A</italic>, <italic>Observed.not.A</italic>) and expected values (<italic>Expected.A</italic>, <italic>Expected.not.A</italic>) of <italic>Event A</italic> are divided into Q groups {<italic>G</italic>
<sub>
<italic>1</italic>
</sub>, <italic>G</italic>
<sub>
<italic>2</italic>
</sub>, &#x2026; , <italic>G</italic>
<sub>
<italic>Q</italic>
</sub>}, the HL statistic and <italic>p</italic>-value (<italic>HL</italic>
<sub>
<italic>p</italic>
</sub>) of HL test are defined as<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>Q</mml:mi>
</mml:msubsup>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>A</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>A</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
<disp-formula id="e3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>q</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>Q</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <italic>chisq</italic> (<italic>HL</italic>
<sub>
<italic>s</italic>
</sub>, <italic>Q</italic>-2) is the chi-square distribution with <italic>Q</italic>-2 degrees of freedom (<xref ref-type="disp-formula" rid="e3">Equation 3</xref> more details in <xref ref-type="sec" rid="s12">Supplementary Material S3</xref>).</p>
<p>Since the HL based loss function (HL loss) is partially non-differentiable, we use the mean square error loss function to replace the non-differentiable points of the HL loss function (<xref ref-type="disp-formula" rid="e4">Equation 4</xref>). Research indicates that the output value of a group in neural network is determined by the maximum output value of the instance in the group (<xref ref-type="bibr" rid="B1">Amar et al., 2001</xref>). Therefore, to simulate the above rules, the output error of group <italic>G</italic>
<sub>
<italic>i</italic>
</sub> is defined as<disp-formula id="e4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="&#x7c;">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:munder>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mtext>&#x2009;</mml:mtext>
<mml:munder>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2260;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:munder>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mtext>&#x2009;</mml:mtext>
<mml:munder>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <italic>o</italic>
<sub>
<italic>ij</italic>
</sub> is the network output corresponding to the instance <italic>G</italic>
<sub>
<italic>ij</italic>
</sub>.</p>
<p>The input of the model is all the training groups, and the output is the drug effective probability of the patients in the group, which is used to measure the patient&#x2019;s drug use risk (1- drug effective probability). The hyperparameters used for SAFE-MIL are summarized in <xref ref-type="table" rid="T2">Table 2</xref>. In addition, we explored the correlation between the risk stratification and the survival benefits of the patients. In all analyses, <italic>P</italic> &#x3c; 0.05 was considered statistically significant.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Hyperparameter of the SAFE-MIL.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Hyper-parameters</th>
<th align="left">Setting</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Learning rate</td>
<td align="left">0.001</td>
</tr>
<tr>
<td align="left">Epochs</td>
<td align="left">200</td>
</tr>
<tr>
<td align="left">Optimizer</td>
<td align="left">Relu</td>
</tr>
<tr>
<td align="left">Hidden layer neuron</td>
<td align="left">50</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In addition, after training the model, we search for the optimal positive threshold that can differentiate the drug use risk of patients based on the model&#x2019;s outputs on the training set. Firstly, we define the search space for the optimal positive threshold as [<inline-formula id="inf1">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>- <italic>margin</italic>, &#x2b; <inline-formula id="inf2">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> <italic>margin</italic>], where MA represents the mutation abundance and the margin boundaries determine the size of the search space (<xref ref-type="disp-formula" rid="equ1">Equation 5</xref>). The objective function of optimal positive threshold is defined as follows:<disp-formula id="equ1">
<mml:math id="m7">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:munder>
<mml:mi mathvariant="italic">max</mml:mi>
<mml:mi>t</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>t</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ2">
<mml:math id="m8">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>Where <inline-formula id="inf3">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the drug failure risk, and <inline-formula id="inf4">
<mml:math id="m10">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the optimal positive threshold. When t is set to <inline-formula id="inf5">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the samples are divided into two groups based on <inline-formula id="inf6">
<mml:math id="m12">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and the average drug failure risk difference between them is the largest.</p>
</sec>
<sec id="s2-6">
<title>2.6 Baselines</title>
<p>We evaluated SAFE-MIL with the four classical regression loss functions, including mean absolute error loss (MAE loss) (<xref ref-type="bibr" rid="B12">Fisher, 1915</xref>) (<xref ref-type="disp-formula" rid="e6">Equation 6</xref>), mean square error loss (MSE loss) (<xref ref-type="disp-formula" rid="e7">Equation 7</xref>) (<xref ref-type="bibr" rid="B13">Fisher, 1921</xref>), Huber loss (<xref ref-type="disp-formula" rid="e8">Equation 8</xref>) (<xref ref-type="bibr" rid="B17">Huber, 1992</xref>) and Log-cosh loss (<xref ref-type="disp-formula" rid="e9">Equation 9</xref>) (<xref ref-type="bibr" rid="B31">Shen et al., 2018</xref>). During one neural network iteration, the formulas of the four regression loss functions are defined as<disp-formula id="e6">
<mml:math id="m13">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="e7">
<mml:math id="m14">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
<disp-formula id="e8">
<mml:math id="m15">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="&#x7c;">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>f</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mi>&#x3b4;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
<disp-formula id="e9">
<mml:math id="m16">
<mml:mrow>
<mml:mi mathvariant="italic">Log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>cosh</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="italic">cosh</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>where <italic>y</italic>
<sub>
<italic>p</italic>
</sub> is the predicted value of the group, and <italic>y</italic>
<sub>
<italic>t</italic>
</sub> is the true value of the group. In Huber loss, the value of <inline-formula id="inf7">
<mml:math id="m17">
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is determined by cross-validation. In addition to evaluating the model using the mean square error (MSE), we also use the HL statistic and its corresponding <italic>p</italic>-value to test the accuracy and rationality of the predicted value. In the HL test, if the <italic>p</italic>-value is greater than 0.05, it means that the model has passed the HL test, which means that there is no significant difference between the predicted value and the real value, otherwise it means that the model fit is poor. The complete calculation process of the risk based on the SAFE-MIL framework is given in <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref>.</p>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1</label>
<p>SAFE-MIL.<list list-type="simple">
<list-item>
<p>
<bold>Input</bold>: <italic>N</italic> training groups{<italic>G</italic>
<sub>
<italic>1</italic>
</sub>, <italic>G</italic>
<sub>
<italic>2</italic>
</sub>,&#x2026;,<italic>G</italic>
<sub>
<italic>N</italic>
</sub>}.</p>
</list-item>
<list-item>
<p>
<bold>Output</bold>: Drug failure risk of <italic>N</italic> groups {<italic>D</italic>
<sub>
<italic>1</italic>
</sub>, <italic>D</italic>
<sub>
<italic>2</italic>
</sub>, &#x2026; , <italic>D</italic>
<sub>
<italic>N</italic>
</sub>}, Q1, Q2, Q3.</p>
</list-item>
<list-item>
<p>SAFE-MIL (<italic>Epochs, Threshold</italic>)</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Initialize neural network <italic>Net</italic>;</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;for (<italic>epoch&#x3d;1; epoch&#x3c;&#x3d;Epochs; epoch&#x2b;&#x2b;</italic>)</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2002;<italic>GlobalErr</italic>&#x3d;0; //Set the initial value of global error to be zero</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2002;for (<italic>i&#x3d;1; i&#x3c;&#x3d;N; i&#x2b;&#x2b;</italic>)</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;Compute the output error <italic>E</italic>
<sub>
<italic>i</italic>
</sub> of group <italic>B</italic>
<sub>
<italic>i</italic>
</sub> according to Equation <xref ref-type="disp-formula" rid="e2">2</xref>;</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<italic>GlobalErr &#x3d; GlobalErr&#x2b;E</italic>
<sub>
<italic>i</italic>
</sub>;</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;The weights in <italic>Net</italic> are modified according to <italic>E</italic>
<sub>
<italic>i</italic>
</sub> and the</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;weight-updated rule of BP algorithm;</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2002;end</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2002;If (<italic>GlobalErr&#x3c;&#x3d;Threshold</italic>)</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2002;&#x2003;&#x2003;return <italic>Net</italic>;</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2002;end</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;end</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;return <italic>Net</italic>;</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Patient characteristics and somatic variation detection</title>
<p>Cohort 1 included 100 patients with NSCLC accepted first-line EGFR-TKI target therapy (36% male, 90% stage IV) (<xref ref-type="sec" rid="s12">Supplementary Table S1</xref>). The age at diagnosis ranged from 33 to 80&#xa0;years, with a median of 61&#xa0;years (<xref ref-type="table" rid="T1">Table 1</xref>). In the tumor samples, a total of 592 somatic alterations were detected, including 539 SNVs (single nucleotide variants, SNVs) and small indels (insertion and deletions, indels), 50 copy number variants (CNVs), and 1 structural variants (SVs) (<xref ref-type="sec" rid="s12">Supplementary Table S2</xref>). Cohort 2 comprised 237 patients with NSCLC accepted EGFR-TKI target therapy (40% male, stage IV) (<xref ref-type="sec" rid="s12">Supplementary Table S3</xref>), with ages at diagnosis ranging from 26 to 86&#xa0;years and a median age of 60&#xa0;years (<xref ref-type="table" rid="T1">Table 1</xref>). In the tumor samples, a total of 1738 somatic alterations were detected, including 1,454 SNVs, 268 CNVs, and 4 SVs (<xref ref-type="sec" rid="s12">Supplementary Table S4</xref>). Cohort 3 comprised 120 patients with NSCLC accepted EGFR-TKI target therapy (43.3% male, stage IV) (<xref ref-type="sec" rid="s12">Supplementary Table S5</xref>), with ages at diagnosis ranging from 35 to 83&#xa0;years (<xref ref-type="table" rid="T1">Table 1</xref>). In the tumor samples, a total of 930 SNVs or Indels, and 12 structural variants (SVs) were detected (<xref ref-type="sec" rid="s12">Supplementary Table S6</xref>). All patients were found to carry EGFR-sensitive mutations. Cohorts 1 and 2, 3 have recorded slightly different survival information, with the cohort one documenting patients&#x2019; progression-free survival (PFS), while the cohort 2/3 records time to treatment failure (TTF). In our study, both PFS and TTF were used as evaluation criteria to assess the predictive performance of our algorithm. These two outcomes are related, they showed a high correlation in NSCLC of targeted therapy, immunotherapy, and chemotherapy. Especially in targeted therapies of EGFRm-TKI, the correlation is even higher (r &#x3d; 0.91, 95% CI 0.90,0.92) (<xref ref-type="bibr" rid="B4">Blumenthal et al., 2019</xref>).While PFS focuses specifically on disease progression, TTF offers a broader perspective by considering various factors influencing treatment continuation. The study utilized clinical data from patients and somatic alterations for the SAFE-MIL investigation. The study workflow is illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Flow of participants in the study.</p>
</caption>
<graphic xlink:href="fgene-15-1381851-g001.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 The result of SAFE-MIL for predicting EGFR-TKI failure risk in patients with EGFR-mutated NSCLC</title>
<p>The workflow of this study is shown in <xref ref-type="fig" rid="F2">Figure 2B</xref>. We collected clinical data and SNV mutation data of 457 EGFR-mutated NSCLC patients and conducted data preprocessing. Following the mentioned criteria from the preceding text, we calculated the drug effectiveness label for each patient. Subsequently, an unsupervised clustering of patients was performed using the k-means algorithm based on the features including age, gender, mutation abundance, TP53 co-mutated or not, and the number of co-mutations, which were outlined in <xref ref-type="table" rid="T3">Table 3</xref>. Consequently, we normalized these features using min-max normalization and employed them for unsupervised clustering using the k-means algorithm. After the feature selection process, mutation abundance was identified as the most significant feature and selected as a key predictor. Although the presence of co-TP53 mutation and the number of co-mutations are two important genomic molecular features, but in this study the two features were indicated not important. The reasons may be from two aspects: firstly, the enrolled population was exclusively EGFR-mutant, with a relatively homogeneous molecular subtype; secondly, the data used in this study originated from a panel sequencing of &#x223c; 1Mb, which may have limited the representation of the feature regarding the number of co-mutations compared to whole-exome sequencing data. Finally, we randomly selected 1 to 10 patients from each cluster using a with-replacement sampling method to generate groups. The drug effectiveness label of each group was determined by the average effectiveness label of the instances within the group.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The flowchart of SAFE-MIL. <bold>(A)</bold> Data preprocessing process. First, cluster patients into 4 clusters using unsupervised learning. Then, randomly select 1&#x2013;10 patients from each cluster to form a group, with each patient serving as an instance. <bold>(B)</bold> SAFE-MIL workflow. SAFE-MIL learns the representation of drug failure risk for each patient group through the HL loss function and determines the optimal positive threshold based on mutation abundance level.</p>
</caption>
<graphic xlink:href="fgene-15-1381851-g002.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>The input features of SAFE-MIL.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Feature</th>
<th align="left">Data type</th>
<th align="left">Description</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Age</td>
<td align="left">Numerical</td>
<td align="left">The age of patients</td>
</tr>
<tr>
<td align="left">Gender</td>
<td align="left">Categorical</td>
<td align="left">The gender of patients. A value of 1 corresponds to female, while 0 corresponds to male</td>
</tr>
<tr>
<td align="left">Mutation abundance level</td>
<td align="left">Numerical</td>
<td align="left">Mutation Abundance level/Copy Number</td>
</tr>
<tr>
<td align="left">TP53</td>
<td align="left">Categorical</td>
<td align="left">Does the patient have a TP53 gene mutation? A value of 1 indicates presence, while 0 indicates absence</td>
</tr>
<tr>
<td align="left">The number of co-mutated genes</td>
<td align="left">Numerical</td>
<td align="left">The number of co-mutated genes</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We compared the performance of the HL loss function and four traditional regression loss functions in terms of mean squared error (MSE), HL statistic, and <italic>P</italic>-value. In the cohort1, overall, all loss functions passed the HL test with a <italic>p</italic>-value greater than 0.05 (<xref ref-type="fig" rid="F3">Figure 3A</xref>), and achieved low mean squared error (MSE) (<xref ref-type="fig" rid="F3">Figure 3B</xref>). SAFE-MIL demonstrated the lowest MSE and HL statistics among the two generated datasets in predicting drug effectiveness (<xref ref-type="fig" rid="F3">Figures 3B, C</xref>), indicating that the HL-based loss functions can capture differences among instances within groups, ensuring accurate estimation results while maintaining statistical interpretability. Similarly, in the cohort2, SAFE-MIL based on the HL loss also achieved lower MSE (<xref ref-type="fig" rid="F4">Figure 4</xref>). Due to the larger sample size of cohort2 compared to cohort1, theoretically, the model of HL-based SAFE-MIL predictive performance should be somewhat better. In summary, HL-based SAFE-MIL achieved the lowest MSE and passed the HL test in both datasets, obtaining optimal HL statistics. Similarly, in the cohort2 and cohort3, SAFE-MIL based on the HL loss also achieved lower MSE (<xref ref-type="fig" rid="F4">Figures 4</xref>, <xref ref-type="fig" rid="F5">5</xref>). Due to the larger sample size of cohort 2 compared to cohort1, theoretically, the model of HL-based SAFE-MIL predictive performance should be somewhat better. In summary, HL-based SAFE-MIL achieved the lowest MSE and passed the HL test in all three datasets, obtaining optimal HL statistics.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The performance of the HL loss function and four traditional regression loss functions were compared on cohort one in terms of <italic>P</italic>-value <bold>(A)</bold>, mean squared error <bold>(B)</bold>, and HL statistic <bold>(C)</bold>.</p>
</caption>
<graphic xlink:href="fgene-15-1381851-g003.tif"/>
</fig>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>The performance of the HL loss function and four traditional regression loss functions were compared on cohort 2 in terms of <italic>P</italic>-value <bold>(A)</bold>, mean squared error <bold>(B)</bold>, and HL statistic <bold>(C)</bold>.</p>
</caption>
<graphic xlink:href="fgene-15-1381851-g004.tif"/>
</fig>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>The performance of the HL loss function and four traditional regression loss functions were compared on cohort 3 in terms of <italic>P</italic>-value <bold>(A)</bold>, mean squared error <bold>(B)</bold>, and HL statistic <bold>(C)</bold>.</p>
</caption>
<graphic xlink:href="fgene-15-1381851-g005.tif"/>
</fig>
<p>Additionally, we selected cohort1-600 as examples to illustrate the relationship between the optimal positive threshold and drug efficacy. Based on the optimal positive threshold determined of mutation abundance, we divided patients in cohort1-600 into two groups to observe differences in drug failure risk and survival outcomes. Following the calculation rules mentioned earlier, the optimal positive threshold for mutation abundance in cohort1-600 was determined to be &#x3c4; &#x3d; 0.479. Firstly, using the threshold &#x3c4;, we categorized patients into high-risk (mutation abundance level &#x3c; &#x3c4;) and low-risk (mutation abundance level &#x3e;&#x3d; &#x3c4;) groups. In cohort1-600, the average drug failure risk in the high-risk group was 0.715, significantly higher than the 0.332 in the low-risk group (t-test, <italic>p</italic>&#x3c; 2.2e-16, <xref ref-type="fig" rid="F6">Figure 6A</xref>). Since survival information for patients in cohort1-600 was recorded, we evaluated the survival differences between the high-risk and low-risk groups in cohort1-600. The results indicated that the average progression-free survival in the high-risk group in cohort1-600 was 8.4&#xa0;months, significantly lower than the 16.4&#xa0;months in the low-risk group (Log-rank test, <italic>p</italic> &#x3d; 2.8e-9, <xref ref-type="fig" rid="F6">Figure 6B</xref>).</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>
<bold>(A)</bold> The distribution of drug failure risk under the optimal positive threshold grouping (Independent Samples t-test, <italic>p</italic> &#x3c; 2.2e-16). <bold>(B)</bold> The survival analysis of patients according to the optimal positive threshold (Log-rank test, <italic>p</italic> &#x3d; 2.8e-9).</p>
</caption>
<graphic xlink:href="fgene-15-1381851-g006.tif"/>
</fig>
<p>The optimal positive threshold identified by SAFE-MIL represents the mutation abundance level at which patients are stratified into high-risk and low-risk groups with respect to treatment failure. This threshold is determined based on statistical analysis and represents a critical point for distinguishing patients who are more likely to experience treatment failure from those who are likely to respond favorably to therapy. Patients identified as high-risk above the threshold may benefit from closer monitoring, alternative treatment options, or enrollment in clinical trials for novel therapies. Conversely, patients below the threshold may be considered lower risk and may continue with standard therapy with confidence in treatment efficacy. The use of the optimal positive threshold enhances patient risk stratification by providing a quantitative measure that correlates with treatment outcomes. This allows for more accurate risk assessment and personalized treatment approaches, ultimately improving patient outcomes and optimizing resource allocation in healthcare settings. The identification of an optimal positive threshold enables clinicians to make informed decisions regarding treatment selection by considering individual patient risk profiles. SAFE-MIL facilitates precision medicine approaches, ensuring that patients receive the most appropriate and effective treatments based on their unique characteristics.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>The traditional clinical decision-making approaches often simplify drug efficacy into binary classifications, which fails to capture the nuanced risks associated with drug failure in individual patients. To address this limitation, we proposed that the clinical drug response for patients should be a regression value, reflecting individual variability in drug failure risk. This study introduces the SAFE-MIL framework, a new method for calculating and predicting the risk of drug failure in patients, which designs drug effectiveness labels based on clinical features of patients and builds a model to predict the risk of drug failure using multiple instance learning. A critical enhancement in SAFE-MIL is the integration of a loss function derived from the Hosmer-Lemeshow test. This integration not only boosts model performance but also ensures statistical interpretability, enabling robust assessments of drug failure risks and facilitating patient stratification by identifying the optimal positive threshold. Our results indicate that SAFE-MIL outperforms traditional regression methods in accuracy, particularly in capturing inter-patient variability in risk assessments, and successfully passes the HL test.</p>
<p>The concept of drug failure risk introduced in this study provides a more comprehensive characterization of drug effectiveness. Due to the significant molecular heterogeneity observed in tumors, there are often many different molecular features and feature combinations that can lead to the model predicting specific drug responses. However, elucidating these features and discerning whether they are distinct or functionally linked can be challenging. This is primarily due to the &#x201c;black box&#x201d; nature of most machine learning models, which prioritize predictive accuracy over an understanding of the underlying biological mechanisms (<xref ref-type="bibr" rid="B8">Ching et al., 2018</xref>). SAFE-MIL addresses this by enhancing clinical interpretability through its novel use of MIL and HL test integration, thereby somewhat bridging the gap between clinical needs and machine learning methodologies.</p>
<p>Furthermore, the study critically evaluates and expands upon existing methods for interpreting deep learning algorithms in clinical settings [(<xref ref-type="bibr" rid="B41">Zhang et al., 2018</xref>)]. Our approach incorporates the Hosmer-Lemeshow test into the MIL framework to estimate the risk of treatment failure, significantly improving the model&#x2019;s predictive calibration performance and offering a unique risk assessment perspective beyond traditional interpretation methods. The results demonstrate that our framework outperforms traditional regression loss functions in terms of accuracy and successfully passes the Hosmer-Lemeshow test, highlighting its capability to accurately capture inter-patient variability in risk while ensuring statistical interpretability. Additionally, our framework enables the identification of an optimal threshold for mutation abundance level, allowing for effective patient stratification into high-risk and low-risk groups. This stratification reveals significant disparities in drug failure risk and progression-free survival, showcasing the utility of our approach in guiding treatment decisions.</p>
<p>Although our study primarily focuses on patients with EGFR-mutant non-small cell lung cancer treated with EGFR tyrosine kinase inhibitors, the potential of SAFE-MIL extends far beyond this specific context. Moving forward, efforts to validate and refine the SAFE-MIL framework across different cancer types and therapeutic modalities will further enhance its utility and broaden its impact on clinical practice. Beyond its predictive capabilities, SAFE-MIL offers early risk identification and optimizing clinical trial design. For instance, SAFE-MIL enables the identification of patient subgroups with distinct risk profiles, paving the way for targeted interventions and precision medicine approaches. Moreover, integrating automated risk assessment tools into clinical workflows holds promise for streamlining decision-making processes and improving patient outcomes in real-world practice. Consequently, the integration of SAFE-MIL is poised to transform clinical practice by boosting treatment efficacy and ensuring patient safety. The drug failure risk metrics derived from SAFE-MIL appear to correlate with the patient&#x2019;s resistance mechanisms, a relationship that warrants deeper investigation.</p>
<p>Despite these promising results, our study acknowledges several limitations. Firstly, due to the limited number of patients, we could only generate groups through bootstrap sampling, and we constrained the number of groups and the number of instances per group. Additionally, the availability and quality of data sources pose significant challenges to model development and validation. Future efforts should focus on employing SAFE-MIL in larger-scale studies to explore more robust and efficient risk calculation models. Furthermore, implementing SAFE-MIL in real-world healthcare settings necessitates careful consideration of practical challenges such as workflow integration, clinician training, and patient acceptance.</p>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>This paper presented SAFE-MIL, a novel risk assessment framework tailored to address the issue of varying survival benefits among patients with certain target gene mutations undergoing targeted therapy. By incorporating multiple instance learning (MIL), SAFE-MIL constructs patient-specific effectiveness labels and designs a novel interpretable loss function based on the Hosmer-Lemeshow test. This risk assessment framework accurately estimates the risk of treatment failure and also provides the optimal threshold for risk stratification. A comprehensive case study involving 457 non-small cell lung cancer patients with EGFR mutations treated with EGFR tyrosine kinase inhibitors demonstrated that SAFE-MIL outperforms conventional regression methods in accuracy, effectively capturing inter-patient variability in risk. This computational framework possesses statistical interpretability and adaptability, rendering it a helpful tool in clinical decision-making for targeted therapy. It has the potential to advance personalized medicine by enhancing patient stratification methods.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s12">Supplementary Material</xref>, and the data presented in the study are deposited in the GSA- human repository, accession number HRA006927.</p>
</sec>
<sec id="s7">
<title>Ethics statement</title>
<p>The studies involving humans were approved by the Institutional Review Board of Shanghai Chest Hospital (No. KS1707). The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study.</p>
</sec>
<sec id="s8">
<title>Author contributions</title>
<p>YG: Data curation, Methodology, Software, Writing&#x2013;original draft, Writing&#x2013;review and editing. ZX: Formal Analysis, Methodology, Software, Writing&#x2013;original draft, Writing&#x2013;review and editing. JW: Conceptualization, Methodology, Supervision, Writing&#x2013;review and editing. XA: Data curation, Resources, Writing&#x2013;review and editing. RC: Resources, Supervision, Writing&#x2013;review and editing. XY: Resources, Supervision, Writing&#x2013;review and editing. SL: Conceptualization, Resources, Supervision, Writing&#x2013;review and editing. YL: Conceptualization, Formal Analysis, Methodology, Supervision, Writing&#x2013;original draft, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work was supported by National Natural Science Foundation of China, grant numbers 72274152, 82030045, 82241227, National Multi-disciplinary Treatment Project for Major Diseases, grant number 2020NMDTP, Collaborative Innovation Center for Clinical and Translational Science by Ministry of Education and Shanghai, grant numbers TM202112, CCTS202204, General program of Natural Science Foundation of Shanghai, grant number 22ZR1457500.</p>
</sec>
<ack>
<p>We thank all faculty members and graduate students who discussed the mathematical and statistical issues in seminars.</p>
</ack>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>Authors YG, RC and XY were employed by GenePlus Beijing Institute.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2024.1381851/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2024.1381851/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet2.PDF" id="SM1" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="DataSheet1.ZIP" id="SM2" mimetype="application/ZIP" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Amar</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Dooly</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Goldman</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Multiple-instance learning of real-valued data</article-title>. <source>InICML</source> <volume>28</volume>, <fpage>3</fpage>&#x2013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.5555/944919.944949</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Banerjee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mohammed</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>H. R.</given-names>
</name>
<name>
<surname>Palaniyar</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kamaleswaran</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Machine learning identifies complicated sepsis course and subsequent mortality based on 20 genes in peripheral blood immune cells at 24 H post-ICU admission</article-title>. <source>Front. Immunol.</source> <volume>12</volume>, <fpage>592303</fpage>. <pub-id pub-id-type="doi">10.3389/fimmu.2021.592303</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Blakely</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Watkins</surname>
<given-names>T. B. K.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Gini</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chabon</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>McCoach</surname>
<given-names>C. E.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Evolution and clinical impact of co-occurring genetic alterations in advanced-stage EGFR-mutant lung cancers</article-title>. <source>Nat. Genet.</source> <volume>49</volume> (<issue>12</issue>), <fpage>1693</fpage>&#x2013;<lpage>1704</lpage>. <pub-id pub-id-type="doi">10.1038/ng.3990</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Blumenthal</surname>
<given-names>G. M.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kehl</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Mishra-Kalyani</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Goldberg</surname>
<given-names>K. B.</given-names>
</name>
<name>
<surname>Khozin</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Analysis of time-to-treatment discontinuation of targeted therapy, immunotherapy, and chemotherapy in clinical trials of patients with non-small-cell lung cancer</article-title>. <source>Ann. Oncol.</source> <volume>30</volume> (<issue>5</issue>), <fpage>830</fpage>&#x2013;<lpage>838</lpage>. <pub-id pub-id-type="doi">10.1093/annonc/mdz060</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Moustafa</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Bashir</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>AI-driven synthetic biology for non-small cell lung cancer drug effectiveness-cost analysis in intelligent assisted medical systems</article-title>. <source>IEEE J. Biomed. Health Inf.</source> <volume>26</volume> (<issue>10</issue>), <fpage>5055</fpage>&#x2013;<lpage>5066</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2021.3133455</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Machine learning reveals ferroptosis features and a novel ferroptosis classifier in patients with sepsis</article-title>. <source>Immun. Inflamm. Dis.</source> <volume>12</volume> (<issue>5</issue>), <fpage>e1279</fpage>. <pub-id pub-id-type="doi">10.1002/iid3.1279</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Comprehensive characterization and clinical impact of concomitant genomic alterations in EGFR-mutant NSCLCs treated with EGFR kinase inhibitors</article-title>. <source>Lung Cancer</source> <volume>145</volume>, <fpage>63</fpage>&#x2013;<lpage>70</lpage>. <pub-id pub-id-type="doi">10.1016/j.lungcan.2020.04.004</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ching</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Himmelstein</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Beaulieu-Jones</surname>
<given-names>B. K.</given-names>
</name>
<name>
<surname>Kalinin</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Do</surname>
<given-names>B. T.</given-names>
</name>
<name>
<surname>Way</surname>
<given-names>G. P.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Opportunities and obstacles for deep learning in biology and medicine</article-title>. <source>J. R. Soc. Interface</source> <volume>15</volume> (<issue>141</issue>), <fpage>20170387</fpage>. <pub-id pub-id-type="doi">10.1098/rsif.2017.0387</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Daoud</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mdhaffar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jmaiel</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Freisleben</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Q-rank: reinforcement learning for recommending algorithms to predict drug sensitivity to cancer therapy</article-title>. <source>IEEE J. Biomed. Health Inf.</source> <volume>24</volume> (<issue>11</issue>), <fpage>3154</fpage>&#x2013;<lpage>3161</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2020.3004663</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Davies</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Redfield</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Scott</surname>
<given-names>C. G.</given-names>
</name>
<name>
<surname>Minamisawa</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Grogan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dispenzieri</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>A simple score to identify increased risk of transthyretin amyloid cardiomyopathy in heart failure with preserved ejection fraction</article-title>. <source>JAMA Cardiol.</source> <volume>7</volume> (<issue>10</issue>), <fpage>1036</fpage>&#x2013;<lpage>1044</lpage>. <pub-id pub-id-type="doi">10.1001/jamacardio.2022.1781</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Diao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Huo</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A simplified machine learning model utilizing platelet-related genes for predicting poor prognosis in sepsis</article-title>. <source>Front. Immunol.</source> <volume>14</volume>, <fpage>1286203</fpage>. <pub-id pub-id-type="doi">10.3389/fimmu.2023.1286203</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fisher</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>1915</year>). <article-title>Frequency distribution of the values of the correlation coefficient in samples from an indefinitely large population</article-title>. <source>Biometrika</source> <volume>10</volume> (<issue>4</issue>), <fpage>507</fpage>&#x2013;<lpage>521</lpage>. <pub-id pub-id-type="doi">10.2307/2331838</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fisher</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>1921</year>). <article-title>On the &#x27;probable error&#x27; of a coefficient of correlation deduced from a small sample</article-title>. <source>Contributions Math. Statistics</source>, <fpage>3</fpage>&#x2013;<lpage>32</lpage>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Nan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>R. Y.</given-names>
</name>
<name>
<surname>Daga</surname>
<given-names>P. R.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Implementation of multiple-instance learning in drug activity prediction</article-title>. <source>BMC Bioinforma.</source> <volume>13</volume> (<issue>15</issue>), <fpage>S3</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-13-S15-S3</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>G&#xf6;ttlich</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>M&#xfc;ller</surname>
<given-names>L. C.</given-names>
</name>
<name>
<surname>Kunz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schmitt</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Walles</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Walles</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>A combined 3D tissue engineered <italic>in vitro</italic>/<italic>in silico</italic> lung tumor model for predicting drug effectiveness in specific mutational backgrounds</article-title>. <source>J. Vis. Exp.</source> <volume>110</volume>, <fpage>e53885</fpage>. <pub-id pub-id-type="doi">10.3791/53885</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hosmer</surname>
<given-names>D. W.</given-names>
</name>
<name>
<surname>Lemesbow</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>1980</year>). <article-title>Goodness of fit tests for the multiple logistic regression model</article-title>. <source>Commun. Statistics - Theory Methods</source> <volume>9</volume> (<issue>10</issue>), <fpage>1043</fpage>&#x2013;<lpage>1069</lpage>. <pub-id pub-id-type="doi">10.1080/03610928008827941</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huber</surname>
<given-names>P. J.</given-names>
</name>
</person-group> (<year>1992</year>). <article-title>Robust estimation of a location parameter</article-title>. <source>Break. statistics Methodol. distribution</source>, <fpage>492</fpage>&#x2013;<lpage>518</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-4612-4380-9_35</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kramer</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Zimmerman</surname>
<given-names>J. E.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Assessing the calibration of mortality benchmarks in critical care: the Hosmer-Lemeshow test revisited</article-title>. <source>Crit. Care Med.</source> <volume>35</volume> (<issue>9</issue>), <fpage>2052</fpage>&#x2013;<lpage>2056</lpage>. <pub-id pub-id-type="doi">10.1097/01.CCM.0000275267.64078.B0</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kuenzi</surname>
<given-names>B. M.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fong</surname>
<given-names>S. H.</given-names>
</name>
<name>
<surname>Sanchez</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kreisberg</surname>
<given-names>J. F.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Predicting drug response and synergy using a deep learning model of human cancer cells</article-title>. <source>Cancer Cell</source> <volume>38</volume> (<issue>5</issue>), <fpage>672</fpage>&#x2013;<lpage>684.e6</lpage>. <pub-id pub-id-type="doi">10.1016/j.ccell.2020.09.014</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>EGFR mutation types and abundance were associated with the overall survival of advanced lung adenocarcinoma patients receiving first-line tyrosine kinase inhibitors</article-title>. <source>J. Thorac. Dis.</source> <volume>14</volume> (<issue>6</issue>), <fpage>2254</fpage>&#x2013;<lpage>2267</lpage>. <pub-id pub-id-type="doi">10.21037/jtd-22-755</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>&#x141;osi&#x144;ska</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wilk</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pripp</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Korkosz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Haugeberg</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Long-term drug effectiveness and survival for reference rituximab in rheumatoid arthritis patients in an ordinary outpatient clinic</article-title>. <source>Sci. Rep.</source> <volume>12</volume> (<issue>1</issue>), <fpage>8283</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-022-12271-9</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nemati</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Holder</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Razmi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Stanley</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Clifford</surname>
<given-names>G. D.</given-names>
</name>
<name>
<surname>Buchman</surname>
<given-names>T. G.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>An interpretable machine learning model for accurate prediction of sepsis in the ICU</article-title>. <source>Crit. care Med.</source> <volume>46</volume> (<issue>4</issue>), <fpage>547</fpage>&#x2013;<lpage>553</lpage>. <pub-id pub-id-type="doi">10.1097/CCM.0000000000002936</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Guan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yi</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Circulating tumor DNA analysis depicts subclonal architecture and genomic evolution of small cell lung cancer [published correction appears in Nat Commun 2019 Jan 29;10(1):552]</article-title>. <source>Nat. Commun.</source> <volume>9</volume> (<issue>1</issue>), <fpage>3114</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-018-05327-w</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paz-Ares</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Luft</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Vicente</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Tafreshi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>G&#xfc;m&#xfc;&#x15f;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mazi&#xe8;res</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Pembrolizumab plus chemotherapy for squamous non-small-cell lung cancer</article-title>. <source>N. Engl. J. Med.</source> <volume>379</volume> (<issue>21</issue>), <fpage>2040</fpage>&#x2013;<lpage>2051</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMoa1810865</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Predicting drug response based on multi-omics fusion and graph convolution</article-title>. <source>IEEE J. Biomed. Health Inf.</source> <volume>26</volume> (<issue>3</issue>), <fpage>1384</fpage>&#x2013;<lpage>1393</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2021.3102186</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Robichaux</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Vijayan</surname>
<given-names>R. S. K.</given-names>
</name>
<name>
<surname>Hicks</surname>
<given-names>J. K.</given-names>
</name>
<name>
<surname>Heeke</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Elamin</surname>
<given-names>Y. Y.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Structure-based classification predicts drug response in EGFR-mutant NSCLC</article-title>. <source>Nature</source> <volume>597</volume> (<issue>7878</issue>), <fpage>732</fpage>&#x2013;<lpage>737</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-021-03898-1</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rubin</surname>
<given-names>E. H.</given-names>
</name>
<name>
<surname>Gilliland</surname>
<given-names>D. G.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Drug development and clinical trials--the path to an approved cancer drug</article-title>. <source>Nat. Rev. Clin. Oncol.</source> <volume>9</volume> (<issue>4</issue>), <fpage>215</fpage>&#x2013;<lpage>222</lpage>. <pub-id pub-id-type="doi">10.1038/nrclinonc.2012.22</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saberian</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Moriarty</surname>
<given-names>K. P.</given-names>
</name>
<name>
<surname>Olmstead</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Hallgrimson</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Jean</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Nabi</surname>
<given-names>I. R.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>DEEMD: drug efficacy estimation against SARS-CoV-2 based on cell morphology with deep multiple instance learning</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>41</volume> (<issue>11</issue>), <fpage>3128</fpage>&#x2013;<lpage>3145</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2022.3178523</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Samstein</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Shoushtari</surname>
<given-names>A. N.</given-names>
</name>
<name>
<surname>Hellmann</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Janjigian</surname>
<given-names>Y. Y.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Tumor mutational load predicts survival after immunotherapy across multiple cancer types</article-title>. <source>Nat. Genet.</source> <volume>51</volume> (<issue>2</issue>), <fpage>202</fpage>&#x2013;<lpage>206</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-018-0312-8</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schnipper</surname>
<given-names>L. E.</given-names>
</name>
<name>
<surname>Davidson</surname>
<given-names>N. E.</given-names>
</name>
<name>
<surname>Wollins</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Tyne</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Blayney</surname>
<given-names>D. W.</given-names>
</name>
<name>
<surname>Blum</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>American society of clinical oncology statement: a conceptual framework to assess the value of cancer treatment options</article-title>. <source>J. Clin. Oncol.</source> <volume>33</volume> (<issue>23</issue>), <fpage>2563</fpage>&#x2013;<lpage>2577</lpage>. <pub-id pub-id-type="doi">10.1200/JCO.2015.61.6706</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Tao</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Continuous dropout</article-title>. <source>IEEE Trans. Neural Netw. Learn Syst.</source> <volume>29</volume> (<issue>9</issue>), <fpage>3926</fpage>&#x2013;<lpage>3937</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2017.2750679</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sotudian</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Paschalidis</surname>
<given-names>I. C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Machine learning for pharmacogenomics and personalized medicine: a ranking model for drug sensitivity prediction</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinform</source> <volume>19</volume> (<issue>4</issue>), <fpage>2324</fpage>&#x2013;<lpage>2333</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2021.3084562</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ming</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Self-internal-reference probe system for control-free quantification of mutation abundance</article-title>. <source>Anal. Chem.</source> <volume>93</volume> (<issue>39</issue>), <fpage>13274</fpage>&#x2013;<lpage>13283</lpage>. <pub-id pub-id-type="doi">10.1021/acs.analchem.1c02877</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Modeling DTA by combining multiple-instance learning with a private-public mechanism</article-title>. <source>Int. J. Mol. Sci.</source> <volume>23</volume> (<issue>19</issue>), <fpage>11136</fpage>. <pub-id pub-id-type="doi">10.3390/ijms231911136</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Plasma EGFR mutation abundance affects clinical response to first-line EGFR-TKIs in patients with advanced non-small cell lung cancer</article-title>. <source>Ann. Transl. Med.</source> <volume>9</volume> (<issue>8</issue>), <fpage>635</fpage>. <pub-id pub-id-type="doi">10.21037/atm-20-7155</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Efficacy of first-line treatment with epidermal growth factor receptor-tyrosine kinase inhibitor (EGFR-TKI) alone or in combination with chemotherapy for advanced non-small cell lung cancer (NSCLC) with low-abundance mutation</article-title>. <source>Lung Cancer</source> <volume>128</volume>, <fpage>6</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1016/j.lungcan.2018.12.007</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>DRONet: effectiveness-driven drug repositioning framework using network embedding and ranking learning</article-title>. <source>Brief. Bioinform</source> <volume>24</volume> (<issue>1</issue>), <fpage>bbac518.</fpage> <pub-id pub-id-type="doi">10.1093/bib/bbac518</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2021b</year>). <article-title>Construction and evaluation of a sepsis risk prediction model for urinary tract infection</article-title>. <source>Front. Med. (Lausanne)</source> <volume>8</volume>, <fpage>671184</fpage>. <pub-id pub-id-type="doi">10.3389/fmed.2021.671184</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Guan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>The correlations of tumor mutational burden among single-region tissue, multi-region tissues and blood in non-small cell lung cancer</article-title>. <source>J. Immunother. Cancer</source> <volume>7</volume> (<issue>1</issue>), <fpage>98</fpage>. <pub-id pub-id-type="doi">10.1186/s40425-019-0581-5</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>C-X.</given-names>
</name>
<name>
<surname>Abudushalamu</surname>
</name>
<name>
<surname>Jin</surname>
<given-names>X-L.</given-names>
</name>
<etal/>
</person-group> (<year>2021a</year>). <article-title>International clinical practice guideline of Chinese medicine Alzheimer</article-title>. <source>World J. Trad. Chin. Med.</source> <volume>7</volume> (<issue>2</issue>), <fpage>265</fpage>&#x2013;<lpage>275</lpage>. <pub-id pub-id-type="doi">10.4103/wjtcm.wjtcm_28_21</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Beck</surname>
<given-names>M. W.</given-names>
</name>
<name>
<surname>Winkler</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Sibanda</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Goyal</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Opening the black box of neural networks: methods for interpreting neural network models in clinical applications</article-title>. <source>Ann. Transl. Med.</source> <volume>6</volume> (<issue>11</issue>), <fpage>216</fpage>. <pub-id pub-id-type="doi">10.21037/atm.2018.05.32</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Ge</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Deep learning-based clustering robustly identified two classes of sepsis with both prognostic and predictive values</article-title>. <source>EBioMedicine</source> <volume>62</volume>, <fpage>103081</fpage>. <pub-id pub-id-type="doi">10.1016/j.ebiom.2020.103081</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Elokely</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Doerksen</surname>
<given-names>R. J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Drug activity prediction using multiple-instance learning via joint instance and feature selection</article-title>. <source>BMC Bioinforma.</source> <volume>14</volume> (<issue>Suppl. 14</issue>), <fpage>S16</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-14-S14-S16</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X. C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z. H.</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>X. L.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>C. R.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Relative abundance of EGFR mutations predicts benefit from gefitinib treatment for advanced non-small-cell lung cancer</article-title>. <source>J. Clin. Oncol.</source> <volume>29</volume> (<issue>24</issue>), <fpage>3316</fpage>&#x2013;<lpage>3321</lpage>. <pub-id pub-id-type="doi">10.1200/JCO.2010.33.3757</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>