<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Med.</journal-id>
<journal-title>Frontiers in Medicine</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Med.</abbrev-journal-title>
<issn pub-type="epub">2296-858X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmed.2024.1482726</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Medicine</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>BreCML: identifying breast cancer cell state in scRNA-seq via machine learning</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Ke</surname> <given-names>Shanbao</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn0001"><sup>&#x2020;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Huang</surname> <given-names>Yuxuan</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn0001"><sup>&#x2020;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2709591/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Wang</surname> <given-names>Dong</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="author-notes" rid="fn0001"><sup>&#x2020;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Jiang</surname> <given-names>Qiang</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Luo</surname> <given-names>Zhangyang</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1322996/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Baiyu</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Yan</surname> <given-names>Danfang</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2133784/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Zhou</surname> <given-names>Jianwei</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2820143/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Oncology, Henan Provincial People&#x2019;s Hospital, Zhengzhou University People&#x2019;s Hospital</institution>, <addr-line>Zhengzhou</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Neuroscience in the Behavioral Sciences, Duke University and Duke Kunshan University</institution>, <addr-line>Suzhou</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Pudong Institute for Health Development</institution>, <addr-line>Shanghai</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Radiation Oncology, The First Affiliated Hospital, College of Medicine, Zhejiang University</institution>, <addr-line>Hangzhou</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0002">
<p>Edited by: Udhaya Kumar, Baylor College of Medicine, United States</p>
</fn>
<fn fn-type="edited-by" id="fn0003">
<p>Reviewed by: Muthu Kumar Krishnamoorthi, Houston Methodist Research Institute, United States</p>
<p>Xingjian Chen, City University of Hong Kong, Hong Kong SAR, China</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Jianwei Zhou, <email>18037790277@163.com</email></corresp>
<fn fn-type="equal" id="fn0001"><p><sup>&#x2020;</sup>These authors have contributed equally to this work</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>06</day>
<month>11</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>11</volume>
<elocation-id>1482726</elocation-id>
<history>
<date date-type="received">
<day>18</day>
<month>08</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>15</day>
<month>10</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2024 Ke, Huang, Wang, Jiang, Zhanyang, Li, Yan and Zhou.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Ke, Huang, Wang, Jiang, Zhanyang, Li, Yan and Zhou</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Breast cancer is a prevalent malignancy and one of the leading causes of cancer-related mortality among women worldwide. This disease typically manifests through the abnormal proliferation and dissemination of malignant cells within breast tissue. Current diagnostic and therapeutic strategies face significant challenges in accurately identifying and localizing specific subtypes of breast cancer. In this study, we developed a novel machine learning-based predictor, BreCML, designed to accurately classify subpopulations of breast cancer cells and their associated marker genes. BreCML exhibits outstanding predictive performance, achieving an accuracy of 98.92% on the training dataset. Utilizing the XGBoost algorithm, BreCML demonstrates superior accuracy (98.67%), precision (99.15%), recall (99.49%), and F1-score (99.79%) on the test dataset. Through the application of machine learning and feature selection techniques, BreCML successfully identified new key genes. This predictor not only serves as a powerful tool for assessing breast cancer cellular status but also offers a rapid and efficient means to uncover potential biomarkers, providing critical insights for precision medicine and therapeutic strategies.</p>
</abstract>
<kwd-group>
<kwd>breast cancer</kwd>
<kwd>machine learning</kwd>
<kwd>scRNA-seq</kwd>
<kwd>cell subpopulations</kwd>
<kwd>feature selection</kwd>
</kwd-group>
<contract-num rid="cn1">PWZxk2022-27</contract-num>
<contract-sponsor id="cn1">Key Discipline Construction Project of Pudong Health Bureau of Shanghai: Clinical Pharmacy</contract-sponsor>
<counts>
<fig-count count="6"/>
<table-count count="2"/>
<equation-count count="6"/>
<ref-count count="32"/>
<page-count count="9"/>
<word-count count="4879"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Precision Medicine</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<title>Introduction</title>
<p>Breast cancer is considered the most common malignant tumor worldwide and is one of the leading causes of cancer-related deaths among women globally (<xref ref-type="bibr" rid="ref1">1</xref>, <xref ref-type="bibr" rid="ref2">2</xref>). The incidence of breast cancer is influenced by multiple factors, including age, genetic background, and reproductive history and so on. Long-term exposure to ovarian steroids is widely recognized as a risk factor for breast cancer in women, and studies have shown a significant correlation between the total number of menstrual cycles and the risk of breast cancer (<xref ref-type="bibr" rid="ref3 ref4 ref5">3&#x2013;5</xref>). Breast cancer is commonly described as an &#x201C;immunologically cold&#x201D; tumor (<xref ref-type="bibr" rid="ref6">6</xref>), characterized by a low mutation count, limited immune cell infiltration, and immunosuppressive features in the tumor microenvironment (<xref ref-type="bibr" rid="ref7">7</xref>).</p>
<p>A detailed exploration of the cellular subtypes of breast cancer is crucial for developing more precise clinical treatment protocols and for advancing pathophysiological research. The genetic heterogeneity of breast cancer has been confirmed at a single-cell resolution, a process dependent on high-density genome coverage (<xref ref-type="bibr" rid="ref8">8</xref>). With the ongoing advancement of single-cell sequencing technology, we are now able to explore the cellular heterogeneity of this cancer at an even higher resolution.</p>
<p>Through single-cell transcriptomic analysis, Chung et al. (<xref ref-type="bibr" rid="ref9">9</xref>) explored the heterogeneity of tumor cells and their neighboring immune and stromal cells, revealing significant heterogeneity both within the tumor and among immune cells. Jang et al. (<xref ref-type="bibr" rid="ref10">10</xref>) utilized single-cell RNA sequencing (scRNA-seq) technology to analyze the transcriptional and mutational features of breast cancer and immune cells. They identified high PD-L1 expression and significant microsatellite instability in radioresistant cells, along with complex interactions at immune checkpoints. These findings provide potential biomarkers and therapeutic strategies for immunotherapy and radiation therapy tailored to different subtypes of breast cancer. Liu et al. (<xref ref-type="bibr" rid="ref11">11</xref>) combined scRNA-seq with spatial transcriptomics to analyze the cell populations and their spatial distribution in breast cancer. They identified subpopulations of malignant cells, revealing their locations and the relationships with patient survival and therapeutic responses, which provided new insights into the heterogeneity of breast cancer and potential personalized treatment strategies. Ding et al. (<xref ref-type="bibr" rid="ref12">12</xref>) discussed the application of scRNA-seq in breast cancer research. Through technological advancements, scRNA-seq has revealed cellular heterogeneity in the tumor microenvironment and identified disease-related rare cell types. This technique has demonstrated its potential in classifying breast cancer subtypes, recognizing immune cell subgroups, and identifying therapeutic targets, thereby facilitating the development of personalized treatment strategies.</p>
<p>Although existing research technologies in the field of breast cancer are relatively advanced, manual methods remain time-consuming and labor-intensive when it comes to mining marker genes and identifying cell subgroups. Consequently, there is an urgent need for the development of computational methods to assist researchers in efficiently identifying breast cancer cell subgroups and thoroughly exploring their potential marker genes. To address these challenges, we introduced a computational framework named BreCML (<xref ref-type="fig" rid="fig1">Figure 1</xref>). This framework is designed to identify biomarkers within breast cancer cell subpopulations and infer their cellular developmental stages, thereby enhancing the accuracy and depth of research in this area. To achieve optimal predictive modeling results, we employed a combined feature selection and incremental feature selection (IFS) strategy. This strategy incorporates the use of four fundamental classification methods: K-nearest neighbors (KNN), extreme gradient boosting (XGBoost), support vector machine (SVM), and random forest classification (RFC).</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>The workflow of constructing BreCML.</p>
</caption>
<graphic xlink:href="fmed-11-1482726-g001.tif"/>
</fig>
</sec>
<sec sec-type="results" id="sec2">
<title>Results</title>
<sec id="sec3">
<title>Identify important genes by BreCML</title>
<p>To identify key genes associated with subpopulations of breast cancer cells, we employed three feature selection methods <italic>F</italic>-score, coefficient of variation squared (CV<sup>2</sup>), and principal component analysis (PCA) to evaluate the significance of 29,733 genes and rank them according to their contribution (<xref ref-type="fig" rid="fig2">Figures 2A</xref>&#x2013;<xref ref-type="fig" rid="fig2">C</xref>). Genes with importance scores less than or equal to zero were excluded from further analysis. The CV<sup>2</sup>, PCA, and <italic>F</italic>-score extracted 22,000 important genes. Subsequently, machine learning models combined with incremental feature selection (IFS) were utilized to identify the optimal subset of genes. Using five-fold cross-validation, the machine learning models (SVM, RFC, XGBoost, and KNN) were trained with single-cell gene expression matrices as input features.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>The results of feature selection. <bold>(A&#x2013;C)</bold> The IFS curves show the performance of three feature selections (<italic>F</italic>-score, CV<sup>2</sup>, and PCA) and the four classifiers in different gene subsets. <bold>(D)</bold> Comparative Venn diagram of the top 100 genes in <italic>F</italic>-score, CV<sup>2</sup>, and PCA.</p>
</caption>
<graphic xlink:href="fmed-11-1482726-g002.tif"/>
</fig>
<p>The analysis of the training dataset showed that the combination of <italic>F</italic>-score and XGBoost model (BreCML) using the top 360 genes achieved the best predictive performance, successfully classifying breast cancer cell subpopulations with 98.92% accuracy (<xref ref-type="supplementary-material" rid="SM1">Supplementary Table S3</xref>). Notably, significant predictive performance was also achieved when the four machine learning models were combined with PCA. However, BreCML uses only 360 feature genes, while XGBoost combined with PCA uses 20,000 feature genes, meaning that the complexity of the model was greatly reduced. Therefore, BreCML was selected as the classifier by us. To prevent the feature selection methods from exhibiting similar scoring preferences, we compared the top 100 genes ranked by each feature selection method. As demonstrated in <xref ref-type="fig" rid="fig2">Figure 2D</xref>, the top 100 genes selected by PCA, CV<sup>2</sup>, and <italic>F</italic>-score exhibit minimal overlap, thereby validating the distinct effectiveness of each feature selection method. These findings underscore the utility of combining diverse feature selection techniques to enhance the robustness and accuracy of predictive models in breast cancer research.</p>
</sec>
<sec id="sec4">
<title>BreCML performance on test dataset</title>
<p>BreCML demonstrated exceptional performance on the test dataset, achieving outstanding results across several key metrics: accuracy of 98.67%, precision of 99.15%, recall of 99.49%, and F1-score of 99.79% (<xref ref-type="table" rid="tab1">Table 1</xref>). To further evaluate the model&#x2019;s effectiveness, we assessed its predictive capabilities using Receiver Operating Characteristic (ROC) curves and confusion matrices. The ROC analysis revealed an impressive area under the curve (AUC) of 0.97 for the BreCML model, as shown in <xref ref-type="fig" rid="fig3">Figure 3A</xref>. Additionally, the confusion matrix provided a detailed breakdown of the model&#x2019;s performance across different breast cancer subgroups, highlighting a notably low misclassification rate (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). This strong performance underscores the robustness and reliability of the BreCML model in clinical diagnostics.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Performance comparison of different algorithms and feature selection strategies (test dataset).</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Method</th>
<th align="left" valign="top">Feature selection</th>
<th align="center" valign="top">No. of feature</th>
<th align="center" valign="top">Accuracy</th>
<th align="center" valign="top">Precision</th>
<th align="center" valign="top">Recall</th>
<th align="center" valign="top">F1-measure</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">KNN</td>
<td align="left" valign="top"><italic>F</italic>-score</td>
<td align="center" valign="top">360</td>
<td align="char" valign="top" char=".">94.78</td>
<td align="char" valign="top" char=".">94.98</td>
<td align="char" valign="top" char=".">95.54</td>
<td align="char" valign="top" char=".">95.79</td>
</tr>
<tr>
<td align="left" valign="top">RFC</td>
<td align="left" valign="top"><italic>F</italic>-score</td>
<td align="center" valign="top">360</td>
<td align="char" valign="top" char=".">96.72</td>
<td align="char" valign="top" char=".">97.89</td>
<td align="char" valign="top" char=".">97.54</td>
<td align="char" valign="top" char=".">96.97</td>
</tr>
<tr>
<td align="left" valign="top">SVM</td>
<td align="left" valign="top"><italic>F</italic>-score</td>
<td align="center" valign="top">860</td>
<td align="char" valign="top" char=".">99.08</td>
<td align="char" valign="top" char=".">99.78</td>
<td align="char" valign="top" char=".">99.68</td>
<td align="char" valign="top" char=".">99.82</td>
</tr>
<tr>
<td align="left" valign="top">XGBoost</td>
<td align="left" valign="top"><italic>F</italic>-score</td>
<td align="center" valign="top">360</td>
<td align="char" valign="top" char=".">98.67</td>
<td align="char" valign="top" char=".">99.15</td>
<td align="char" valign="top" char=".">99.49</td>
<td align="char" valign="top" char=".">99.79</td>
</tr>
<tr>
<td align="left" valign="top">KNN</td>
<td align="left" valign="top">CV<sup>2</sup></td>
<td align="center" valign="top">1,500</td>
<td align="char" valign="top" char=".">88.92</td>
<td align="char" valign="top" char=".">89.22</td>
<td align="char" valign="top" char=".">89.74</td>
<td align="char" valign="top" char=".">89.18</td>
</tr>
<tr>
<td align="left" valign="top">RFC</td>
<td align="left" valign="top">CV<sup>2</sup></td>
<td align="center" valign="top">1,200</td>
<td align="char" valign="top" char=".">95.18</td>
<td align="char" valign="top" char=".">95.73</td>
<td align="char" valign="top" char=".">95.79</td>
<td align="char" valign="top" char=".">96.39</td>
</tr>
<tr>
<td align="left" valign="top">SVM</td>
<td align="left" valign="top">CV<sup>2</sup></td>
<td align="center" valign="top">1,200</td>
<td align="char" valign="top" char=".">98.15</td>
<td align="char" valign="top" char=".">98.76</td>
<td align="char" valign="top" char=".">99.27</td>
<td align="char" valign="top" char=".">98.94</td>
</tr>
<tr>
<td align="left" valign="top">XGBoost</td>
<td align="left" valign="top">CV<sup>2</sup></td>
<td align="center" valign="top">22,000</td>
<td align="char" valign="top" char=".">98.97</td>
<td align="char" valign="top" char=".">99.12</td>
<td align="char" valign="top" char=".">99.67</td>
<td align="char" valign="top" char=".">99.81</td>
</tr>
<tr>
<td align="left" valign="top">KNN</td>
<td align="left" valign="top">PCA</td>
<td align="center" valign="top">160</td>
<td align="char" valign="top" char=".">70.67</td>
<td align="char" valign="top" char=".">71.49</td>
<td align="char" valign="top" char=".">70.91</td>
<td align="char" valign="top" char=".">71.59</td>
</tr>
<tr>
<td align="left" valign="top">RFC</td>
<td align="left" valign="top">PCA</td>
<td align="center" valign="top">18,000</td>
<td align="char" valign="top" char=".">93.44</td>
<td align="char" valign="top" char=".">94.78</td>
<td align="char" valign="top" char=".">94.16</td>
<td align="char" valign="top" char=".">93.87</td>
</tr>
<tr>
<td align="left" valign="top">SVM</td>
<td align="left" valign="top">PCA</td>
<td align="center" valign="top">3,200</td>
<td align="char" valign="top" char=".">95.59</td>
<td align="char" valign="top" char=".">96.48</td>
<td align="char" valign="top" char=".">96.29</td>
<td align="char" valign="top" char=".">95.91</td>
</tr>
<tr>
<td align="left" valign="top">XGBoost</td>
<td align="left" valign="top">PCA</td>
<td align="center" valign="top">20,000</td>
<td align="char" valign="top" char=".">99.18</td>
<td align="char" valign="top" char=".">99.94</td>
<td align="char" valign="top" char=".">99.87</td>
<td align="char" valign="top" char=".">99.64</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Predictive performance of BreCML. <bold>(A)</bold> ROC curves for BreCML on test dataset. <bold>(B)</bold> The confusion matrix shows the accuracy of BreCML using 360 genes from BreCML algorithm on test dataset. <bold>(C)</bold> ROC curves and AUC show the performance of the BreCML with other state-of-the-art methods on an independent dataset. <bold>(D)</bold> Based on the BreCML optimal gene set, the confusion matrix of BreCML on the independent dataset.</p>
</caption>
<graphic xlink:href="fmed-11-1482726-g003.tif"/>
</fig>
</sec>
<sec id="sec5">
<title>Predictive performance of BreCML on an independent test set</title>
<p>To evaluate the robustness of the proposed BreCML for breast cancer cell subpopulation prediction, we assessed the performance of BreCML in an independent dataset and compared it with two state-of-the-art methods: eHSCPr, and HelPredictor. To ensure a fair comparison, these models were executed and evaluated using the same independent test set containing 360 genes. As shown in <xref ref-type="table" rid="tab2">Table 2</xref>, BreCML achieved the best performance among all of the tested methods, with an accuracy of 94.78%, precision of 94.98%, recall of 95.54%, and F1-measure of 95.79%. Specifically, compared to other existing methods, the accuracy of our method is higher by 4.66 to 6.53%. The AUC of the three methods is shown in <xref ref-type="fig" rid="fig3">Figure 3C</xref>, and the AUC of BreCML is 0.97, which is outperformed by the other prediction models. Furthermore, the confusion matrix further validated the predictive performance of the model for each cell subpopulation, and the low misclassification rate demonstrated the power of the BreCML model (<xref ref-type="fig" rid="fig3">Figure 3D</xref>). Therefore, we conclude that our method is more effective than eHSCPr, and HelPredictor in predicting breast cancer cell subpopulations.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Performance comparison between BreCML and the other algorithms (independent dataset).</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Method</th>
<th align="center" valign="top">Accuracy</th>
<th align="center" valign="top">Precision</th>
<th align="center" valign="top">Recall</th>
<th align="center" valign="top">F1-measure</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">BreCML</td>
<td align="char" valign="top" char=".">94.78</td>
<td align="char" valign="top" char=".">94.98</td>
<td align="char" valign="top" char=".">95.54</td>
<td align="char" valign="top" char=".">95.79</td>
</tr>
<tr>
<td align="left" valign="top">eHSCPr</td>
<td align="char" valign="top" char=".">88.25</td>
<td align="char" valign="top" char=".">87.29</td>
<td align="char" valign="top" char=".">87.16</td>
<td align="char" valign="top" char=".">85.71</td>
</tr>
<tr>
<td align="left" valign="top">HelPredictor</td>
<td align="char" valign="top" char=".">90.12</td>
<td align="char" valign="top" char=".">88.02</td>
<td align="char" valign="top" char=".">88.32</td>
<td align="char" valign="top" char=".">96.33</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec6">
<title>Expression analysis of the BreCML gene set</title>
<p>Further analysis was performed using Uniform Manifold Approximation and Projection (UMAP) on 4,874 single cells to evaluate the comparative performance of the 360 selected genes against the entire gene set. The results demonstrated that the 360 selected genes significantly outperformed the full gene set in terms of clustering efficiency and specificity. When clustering was conducted using all genes, samples from different categories were almost entirely intermingled, resulting in poor classification outcomes (<xref ref-type="fig" rid="fig4">Figure 4A</xref>). In contrast, the application of the top 360 genes produced a clear and distinct distribution of cell subpopulations (<xref ref-type="fig" rid="fig4">Figure 4B</xref>). This enhanced clustering not only improved the visual differentiation of categories but also underscored the effectiveness of selecting key marker genes for precise subpopulation identification.</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>The clustering effect on 4,874 cells was evaluated using 110 marker genes and all genes (<bold>A</bold> represents all genes, <bold>B</bold> represents the 110 marker genes). Each point represents a sample in the dataset, and different categories of samples are given different colors.</p>
</caption>
<graphic xlink:href="fmed-11-1482726-g004.tif"/>
</fig>
<p>Additionally, we explored the representation of the 360 marker genes across the biological landscape, identifying several key genes that serve as robust markers for specific cell types within the immune system. Notably, genes such as <italic>CD4</italic>, <italic>IL7R</italic>, and <italic>CD3D</italic> were found to be highly expressed in T-cells, underscoring their importance in cellular immunity functions. Similarly, <italic>CD68</italic> was predominantly expressed in myeloid cells, while <italic>MS4A1</italic> was identified as a characteristic gene of a B-cell subpopulation (<xref ref-type="fig" rid="fig5">Figure 5</xref>). These genes have undergone rigorous validation, and their expression patterns have been corroborated by extensive literature, highlighting their biological relevance and utility in cellular characterization.</p>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>UMAP shows marker genes for human breast cancer cell fate determination.</p>
</caption>
<graphic xlink:href="fmed-11-1482726-g005.tif"/>
</fig>
<p>Using multiple genes to characterize cellular subpopulations significantly enhances accuracy. For instance, <italic>GAPDH</italic>, <italic>RPL22</italic>, <italic>RPS12</italic>, <italic>RPS6</italic>, and <italic>RPS18</italic> were crucial in identifying epithelial subpopulations. Similarly, <italic>CD74</italic>, <italic>HLA-DRA</italic>, and <italic>HLA-DPA1</italic> exhibited high expression levels in B-cells, while <italic>SSR4</italic>, <italic>B2M</italic>, <italic>MZB1</italic>, <italic>HERPUD1</italic>, and <italic>XBP1</italic> were prominently expressed in plasmablasts. Additionally, <italic>TMSB4X</italic>, <italic>ZFP36L2</italic>, <italic>HLA-A</italic>, and <italic>CCL5</italic> were highly expressed in T-cells (<xref ref-type="fig" rid="fig6">Figure 6</xref>). This multi-gene approach not only improves the precision of cell type identification but also provides a more comprehensive understanding of the molecular signatures associated with different cellular subpopulations.</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>High expression marker genes screened by Scanpy.</p>
</caption>
<graphic xlink:href="fmed-11-1482726-g006.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="conclusions" id="sec7">
<title>Conclusion</title>
<p>Breast cancer is one of the most prevalent malignant tumors in women, and single-cell RNA sequencing technology plays a crucial role in uncovering its tumor heterogeneity and developmental mechanisms. In this study, we utilized single-cell sequencing technology to conduct an in-depth analysis of various cell subpopulations and their molecular characteristics within breast cancer tissues. We designed and developed a machine learning-based prediction model, BreCML, which demonstrated exceptional performance in predicting breast cancer cell subpopulations. This was evidenced by the results from an independent dataset, where BreCML achieved an accuracy of 98.92% and an ROC of 0.97. BreCML addresses the computational inefficiency and overfitting issues typically associated with the high-dimensional feature space, thereby significantly enhancing the prediction accuracy and robustness of the model. Moreover, by analyzing the BreCML model, we identified a set of key genes that can serve as biomarkers for breast cancer cell subpopulations. These markers hold promise for providing new breakthroughs in early diagnosis and personalized treatment.</p>
<p>However, this study is certainly not without its limitations. A major limitation is the small sample size, and collaborative efforts in data collection may help to improve the model. Despite this potential limitation of the current study, our work provides a resource for studying biomarkers of breast cancer cell subpopulations at single-cell resolution. This not only enhances our understanding of the molecular mechanisms underlying breast cancer but also provides a vital molecular tool for assessing the complexity of breast cancer cell subpopulations, with profound implications for future clinical research.</p>
</sec>
<sec sec-type="materials|methods" id="sec8">
<title>Methods and materials</title>
<sec id="sec9">
<title>Dataset construction and preprocessing</title>
<p>Single-cell transcriptome data for human breast cancer were obtained from the National Center for Biotechnology Information (GSE176078) and include 4,874 cells (<xref ref-type="bibr" rid="ref13">13</xref>). GSE176078 is one of the most comprehensive scRNA-seq dataset specifically focused on breast cancer, including a wide variety of breast cancer subpopulations. This dataset offers detailed transcriptional profiles of thousands of individual cells from multiple patients, making it highly suitable for studying intratumoral heterogeneity, identifying distinct cell subpopulations, and exploring cell-specific biomarkers. The BCL files were demultiplexed and aligned to the GRCh38 reference genome using Cell Ranger Single Cell software v2.0 (10&#x00D7; Genomics). Cell filtering was performed with the EmptyDrops method from the DropletUtils package v1.2.2, applying additional criteria: cells with more than 200 genes and 250 unique molecular identifiers, and a mitochondrial gene percentage below 20%. The dataset included five different cell subpopulations: B-cells (773), Cancer Epithelisl (1,184), Myeloid (897), Plasmablasts (1,020), and T-cells (1,000). The dataset was split into a training dataset and a test dataset at a 8:2 ratio. More dataset details are provided in the <xref ref-type="supplementary-material" rid="SM1">Supplementary Table S1</xref>. The Python packages Numpy (version 1.21.6), Pandas (version 1.3.5) and Scanpy (version 1.9.1) were used to read and process the data.</p>
<p>To further validate the robustness of BreCML, we collected single-cell transcriptome data of breast cancer from Wu et al. (<xref ref-type="bibr" rid="ref14">14</xref>) in the NCBI database (GSE158677). The dataset also included five different cell subpopulations: B cells (598), Cancer epithelial cells (600), myeloid cells (601), Plasmablasts cells (600), and T cells (599). This dataset was used as an independent test set to evaluate the BreCML performance (<xref ref-type="supplementary-material" rid="SM1">Supplementary Table S2</xref>).</p>
</sec>
<sec id="sec10">
<title>Biological analysis and visualization</title>
<p>In this study, we performed an extensive analysis to evaluate the predictive capability of 360 marker genes in identifying cell subpopulations. For the clustering analysis in <xref ref-type="fig" rid="fig4">Figure 4</xref>, UMAP visualization was executed using the python package umap-learn (version 0.3.9), with all settings maintained at default values. For the clustering analysis in <xref ref-type="fig" rid="fig5">Figure 5</xref>, we used the Preprocessing and clustering module in Scanpy (version 1.9.1), which facilitated to identify specific subpopulations of cells associated with these marker genes; default parameters were used throughout. Pearson correlation analysis was conducted on five distinct human breast cancer cell populations, based on the expression profiles of the 360 marker genes, using Pandas (version 1.4.4).</p>
</sec>
<sec id="sec11">
<title>Principal component analysis</title>
<p>Feature-scML is a scalable and friendly toolkit that allows users to comprehensively score and rank each feature in scRNA-seq data. The PCA module of Feature-scML was used to assess the feature importance of each gene. The source code is available at <ext-link xlink:href="https://github.com/liameihao/Feature-scML" ext-link-type="uri">https://github.com/liameihao/Feature-scML</ext-link>.</p>
</sec>
<sec id="sec12">
<title><italic>F</italic>-score algorithm</title>
<p>The <italic>F</italic>-score can be used to measure the degree of differentiation of features in different categories and has been shown to be a simple and effective method for feature selection. This method significantly improves the interpretability and classification performance of the model while reducing the bias (<xref ref-type="bibr" rid="ref15">15</xref>). The <italic>F</italic>-score of the <italic>i</italic>th feature is defined as (<xref ref-type="bibr" rid="ref16">16</xref>, <xref ref-type="bibr" rid="ref17">17</xref>):</p><disp-formula id="E1">
<label>(1)</label>
<mml:math id="M1">
<mml:mi>F</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mover accent="true">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
<mml:mfenced open="(" close=")">
<mml:mo>+</mml:mo>
</mml:mfenced>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mover accent="true">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
<mml:mfenced open="(" close=")">
<mml:mo>&#x2212;</mml:mo>
</mml:mfenced>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mo>+</mml:mo>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>+</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>X</mml:mi>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mo>+</mml:mo>
</mml:mfenced>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mover accent="true">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
<mml:mfenced open="(" close=")">
<mml:mo>+</mml:mo>
</mml:mfenced>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mover accent="true">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
<mml:mfenced open="(" close=")">
<mml:mo>&#x2212;</mml:mo>
</mml:mfenced>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula><p>where <inline-formula>
<mml:math id="M3">
<mml:mover accent="true">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> represents the average of the ith feature of the whole <inline-formula>
<mml:math id="M4">
<mml:msup>
<mml:mover accent="true">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
<mml:mfenced open="(" close=")">
<mml:mo>+</mml:mo>
</mml:mfenced>
</mml:msup>
</mml:math>
</inline-formula> is the number of positive samples, <inline-formula>
<mml:math id="M5">
<mml:msup>
<mml:mover accent="true">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
<mml:mfenced open="(" close=")">
<mml:mo>&#x2212;</mml:mo>
</mml:mfenced>
</mml:msup>
</mml:math>
</inline-formula> is the number of negative samples. <inline-formula>
<mml:math id="M6">
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>X</mml:mi>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mo>+</mml:mo>
</mml:mfenced>
</mml:msubsup>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M7">
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>X</mml:mi>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mo>&#x2212;</mml:mo>
</mml:mfenced>
</mml:msubsup>
</mml:math>
</inline-formula> are the ith feature of the <italic>k</italic>th positive and negative instances, respectively. The larger the <italic>F</italic>-score value, the stronger the distinguishing degree of the feature among different categories.</p>
</sec>
<sec id="sec13">
<title>Squared coefficient of variation</title>
<p>The squared coefficient of variation (CV<sup>2</sup>) is a quantitative statistical method for quantifying technical variation at the gene level and assessing variability in cell biology, and is widely used in the field of single-cell experiments (<xref ref-type="bibr" rid="ref18">18</xref>). The CV<sup>2</sup> method operates by calculating the squares of the coefficients of variation and curve-fitting the observations using the generalized linear model (GLM) in the R package statmod,</p><disp-formula id="E2">
<label>(2)</label>
<mml:math id="M8">
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">V</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>&#x03BC;</mml:mi>
</mml:mfrac>
<mml:mo>+</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:mn>0</mml:mn>
</mml:math>
</disp-formula>
</sec>
<sec id="sec14">
<title>Extreme gradient boosting</title>
<p>Extreme gradient boosting (XGBoost) is recognized as an exceedingly complex and efficient machine learning algorithm, widely acknowledged for its outstanding performance in predictive modeling competitions (<xref ref-type="bibr" rid="ref19">19</xref>). XGBoost attracts significant attention primarily due to its efficiency and effectiveness demonstrated in various competitive settings. The algorithm operates by sequentially constructing a series of decision trees, each designed to correct the errors of its predecessor. This approach allows the model to capture complex patterns in the data, thereby enhancing predictive accuracy. One major advantage of XGBoost is its ability to quickly and accurately process large datasets, making it an ideal tool for our research. Furthermore, compared to other models such as KNN and SVM, XGBoost also offers a direct method for assessing the importance of each input variable.</p>
</sec>
<sec id="sec15">
<title>Model construction of BreCML</title>
<p>During the exploratory data analysis, it was crucial to identify key relationships and assign appropriate weights to features to filter out less relevant or weaker information. We employed three feature selection techniques PCA, CV<sup>2</sup>, and <italic>F</italic>-score&#x2014;to evaluate and rank the importance of genes in descending order. Genes with weights equal to or below zero were excluded from further analysis. The sorted gene expression profiles of breast cancer cell subpopulations served as input features for training machine learning models. Utilizing the incremental feature selection (IFS) strategy, we formed 12 combinations by integrating the three feature selection methods with four machine learning models: KNN, XGBoost, SVM, and RFC. Grid search was used to determine the optimal parameters for each combination. The optimal gene set for each combination was identified when the accuracy no longer showed improvement with the addition of more genes. Ultimately, the combination of <italic>F</italic>-score and XGBoost proved to be the most effective and was employed to develop the BreCML model.</p>
</sec>
<sec id="sec16">
<title>Model evaluation</title>
<p>The four classic metrics were used to quantify the performance of the model predictions, namely, the accuracy (Acc), recall (Re), precision (Pre), and F1 measure (F1), defined as (<xref ref-type="bibr" rid="ref20 ref21 ref22 ref23 ref24 ref25 ref26 ref27 ref28">20&#x2013;28</xref>):</p><disp-formula id="E3">
<label>(3)</label>
<mml:math id="M10">
<mml:mi mathvariant="normal">Accuracy</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula><disp-formula id="E4">
<label>(4)</label>
<mml:math id="M12">
<mml:mi mathvariant="normal">Recall</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula><disp-formula id="E5">
<label>(5)</label>
<mml:math id="M14">
<mml:mi mathvariant="normal">Precision</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula><disp-formula id="E6">
<label>(6)</label>
<mml:math id="M16">
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mspace width="thickmathspace"/>
<mml:mi mathvariant="normal">measure</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#x2217;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="normal">precision</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi mathvariant="normal">recall</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">precision</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">recall</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula><p>where TP, TN, FP, and FN represent the numbers of true positives, true negatives, false positives and false negatives, respectively. In addition, the ROC curve was used to evaluate the performance of the BreCML (<xref ref-type="bibr" rid="ref29 ref30 ref31 ref32">29&#x2013;32</xref>).</p>
</sec>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec17">
<title>Data availability statement</title>
<p>Single-cell transcriptome data for human breast cancer were obtained from the National Center for Biotechnology Information (GSE176078 and GSE158677). The processed data from this study are available at <ext-link xlink:href="https://www.jianguoyun.com/p/DV0r_W0QsKH7DBictdkFIAA" ext-link-type="uri">https://www.jianguoyun.com/p/DV0r_W0QsKH7DBictdkFIAA</ext-link>.</p>
</sec>
<sec sec-type="ethics-statement" id="sec18">
<title>Ethics statement</title>
<p>Ethical approval was not required for the studies involving humans because all data for this study were obtained from public databases and are publicly accessible. The studies were conducted in accordance with the local legislation and institutional requirements. Written informed consent was not obtained from the individual(s) for the publication of any potentially identifiable images or data included in this article because All data for this study were obtained from public databases and are publicly accessible.</p>
</sec>
<sec sec-type="author-contributions" id="sec19">
<title>Author contributions</title>
<p>SK: Software, Writing &#x2013; original draft. YH: Formal analysis, Writing &#x2013; original draft. DW: Data curation, Resources, Writing &#x2013; original draft. QJ: Data curation, Formal analysis, Writing &#x2013; original draft. ZL: Data curation, Resources, Writing &#x2013; original draft. BL: Data curation, Formal analysis, Software, Writing &#x2013; original draft. DY: Data curation, Formal analysis, Software, Writing &#x2013; original draft. JZ: Supervision, Writing &#x2013; original draft.</p>
</sec>
<sec sec-type="funding-information" id="sec20">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work is being supported by the Key Discipline Construction Project of Pudong Health Bureau of Shanghai: Clinical Pharmacy (Grant No. PWZxk2022-27).</p>
</sec>
<sec sec-type="COI-statement" id="sec21">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="sec22">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec23">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fmed.2024.1482726/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fmed.2024.1482726/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.DOCX" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1">
<label>1.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sung</surname> <given-names>H</given-names></name> <name><surname>Ferlay</surname> <given-names>J</given-names></name> <name><surname>Siegel</surname> <given-names>RL</given-names></name> <name><surname>Laversanne</surname> <given-names>M</given-names></name> <name><surname>Soerjomataram</surname> <given-names>I</given-names></name> <name><surname>Jemal</surname> <given-names>A</given-names></name> <etal/></person-group>. <article-title>Global cancer statistics 2020: GLOBOCAN estimates of incidence and mortality worldwide for 36 cancers in 185 countries</article-title>. <source>CA Cancer J Clin</source>. (<year>2021</year>) <volume>71</volume>:<fpage>209</fpage>&#x2013;<lpage>49</lpage>. doi: <pub-id pub-id-type="doi">10.3322/caac.21660</pub-id></citation>
</ref>
<ref id="ref2">
<label>2.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bray</surname> <given-names>F</given-names></name> <name><surname>Laversanne</surname> <given-names>M</given-names></name> <name><surname>Sung</surname> <given-names>H</given-names></name> <name><surname>Ferlay</surname> <given-names>J</given-names></name> <name><surname>Siegel</surname> <given-names>RL</given-names></name> <name><surname>Soerjomataram</surname> <given-names>I</given-names></name> <etal/></person-group>. <article-title>Global cancer statistics 2022: GLOBOCAN estimates of incidence and mortality worldwide for 36 cancers in 185 countries</article-title>. <source>CA Cancer J Clin</source>. (<year>2024</year>) <volume>74</volume>:<fpage>229</fpage>&#x2013;<lpage>63</lpage>. doi: <pub-id pub-id-type="doi">10.3322/caac.21834</pub-id>, PMID: <pub-id pub-id-type="pmid">38572751</pub-id></citation>
</ref>
<ref id="ref3">
<label>3.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Clemons</surname> <given-names>M</given-names></name> <name><surname>Goss</surname> <given-names>P</given-names></name></person-group>. <article-title>Estrogen and the risk of breast cancer</article-title>. <source>N Engl J Med</source>. (<year>2001</year>) <volume>344</volume>:<fpage>276</fpage>&#x2013;<lpage>85</lpage>. doi: <pub-id pub-id-type="doi">10.1056/NEJM200101253440407</pub-id></citation>
</ref>
<ref id="ref4">
<label>4.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pal</surname> <given-names>B</given-names></name> <name><surname>Chen</surname> <given-names>Y</given-names></name> <name><surname>Vaillant</surname> <given-names>F</given-names></name> <name><surname>Capaldo</surname> <given-names>BD</given-names></name> <name><surname>Joyce</surname> <given-names>R</given-names></name> <name><surname>Song</surname> <given-names>X</given-names></name> <etal/></person-group>. <article-title>A single-cell RNA expression atlas of normal, preneoplastic and tumorigenic states in the human breast</article-title>. <source>EMBO J</source>. (<year>2021</year>) <volume>40</volume>:<fpage>e107333</fpage>. doi: <pub-id pub-id-type="doi">10.15252/embj.2020107333</pub-id>, PMID: <pub-id pub-id-type="pmid">33950524</pub-id></citation>
</ref>
<ref id="ref5">
<label>5.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hankinson</surname> <given-names>SE</given-names></name> <name><surname>Colditz</surname> <given-names>GA</given-names></name> <name><surname>Willett</surname> <given-names>WC</given-names></name></person-group>. <article-title>Towards an integrated model for breast cancer etiology: the lifelong interplay of genes, lifestyle, and hormones</article-title>. <source>Breast Cancer Res</source>. (<year>2004</year>) <volume>6</volume>:<fpage>213</fpage>&#x2013;<lpage>8</lpage>. doi: <pub-id pub-id-type="doi">10.1186/bcr921</pub-id>, PMID: <pub-id pub-id-type="pmid">15318928</pub-id></citation>
</ref>
<ref id="ref6">
<label>6.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sharma</surname> <given-names>P</given-names></name> <name><surname>Allison</surname> <given-names>JP</given-names></name></person-group>. <article-title>The future of immune checkpoint therapy</article-title>. <source>Science</source>. (<year>2015</year>) <volume>348</volume>:<fpage>56</fpage>&#x2013;<lpage>61</lpage>. doi: <pub-id pub-id-type="doi">10.1126/science.aaa8172</pub-id></citation>
</ref>
<ref id="ref7">
<label>7.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vonderheide</surname> <given-names>RH</given-names></name> <name><surname>Domchek</surname> <given-names>SM</given-names></name> <name><surname>Clark</surname> <given-names>AS</given-names></name></person-group>. <article-title>Immunotherapy for breast cancer: what are we missing?</article-title> <source>Clin Cancer Res</source>. (<year>2017</year>) <volume>23</volume>:<fpage>2640</fpage>&#x2013;<lpage>6</lpage>. doi: <pub-id pub-id-type="doi">10.1158/1078-0432.CCR-16-2569</pub-id>, PMID: <pub-id pub-id-type="pmid">28572258</pub-id></citation>
</ref>
<ref id="ref8">
<label>8.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y</given-names></name> <name><surname>Waters</surname> <given-names>J</given-names></name> <name><surname>Leung</surname> <given-names>ML</given-names></name> <name><surname>Unruh</surname> <given-names>A</given-names></name> <name><surname>Roh</surname> <given-names>W</given-names></name> <name><surname>Shi</surname> <given-names>X</given-names></name> <etal/></person-group>. <article-title>Clonal evolution in breast cancer revealed by single nucleus genome sequencing</article-title>. <source>Nature</source>. (<year>2014</year>) <volume>512</volume>:<fpage>155</fpage>&#x2013;<lpage>60</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nature13600</pub-id>, PMID: <pub-id pub-id-type="pmid">25079324</pub-id></citation>
</ref>
<ref id="ref9">
<label>9.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chung</surname> <given-names>W</given-names></name> <name><surname>Eum</surname> <given-names>HH</given-names></name> <name><surname>Lee</surname> <given-names>HO</given-names></name> <name><surname>Lee</surname> <given-names>KM</given-names></name> <name><surname>Lee</surname> <given-names>HB</given-names></name> <name><surname>Kim</surname> <given-names>KT</given-names></name> <etal/></person-group>. <article-title>Single-cell RNA-seq enables comprehensive tumour and immune cell profiling in primary breast cancer</article-title>. <source>Nat Commun</source>. (<year>2017</year>) <volume>8</volume>:<fpage>15081</fpage>. doi: <pub-id pub-id-type="doi">10.1038/ncomms15081</pub-id></citation>
</ref>
<ref id="ref10">
<label>10.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jang</surname> <given-names>BS</given-names></name> <name><surname>Han</surname> <given-names>W</given-names></name> <name><surname>Kim</surname> <given-names>IA</given-names></name></person-group>. <article-title>Tumor mutation burden, immune checkpoint crosstalk and radiosensitivity in single-cell RNA sequencing data of breast cancer</article-title>. <source>Radiother Oncol</source>. (<year>2020</year>) <volume>142</volume>:<fpage>202</fpage>&#x2013;<lpage>9</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.radonc.2019.11.003</pub-id>, PMID: <pub-id pub-id-type="pmid">31767471</pub-id></citation>
</ref>
<ref id="ref11">
<label>11.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>SQ</given-names></name> <name><surname>Gao</surname> <given-names>ZJ</given-names></name> <name><surname>Wu</surname> <given-names>J</given-names></name> <name><surname>Zheng</surname> <given-names>HM</given-names></name> <name><surname>Li</surname> <given-names>B</given-names></name> <name><surname>Sun</surname> <given-names>S</given-names></name> <etal/></person-group>. <article-title>Single-cell and spatially resolved analysis uncovers cell heterogeneity of breast cancer</article-title>. <source>J Hematol Oncol</source>. (<year>2022</year>) <volume>15</volume>:<fpage>19</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13045-022-01236-0</pub-id></citation>
</ref>
<ref id="ref12">
<label>12.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>S</given-names></name> <name><surname>Chen</surname> <given-names>X</given-names></name> <name><surname>Shen</surname> <given-names>K</given-names></name></person-group>. <article-title>Single-cell RNA sequencing in breast cancer: understanding tumor heterogeneity and paving roads to individualized therapy</article-title>. <source>Cancer Commun</source>. (<year>2020</year>) <volume>40</volume>:<fpage>329</fpage>&#x2013;<lpage>44</lpage>. doi: <pub-id pub-id-type="doi">10.1002/cac2.12078</pub-id>, PMID: <pub-id pub-id-type="pmid">32654419</pub-id></citation>
</ref>
<ref id="ref13">
<label>13.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>SZ</given-names></name> <name><surname>Al-Eryani</surname> <given-names>G</given-names></name> <name><surname>Roden</surname> <given-names>DL</given-names></name> <name><surname>Junankar</surname> <given-names>S</given-names></name> <name><surname>Harvey</surname> <given-names>K</given-names></name> <name><surname>Andersson</surname> <given-names>A</given-names></name> <etal/></person-group>. <article-title>A single-cell and spatially resolved atlas of human breast cancers</article-title>. <source>Nat Genet</source>. (<year>2021</year>) <volume>53</volume>:<fpage>1334</fpage>&#x2013;<lpage>47</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41588-021-00911-1</pub-id>, PMID: <pub-id pub-id-type="pmid">34493872</pub-id></citation>
</ref>
<ref id="ref14">
<label>14.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Papanicolaou</surname> <given-names>M</given-names></name> <name><surname>Parker</surname> <given-names>AL</given-names></name> <name><surname>Yam</surname> <given-names>M</given-names></name> <name><surname>Filipe</surname> <given-names>EC</given-names></name> <name><surname>Wu</surname> <given-names>SZ</given-names></name> <name><surname>Chitty</surname> <given-names>JL</given-names></name> <etal/></person-group>. <article-title>Temporal profiling of the breast tumour microenvironment reveals collagen XII as a driver of metastasis</article-title>. <source>Nat Commun</source>. (<year>2022</year>) <volume>13</volume>:<fpage>4587</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41467-022-32255-7</pub-id>, PMID: <pub-id pub-id-type="pmid">35933466</pub-id></citation>
</ref>
<ref id="ref15">
<label>15.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>W</given-names></name> <name><surname>Jia</surname> <given-names>C</given-names></name> <name><surname>Zou</surname> <given-names>Q</given-names></name></person-group>. <article-title>4mCPred: machine learning methods for DNA N4-methylcytosine sites prediction</article-title>. <source>Bioinformatics</source>. (<year>2019</year>) <volume>35</volume>:<fpage>593</fpage>&#x2013;<lpage>601</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/bty668</pub-id>, PMID: <pub-id pub-id-type="pmid">30052767</pub-id></citation>
</ref>
<ref id="ref16">
<label>16.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>M</given-names></name> <name><surname>Li</surname> <given-names>F</given-names></name> <name><surname>Marquez-Lago</surname> <given-names>TT</given-names></name> <name><surname>Leier</surname> <given-names>A</given-names></name> <name><surname>Fan</surname> <given-names>C</given-names></name> <name><surname>Kwoh</surname> <given-names>CK</given-names></name> <etal/></person-group>. <article-title>MULTiPly: a novel multi-layer predictor for discovering general and specific types of promoters</article-title>. <source>Bioinformatics</source>. (<year>2019</year>) <volume>35</volume>:<fpage>2957</fpage>&#x2013;<lpage>65</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btz016</pub-id>, PMID: <pub-id pub-id-type="pmid">30649179</pub-id></citation>
</ref>
<ref id="ref17">
<label>17.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H</given-names></name> <name><surname>Liang</surname> <given-names>P</given-names></name> <name><surname>Zheng</surname> <given-names>L</given-names></name> <name><surname>Long</surname> <given-names>CS</given-names></name> <name><surname>Li</surname> <given-names>HS</given-names></name> <name><surname>Zuo</surname> <given-names>Y</given-names></name></person-group>. <article-title>eHSCPr discriminating the cell identity involved in endothelial to hematopoietic transition</article-title>. <source>Bioinformatics</source>. (<year>2021</year>) <volume>37</volume>:<fpage>2157</fpage>&#x2013;<lpage>64</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btab071</pub-id>, PMID: <pub-id pub-id-type="pmid">33532815</pub-id></citation>
</ref>
<ref id="ref18">
<label>18.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Conesa</surname> <given-names>A</given-names></name> <name><surname>Madrigal</surname> <given-names>P</given-names></name> <name><surname>Tarazona</surname> <given-names>S</given-names></name> <name><surname>Gomez-Cabrero</surname> <given-names>D</given-names></name> <name><surname>Cervera</surname> <given-names>A</given-names></name> <name><surname>McPherson</surname> <given-names>A</given-names></name> <etal/></person-group>. <article-title>A survey of best practices for RNA-seq data analysis</article-title>. <source>Genome Biol</source>. (<year>2016</year>) <volume>17</volume>:<fpage>13</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13059-016-0881-8</pub-id>, PMID: <pub-id pub-id-type="pmid">26813401</pub-id></citation>
</ref>
<ref id="ref19">
<label>19.</label>
<citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>T</given-names></name> <name><surname>Guestrin</surname> <given-names>C</given-names></name></person-group> <article-title>XGBoost: a scalable tree boosting system</article-title>. (<year>2016</year>). <conf-name>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</conf-name>.</citation>
</ref>
<ref id="ref20">
<label>20.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Joshi</surname> <given-names>P</given-names></name> <name><surname>Masilamani</surname> <given-names>V</given-names></name> <name><surname>Ramesh</surname> <given-names>R</given-names></name></person-group>. <article-title>An ensembled SVM based approach for predicting adverse drug reactions</article-title>. <source>Curr Bioinforma</source>. (<year>2021</year>) <volume>16</volume>:<fpage>422</fpage>&#x2013;<lpage>32</lpage>. doi: <pub-id pub-id-type="doi">10.2174/1574893615999200707141420</pub-id></citation>
</ref>
<ref id="ref21">
<label>21.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Geete</surname> <given-names>K</given-names></name> <name><surname>Pandey</surname> <given-names>M</given-names></name></person-group>. <article-title>Robust transcription factor binding site prediction using deep neural networks</article-title>. <source>Curr Bioinform</source>. (<year>2020</year>) <volume>15</volume>:<fpage>1137</fpage>&#x2013;<lpage>52</lpage>. doi: <pub-id pub-id-type="doi">10.2174/1574893615999200429121156</pub-id></citation>
</ref>
<ref id="ref22">
<label>22.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ao</surname> <given-names>C</given-names></name> <name><surname>Zhou</surname> <given-names>W</given-names></name> <name><surname>Gao</surname> <given-names>L</given-names></name> <name><surname>Dong</surname> <given-names>B</given-names></name> <name><surname>Yu</surname> <given-names>L</given-names></name></person-group>. <article-title>Prediction of antioxidant proteins using hybrid feature representation method and random forest</article-title>. <source>Genomics</source>. (<year>2020</year>) <volume>112</volume>:<fpage>4666</fpage>&#x2013;<lpage>74</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ygeno.2020.08.016</pub-id>, PMID: <pub-id pub-id-type="pmid">32818637</pub-id></citation>
</ref>
<ref id="ref23">
<label>23.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname> <given-names>X</given-names></name> <name><surname>Zhu</surname> <given-names>W</given-names></name> <name><surname>Cai</surname> <given-names>L</given-names></name> <name><surname>Liao</surname> <given-names>B</given-names></name> <name><surname>Peng</surname> <given-names>L</given-names></name> <name><surname>Chen</surname> <given-names>Y</given-names></name> <etal/></person-group>. <article-title>Improved pre-miRNAs identification through mutual information of pre-miRNA sequences and structures</article-title>. <source>Front Genet</source>. (<year>2019</year>) <volume>10</volume>:<fpage>119</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fgene.2019.00119</pub-id>, PMID: <pub-id pub-id-type="pmid">30858864</pub-id></citation>
</ref>
<ref id="ref24">
<label>24.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname> <given-names>X</given-names></name> <name><surname>Liao</surname> <given-names>B</given-names></name> <name><surname>Zhu</surname> <given-names>W</given-names></name> <name><surname>Cai</surname> <given-names>L</given-names></name></person-group>. <article-title>New 3D graphical representation for RNA structure analysis and its application in the pre-miRNA identification of plants</article-title>. <source>RSC Adv</source>. (<year>2018</year>) <volume>8</volume>:<fpage>30833</fpage>&#x2013;<lpage>41</lpage>. doi: <pub-id pub-id-type="doi">10.1039/C8RA04138E</pub-id>, PMID: <pub-id pub-id-type="pmid">35548744</pub-id></citation>
</ref>
<ref id="ref25">
<label>25.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qian</surname> <given-names>Y</given-names></name> <name><surname>Ding</surname> <given-names>Y</given-names></name> <name><surname>Zou</surname> <given-names>Q</given-names></name> <name><surname>Guo</surname> <given-names>F</given-names></name></person-group>. <article-title>Multi-view kernel sparse representation for identification of membrane protein types</article-title>. <source>IEEE/ACM Trans Comput Biol Bioinform</source>. (<year>2023</year>) <volume>20</volume>:<fpage>1234</fpage>&#x2013;<lpage>45</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TCBB.2022.3191325</pub-id>, PMID: <pub-id pub-id-type="pmid">35857734</pub-id></citation>
</ref>
<ref id="ref26">
<label>26.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ai</surname> <given-names>C</given-names></name> <name><surname>Yang</surname> <given-names>H</given-names></name> <name><surname>Ding</surname> <given-names>Y</given-names></name> <name><surname>Tang</surname> <given-names>J</given-names></name> <name><surname>Guo</surname> <given-names>F</given-names></name></person-group>. <article-title>Low rank matrix factorization algorithm based on multi-graph regularization for detecting drug-disease association</article-title>. <source>IEEE/ACM Trans Comput Biol Bioinform</source>. (<year>2023</year>) <volume>20</volume>:<fpage>3033</fpage>&#x2013;<lpage>43</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TCBB.2023.3274587</pub-id>, PMID: <pub-id pub-id-type="pmid">37159322</pub-id></citation>
</ref>
<ref id="ref27">
<label>27.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tang</surname> <given-names>Y</given-names></name> <name><surname>Pang</surname> <given-names>Y</given-names></name> <name><surname>Liu</surname> <given-names>B</given-names></name></person-group>. <article-title>IDP-Seq2Seq: identification of intrinsically disordered regions based on sequence to sequence learning</article-title>. <source>Bioinformatics</source>. (<year>2021</year>) <volume>36</volume>:<fpage>5177</fpage>&#x2013;<lpage>86</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa667</pub-id>, PMID: <pub-id pub-id-type="pmid">32702119</pub-id></citation>
</ref>
<ref id="ref28">
<label>28.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>X</given-names></name> <name><surname>Wang</surname> <given-names>H</given-names></name> <name><surname>Wu</surname> <given-names>S</given-names></name> <name><surname>Han</surname> <given-names>B</given-names></name> <name><surname>Cui</surname> <given-names>D</given-names></name> <name><surname>Liu</surname> <given-names>J</given-names></name> <etal/></person-group>. <article-title>DeepCCR: large-scale genomics-based deep learning method for improving rice breeding</article-title>. <source>Plant Biotechnol J</source>. (<year>2024</year>) <volume>22</volume>:<fpage>2691</fpage>&#x2013;<lpage>3</lpage>. doi: <pub-id pub-id-type="doi">10.1111/pbi.14384</pub-id></citation>
</ref>
<ref id="ref29">
<label>29.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zeng</surname> <given-names>X</given-names></name> <name><surname>Zhang</surname> <given-names>X</given-names></name> <name><surname>Zou</surname> <given-names>Q</given-names></name></person-group>. <article-title>Integrative approaches for predicting microRNA function and prioritizing disease-related microRNA using biological interaction networks</article-title>. <source>Brief Bioinform</source>. (<year>2016</year>) <volume>17</volume>:<fpage>193</fpage>&#x2013;<lpage>203</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bib/bbv033</pub-id></citation>
</ref>
<ref id="ref30">
<label>30.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zulfiqar</surname> <given-names>H</given-names></name> <name><surname>Guo</surname> <given-names>Z</given-names></name> <name><surname>Ahmad</surname> <given-names>RM</given-names></name> <name><surname>Ahmed</surname> <given-names>Z</given-names></name> <name><surname>Cai</surname> <given-names>P</given-names></name> <name><surname>Chen</surname> <given-names>X</given-names></name> <etal/></person-group>. <article-title>Deep-STP: a deep learning-based approach to predict snake toxin proteins by using word embeddings</article-title>. <source>Front Med</source>. (<year>2024</year>) <volume>10</volume>:<fpage>1291352</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fmed.2023.1291352</pub-id>, PMID: <pub-id pub-id-type="pmid">38298505</pub-id></citation>
</ref>
<ref id="ref31">
<label>31.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H</given-names></name> <name><surname>Lin</surname> <given-names>YN</given-names></name> <name><surname>Yan</surname> <given-names>S</given-names></name> <name><surname>Hong</surname> <given-names>JP</given-names></name> <name><surname>Tan</surname> <given-names>JR</given-names></name> <name><surname>Chen</surname> <given-names>YQ</given-names></name> <etal/></person-group>. <article-title>NRTPredictor: identifying rice root cell state in single-cell RNA-seq via ensemble learning</article-title>. <source>Plant Methods</source>. (<year>2023</year>) <volume>19</volume>:<fpage>119</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13007-023-01092-0</pub-id>, PMID: <pub-id pub-id-type="pmid">37925413</pub-id></citation>
</ref>
<ref id="ref32">
<label>32.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H</given-names></name> <name><surname>Zhang</surname> <given-names>Z</given-names></name> <name><surname>Li</surname> <given-names>H</given-names></name> <name><surname>Li</surname> <given-names>J</given-names></name> <name><surname>Li</surname> <given-names>H</given-names></name> <name><surname>Liu</surname> <given-names>M</given-names></name> <etal/></person-group>. <article-title>A cost-effective machine learning-based method for preeclampsia risk assessment and driver genes discovery</article-title>. <source>Cell Biosci</source>. (<year>2023</year>) <volume>13</volume>:<fpage>41</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13578-023-00991-y</pub-id>, PMID: <pub-id pub-id-type="pmid">36849879</pub-id></citation>
</ref>
</ref-list>
</back>
</article>