<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Bioeng. Biotechnol.</journal-id>
<journal-title>Frontiers in Bioengineering and Biotechnology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Bioeng. Biotechnol.</abbrev-journal-title>
<issn pub-type="epub">2296-4185</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">890901</article-id>
<article-id pub-id-type="doi">10.3389/fbioe.2022.890901</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Bioengineering and Biotechnology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Identification of Type 2 Diabetes Biomarkers From Mixed Single-Cell Sequencing Data With Feature Selection Methods</article-title>
<alt-title alt-title-type="left-running-head">Li et al.</alt-title>
<alt-title alt-title-type="right-running-head">Identification of T2D Biomarkers</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Zhandong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/928572/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Pan</surname>
<given-names>Xiaoyong</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/857914/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Cai</surname>
<given-names>Yu-Dong</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/103860/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>College of Biological and Food Engineering</institution>, <institution>Jilin Engineering Normal University</institution>, <addr-line>Changchun</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Key Laboratory of System Control and Information Processing</institution>, <institution>Institute of Image Processing and Pattern Recognition</institution>, <institution>Ministry of Education of China</institution>, <institution>Shanghai Jiao Tong University</institution>, <addr-line>Shanghai</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>School of Life Sciences</institution>, <institution>Shanghai University</institution>, <addr-line>Shanghai</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/886979/overview">Jiaofang Shao</ext-link>, Nanjing Medical University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1506820/overview">Bo Zhou</ext-link>, Shanghai University of Medicine and Health Sciences, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1586193/overview">Gele Aori</ext-link>, University of Toyama, Japan</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Yu-Dong Cai, <email>cai_yud@126.com</email>
</corresp>
<fn fn-type="equal" id="fn1">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work</p>
</fn>
<fn fn-type="other">
<p>This article was submitted to Preclinical Cell and Gene Therapy, a section of the journal Frontiers in Bioengineering and Biotechnology</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>02</day>
<month>06</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>10</volume>
<elocation-id>890901</elocation-id>
<history>
<date date-type="received">
<day>07</day>
<month>03</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>04</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Li, Pan and Cai.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Li, Pan and Cai</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Diabetes is the most common disease and a major threat to human health. Type 2 diabetes (T2D) makes up about 90% of all cases. With the development of high-throughput sequencing technologies, more and more fundamental pathogenesis of T2D at genetic and transcriptomic levels has been revealed. The recent single-cell sequencing can further reveal the cellular heterogenicity of complex diseases in an unprecedented way. With the expectation on the molecular essence of T2D across multiple cell types, we investigated the expression profiling of more than 1,600 single cells (949 cells from T2D patients and 651 cells from normal controls) and identified the differential expression profiling and characteristics at the transcriptomics level that can distinguish such two groups of cells at the single-cell level. The expression profile was analyzed by several machine learning algorithms, including Monte Carlo feature selection, support vector machine, and repeated incremental pruning to produce error reduction (RIPPER). On one hand, some T2D-associated genes (MTND4P24, MTND2P28, and LOC100128906) were discovered. On the other hand, we revealed novel potential pathogenic mechanisms in a rule manner. They are induced by newly recognized genes and neglected by traditional bulk sequencing techniques. Particularly, the newly identified T2D genes were shown to follow specific quantitative rules with diabetes prediction potentials, and such rules further indicated several potential functional crosstalks involved in T2D.</p>
</abstract>
<kwd-group>
<kwd>type 2 diabetes</kwd>
<kwd>single-cell sequencing</kwd>
<kwd>Monte Carlo feature selection</kwd>
<kwd>support vector machine</kwd>
<kwd>RIPPER</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Diabetes mellitus (DM) turns out to be the general term describing metabolic disorders with high blood sugar levels as typical symptoms (<xref ref-type="bibr" rid="B84">Tseng et al., 2012</xref>; <xref ref-type="bibr" rid="B77">Tao et al., 2015</xref>). Due to either lack of insulin or pathogenic insulin reactive responses, diabetes can be divided into three groups: type 1 DM with low insulin production, type 2 DM with insulin resistance, and gestational diabetes with high blood sugar levels induced by diabetes recurrence during pregnancy (<xref ref-type="bibr" rid="B1">American Diabetes Association, 2014</xref>). According to the epidemiologic statistics data in 2015, more than four hundred million people suffered from diabetes, and about five million people died from such disease all over the world (<xref ref-type="bibr" rid="B31">Gao et al., 2016</xref>; <xref ref-type="bibr" rid="B22">Disease and Injury Incidence and Prevalence Collaborators, 2017</xref>; <xref ref-type="bibr" rid="B28">Global Burden of Disease Cancer Collaboration et al., 2017</xref>). Particularly, type 2 diabetes (T2D) makes up about 90% of all cases (392 million) and is the primary subtype of diabetes (<xref ref-type="bibr" rid="B22">Disease and Injury Incidence and Prevalence Collaborators, 2017</xref>; <xref ref-type="bibr" rid="B28">Global Burden of Disease Cancer Collaboration et al., 2017</xref>), indicating such kind of disease is one of the major threats to human health.</p>
<p>Different from type 1 DM and gestational diabetes, the major pathogenesis of type 2 DM is insulin resistance and beta-cell dysfunction accompanied with insufficient insulin secretion (<xref ref-type="bibr" rid="B67">Pandey et al., 2015</xref>), where insulin resistance is generally defined as dysfunctional insulin-mediated glucose clearance (<xref ref-type="bibr" rid="B93">Yabe et al., 2015</xref>). During the pathogenesis of type 2 DM, the typical insulin-associated biological processes and action cascade are usually disturbed by either intracellular signals or extra interferences, including serine phosphorylation of IRS-1, excess glucosamine, mitochondria defects, FA (fatty acid)-induced insulin dysfunction, and alternate fatty acid effects (<xref ref-type="bibr" rid="B79">Taylor, 2013</xref>; <xref ref-type="bibr" rid="B25">Eckardt et al., 2014</xref>; <xref ref-type="bibr" rid="B67">Pandey et al., 2015</xref>). Early in 1997, <xref ref-type="bibr" rid="B4">Boden (1997)</xref> has already demonstrated the significance of fatty acids in diabetes. Further similarly in the same year, functional signaling molecules IRS-1 and IRS-2 were confirmed by <xref ref-type="bibr" rid="B104">Zick (2001)</xref>, revealing the initial biological foundations for diabetes. Apart from such complicated pathogenesis associated with insulin resistance, beta-cell dysfunction has also been widely identified in type 2 DM patients as the other etiological factor. Similar to insulin resistance, such pathogenesis also has various potential mechanisms, including glucose toxicity, beta-cell exhaustion, impaired proinsulin biosynthesis, and lipo-toxicity (<xref ref-type="bibr" rid="B27">Ferrannini, 2009</xref>). In 2003, <xref ref-type="bibr" rid="B41">Kahn (2003)</xref> demonstrated the specific contribution of both insulin resistance and beta-cell dysfunction to the pathogenesis of diabetes, laying a foundation for the basic pathological mechanisms of such disease. Different from the downstream mechanisms of two major pathogeneses, such pathogenic mechanisms can be both attributed to either genetic predisposition or environmental interferences (<xref ref-type="bibr" rid="B2">Andersen et al., 2016</xref>; <xref ref-type="bibr" rid="B75">Stancakova and Laakso, 2016</xref>). They would be involved in the progressive dysfunction of pancreatic islet alpha and beta cells, so that, the pancreatic islet cells actually have specific roles in the initiation and progression of type 2 DM.</p>
<p>Traditionally, the studies on the pathogenic characteristics and contribution of pancreatic islet cells mainly focused on the abnormal biochemical reactions and physiological processes of such cell types in type 2 DM (<xref ref-type="bibr" rid="B5">Borg et al., 2001</xref>; <xref ref-type="bibr" rid="B23">Donath et al., 2003</xref>; <xref ref-type="bibr" rid="B69">Prentki and Nolan, 2006</xref>; <xref ref-type="bibr" rid="B90">Westermark and Westermark, 2008</xref>). According to these studies, there are four major pathogenic characteristics of pancreatic islet cells, including increased islet glucose metabolism (<xref ref-type="bibr" rid="B29">Forst et al., 2014</xref>), abnormal lipid signaling (<xref ref-type="bibr" rid="B10">Chakraborty et al., 2014</xref>), abnormal GLP-1 secretion (<xref ref-type="bibr" rid="B82">Trujillo and Nuffer, 2014</xref>), and compensatory feedback stimulation on parasympathetic and sympathetic neurons (<xref ref-type="bibr" rid="B80">Thorens, 2014</xref>). With the development of high-throughput sequencing technologies, more and more fundamental pathogenesis of type 2 DM at genetic and transcriptomic levels has been revealed. Apart from such transcription factors, genes regulating optimal glucose-responsive insulin secretion, like <italic>IAPP</italic>, <italic>GLUT2</italic>, <italic>GAD65</italic>, and <italic>IA-2</italic>, have also been identified to participate in T2D-associated pathogenesis (<xref ref-type="bibr" rid="B15">Clocquet et al., 2000</xref>). Therefore, the abnormal gene functions of pancreatic islet cells may be one of the major pathogenic factors for type 2 DM. However, as we all know, the cellular components of pancreatic islet cells are quite complicated involving various cell subtypes. Meanwhile, traditional studies all focused on the biological features (either at the cellular level or genetic level) of cell population, no matter pathogenic or not for individual cells. Therefore, these conventional studies may ignore some potential pathogenic factors and mistake non-pathogenic features due to normal or irrelevant cells&#x2019; interferences.</p>
<p>Multiple previous studies have focused on single-cell analyses on pancreatic islets under physical or pathological conditions. With the development of single-cell techniques, the studies on pancreatic islets under either pathological or normal conditions have been extended to the single-cell level. Early in 2016, <xref ref-type="bibr" rid="B72">Segerstolpe et al. (2016)</xref> have identified some typical biomarkers to distinguish pancreatic islets under healthy and diabetic conditions. However, as limitations of this study, the authors only applied differential expression analyses and the t-SNE method to identify some potential biomarkers to reveal the heterogeneity (<xref ref-type="bibr" rid="B72">Segerstolpe et al., 2016</xref>). Apart from this study, further in 2017, another study extended to identify the specific biomarkers for T2D, confirming that genes are differentially expressed at the transcriptomics level not only between patients and controls but also among different cell types (<xref ref-type="bibr" rid="B46">Lawlor et al., 2017</xref>). In 2018, another single-cell gene expression analysis on T2D also tried to identify specific biomarkers for the prediction of cellular states of beta-cells, either healthy or T2D beta-cells (<xref ref-type="bibr" rid="B57">Ma and Zheng, 2018</xref>). The shortcomings of these two studies turn out to be a lack of quantitative standards establishment, making it still quite hard to predict T2D using single-cell transcriptomics data.</p>
<p>To overcome the limitations of previous studies mentioned earlier, in this study, for the first time, we used the single-cell sequencing results from one previous study (<xref ref-type="bibr" rid="B92">Xin et al., 2016</xref>) and tried to extend their analyses at two levels: 1) using multiple machine learning algorithms for deep analysis; 2) taking the pancreatic islets as a whole and did not distinguish different cell subtypes. We extended the classification and prediction of cellular states from just beta cells to multiple cell types, including human pancreatic alpha, beta, delta, and PP cells. Also, different from previous studies, we did not just focus on the pathogenic effects of T2D on beta cells but tried to reveal the general comprehensive pathogenic effects on all the cells from the pancreatic islets. Although most of the previous studies identified that pancreatic islet B cells are the major participants in the pathogenesis of T2D, other cells, including alpha, delta, and PP cells, are also either shown to be correlated with the pathogenesis of T2D or may act as potential biomarkers for T2D due to their typical changes during the pathogenesis. Therefore, it is not only innovative but effective to reveal the comprehensive effects of T2D on pancreatic islets and identify more valuable biomarkers for such disease.</p>
<p>All in all, to remove the interferences caused by conventional bulk sequencing and analysis, we have tried to identify potential pathogenic factors of T2D from the transcriptomic profiling covering multiple cell subtypes at the single-cell level. Relied on single-cell RNA sequencing techniques and related public datasets (<xref ref-type="bibr" rid="B92">Xin et al., 2016</xref>), we investigated such datasets with several powerful machine learning algorithms. Different from previous studies, focusing on identifying biomarkers for distinguishing a tissue under normal or pathological conditions but not an entire tissue, which makes hard to detect biomarkers from a single-cell subtype in clinical applications, this study tried to identify the common transcriptomics characteristics across different cell types at the single-cell level for T2D. Biomarkers identified in this study may not be affected by the cell composition of the islet tissue that may vary among different individuals. In addition, our results revealed novel potential pathogenic mechanisms induced by newly recognized genes in a rule manner, which are always neglected by traditional bulk sequencing techniques. On the one hand, these results deepen our understanding on the etiology and pathogenesis of T2D. On the other hand, such identified new biomarkers can be potential candidates for further clinical application in the diagnosis of T2D using the transcriptomics information of the entire tissue, with no further cell separation and preprocessing required.</p>
</sec>
<sec id="s2">
<title>2 Materials and Methods</title>
<p>In this study, we first used a feature selection method to analyze a RNA sequencing dataset of T2D for ranking the important genes associated with T2D, and these genes were further optimized for diabetes using incremental feature selection (IFS) (<xref ref-type="bibr" rid="B53">Liu and Setiono, 1998</xref>) with some supervised classifiers. In the end, we applied the rule learning method to generate interpretable classification rules for T2D. The whole process is illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Workflow for key gene identification of type 2 diabetes. The MCFS method was used to evaluate the importance of all features (genes). On the one hand, the IFS method with SVM/RF/KNN was applied on the feature list yielded by the MCFS method to extract optimal T2D-associated genes and optimal classifiers. On the other hand, the informative features yielded by the MCFS method were fed into the Johnson reducer and RIPPER algorithms to construct optimal T2D-associated rules.</p>
</caption>
<graphic xlink:href="fbioe-10-890901-g001.tif"/>
</fig>
<sec id="s2-1">
<title>2.1 Datasets</title>
<p>We downloaded the RNA sequencing data of 1,600 human pancreatic islet cells from the GEO (Transcript Expression Omnibus) database under the accession number of GSE81608 at <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE81608">https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc&#x3d;GSE81608</ext-link> (<xref ref-type="bibr" rid="B92">Xin et al., 2016</xref>). There were 949 pancreatic islet cells from six T2D patients and 651 pancreatic islet cells from 12 non-diabetic donors. Within the 949 pancreatic islet cells from T2D patients, there were 569 alpha, 296 beta, 30 delta, and 54 PP cells. Within in the 651 pancreatic islet cells from non-diabetic donors, there were 377 alpha, 207 beta, 28 delta, and 39 PP cells. The expression levels of 39,851 genes were quantified as RPKM (Reads Per Kilo bases per Million reads). The processed gene expression profiles of these cells downloaded from <ext-link ext-link-type="uri" xlink:href="https://ftp.ncbi.nlm.nih.gov/geo/series/GSE81nnn/GSE81608/suppl/GSE81608_human_islets_rpkm.txt.gz">https://ftp.ncbi.nlm.nih.gov/geo/series/GSE81nnn/GSE81608/suppl/GSE81608_human_islets_rpkm.txt.gz</ext-link> were used. Despite islet cells containing different cells, this work expects to identify the common gene signatures for T2D across multiple cell types.</p>
</sec>
<sec id="s2-2">
<title>2.2 Feature Selection</title>
<p>In this study, we first used the Monte Carlo feature selection (MCFS) (<xref ref-type="bibr" rid="B24">Draminski et al., 2008</xref>) to evaluate the importance of all genes, obtaining a feature list and some informative genes expressed in diabetes. For the feature list, it was fed into the IFS (<xref ref-type="bibr" rid="B53">Liu and Setiono, 1998</xref>) with one classification algorithm to extract optimal genes that had a strong discriminate ability between diabetes and non-diabetes samples and construct an efficient classifier. On the other hand, repeated incremental pruning to produce error reduction (RIPPER) was employed to determine interpretable rules on gene expression patterns with informative features.</p>
<sec id="s2-2-1">
<title>2.2.1 Monte Carlo Feature Selection</title>
<p>The investigated data contained 1,600 samples, each of which was represented by expression levels on lots of genes. Accordingly, the data can be summarized as a matrix with low row numbers and high column numbers. MCFS is deemed to be a powerful feature selection method to deal with such data. Thus, it was employed in this study. MCFS is a multivariate feature selection method based on bootstrap samples and decision trees, which focuses on selecting discriminate features for classification with robustness. In this feature selection algorithm, it generates multiple bootstrap sets, and on each bootstrap set, multiple decision trees are grown on smaller feature subsets randomly selected from original features. Then, the involvement of each feature in the decision trees shows a relative importance (RI) score, which indicates the overall number of splits involving this feature in all nodes of all constructed trees. The MCFS program was downloaded from <ext-link ext-link-type="uri" xlink:href="http://www.ipipan.eu/staff/m.draminski/mcfs.html">http://www.ipipan.eu/staff/m.draminski/mcfs.html</ext-link>. For convenience, default parameters were adopted.</p>
<p>The MCFS program was executed on the aforementioned RNA sequencing data. According to the output of the MCFS program, we can obtain the RI values of all features. Accordingly, features can be ranked in a list with the decreasing order of their RI values. Furthermore, it also provides the informative features, which are generated by a permutation test on class labels and one-sided Student&#x2019;s t-test. These features are always the top-ranking features in the list. We would adopt these features to construct classification rules <italic>via</italic> RIPPER.</p>
</sec>
<sec id="s2-2-2">
<title>2.2.2 Incremental Feature Selection</title>
<p>In this study, we performed IFS on the MCFS-generated feature list, denoted by <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> (<italic>N</italic> was the total number of features), to screen out a set of optimal features, which can accurately discriminate between diabetes and non-diabetes samples. Based on such list, we generated a series of feature subsets with step 5. Suppose there are <italic>m</italic> feature subsets <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where the <italic>i</italic>th feature subset contains top <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mn>5</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> features, that is, <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Then, for a given classification algorithm, we built one classifier on samples represented by features from each feature subset and yielded the 10-fold cross-validation performance for evaluating this classifier. After all constructed feature subsets had been tested, the feature subset, on which the classifier provided the best performance, can be obtained. Such a feature subset was called the optimal feature subset for this classification algorithm, and the features inside were named as the optimal features. Furthermore, the classifier with the best performance was termed as the optimal classifier.</p>
</sec>
</sec>
<sec id="s2-3">
<title>2.3 Classification Algorithm</title>
<p>For the IFS method, one classification algorithm was necessary. In this study, we tried three classic classification algorithms: 1) support vector machine (SVM) (<xref ref-type="bibr" rid="B18">Cortes and Vapnik, 1995</xref>), 2) K-nearest neighbor (KNN) (<xref ref-type="bibr" rid="B19">Cover and Hart, 1967</xref>), and 3) random forest (RF) (<xref ref-type="bibr" rid="B6">Breiman, 2001</xref>). Their brief descriptions were as follows.</p>
<sec id="s2-3-1">
<title>2.3.1 Support Vector Machine</title>
<p>The SVM is a supervised learning model based on statistical learning theory and is widely used in many biological problems (<xref ref-type="bibr" rid="B65">Pan and Shen, 2009</xref>; <xref ref-type="bibr" rid="B63">Mirza et al., 2015</xref>; <xref ref-type="bibr" rid="B12">Chen et al., 2017</xref>; <xref ref-type="bibr" rid="B37">Jia et al., 2018</xref>; <xref ref-type="bibr" rid="B89">Wei et al., 2018</xref>; <xref ref-type="bibr" rid="B101">Zhou et al., 2022a</xref>; <xref ref-type="bibr" rid="B100">Zhou et al., 2020b</xref>; <xref ref-type="bibr" rid="B52">Liu et al., 2021</xref>; <xref ref-type="bibr" rid="B88">Wang et al., 2021</xref>; <xref ref-type="bibr" rid="B103">Zhu et al., 2021</xref>; <xref ref-type="bibr" rid="B49">Li X. et al., 2022</xref>; <xref ref-type="bibr" rid="B91">Wu and Chen, 2022</xref>). Given a set of training samples, each training sample is assigned to positives or negatives. The SVM training algorithm fits a hyperplane that has the maximum margin between positives and negatives, where the generalization error becomes smaller when the margin is larger. The SVM generally is good at handling non-linear data, since it can first map the data in non-linear space to high-dimensional linear space by the kernel function and then fit a linear model in the high-dimensional space.</p>
</sec>
<sec id="s2-3-2">
<title>2.3.2 K-Nearest Neighbor</title>
<p>KNN is one of the simplest schemes for classifying samples. However, in many cases, it still can yield good performance. Given a training dataset, KNN directly uses samples in it to make prediction for any query sample, that is, KNN does not contain a learning procedure. Generally, it finds <italic>k</italic> training samples, which have the nearest distances (e.g., Euclidean distance) to the query sample. By counting the classes of these <italic>k</italic> training samples, the class with most votes is assigned to the query sample.</p>
</sec>
<sec id="s2-3-3">
<title>2.3.3 Random Forest</title>
<p>RF is another classic classification algorithm. In fact, it is an integrated algorithm, consisting of several decision trees. For constructing each decision tree, it randomly picks up samples from the training dataset, with replacement, to constitute the basic dataset. The tree is extended at each node by selecting an optimal split on one feature among the randomly selected features. RF integrates the predictions of all decision trees with majority voting. RF is deemed as a powerful classification algorithm and has wide applications in tackling many biological problems (<xref ref-type="bibr" rid="B43">Kandaswamy et al., 2011</xref>; <xref ref-type="bibr" rid="B9">Casanova et al., 2014</xref>; <xref ref-type="bibr" rid="B59">Marques et al., 2016</xref>; <xref ref-type="bibr" rid="B38">Jia et al., 2020</xref>; <xref ref-type="bibr" rid="B51">Liang et al., 2020</xref>; <xref ref-type="bibr" rid="B96">Zhang et al., 2021b</xref>; <xref ref-type="bibr" rid="B13">Chen et al., 2021</xref>; <xref ref-type="bibr" rid="B64">Onesime et al., 2021</xref>; <xref ref-type="bibr" rid="B11">Chen et al., 2022</xref>; <xref ref-type="bibr" rid="B21">Ding et al., 2022</xref>; <xref ref-type="bibr" rid="B94">Yang and Chen, 2022</xref>).</p>
<p>To quickly implement the aforementioned three classification algorithms, we employed the corresponding packages in scikit-learn (<ext-link ext-link-type="uri" xlink:href="https://scikit-learn.org/stable/">https://scikit-learn.org/stable/</ext-link>). Some main parameters were tuned for extracting optimal parameters.</p>
</sec>
</sec>
<sec id="s2-4">
<title>2.4 Johnson Reducer and Repeated Incremental Pruning to Produce Error Reduction Algorithms</title>
<p>Classification algorithms mentioned in <xref ref-type="sec" rid="s2-3">Section 2.3</xref> are powerful to construct efficient classifiers. However, we cannot understand their principles because they are black-box algorithms. In this case, few clues for uncovering essential differences between T2D patients and non-diabetic donors can be obtained. In view of this, we further adopted rule learning algorithms to investigate the RNA sequencing data. Although it is generally weaker than the aforementioned algorithms, it can provide rules that clearly indicate special expression patterns on T2D patients, thereby improving our understanding on T2D. The procedures were described in the following sections.</p>
<p>As mentioned in Section 3.2.1, the MCFS method can select some informative features. These features are quite essential to describe the characteristics of the dataset. Here, we used these features to construct classification rules <italic>via</italic> RIPPER algorithm (<xref ref-type="bibr" rid="B16">Cohen, 1995</xref>). Before that, the Johnson reducer algorithm (<xref ref-type="bibr" rid="B40">Johnson, 1974</xref>) was applied on the informative features to select the most important features, which had the similar classification ability compared to the original informative features. The selected features were fed into the RIPPER algorithm. RIPPER, proposed by <xref ref-type="bibr" rid="B16">Cohen (1995)</xref>, is a rule learning algorithm which is capable of handling large noisy datasets effectively. RIPPER is the improved version of IREP (<xref ref-type="bibr" rid="B39">Johannes and Widmer, 1994</xref>) which combines both the separate-and-conquer technique used first in the relational learner FOIL (<xref ref-type="bibr" rid="B71">Quinlan, 1990</xref>) and the reduced error pruning strategy proposed by <xref ref-type="bibr" rid="B8">Brunk and Pazzani (1991)</xref>. In RIPPER, the training set is first split into growing and pruning sets. Then, repeat the rule grow phase and rule prune phase until no positive samples are left in the growing set, or the description length (DL) is 64 bits greater than the smallest DL found so far, or the error rate is greater than 50%. In the rule grow phase, one rule is generated by greedily adding conditions to the rule that achieves the highest FOIL&#x2019;s information gain. In the rule prune phase, the rule is pruned using reduced error pruning. Finally, global optimization strategy is applied to further prune the rule set. The aforementioned procedures for constructing rules are also implemented in the MCFS program, that is, the set of rules is one output of the MCFS program.</p>
</sec>
<sec id="s2-5">
<title>2.5 Performance Measurement</title>
<p>In this study, we used six measurements to evaluate the performance of all classifiers under 10-fold cross-validation (<xref ref-type="bibr" rid="B45">Kohavi, 1995</xref>; <xref ref-type="bibr" rid="B50">Li Z. et al., 2022</xref>; <xref ref-type="bibr" rid="B21">Ding et al., 2022</xref>; <xref ref-type="bibr" rid="B76">Tang and Chen, 2022</xref>), including sensitivity (SN) (same as recall), specificity (SP), accuracy (ACC), Matthew correlation coefficient (MCC), precision, and F1-measure (<xref ref-type="bibr" rid="B61">Matthews, 1975</xref>; <xref ref-type="bibr" rid="B99">Zhao et al., 2018</xref>; <xref ref-type="bibr" rid="B98">Zhao et al., 2019</xref>; <xref ref-type="bibr" rid="B38">Jia et al., 2020</xref>; <xref ref-type="bibr" rid="B51">Liang et al., 2020</xref>; <xref ref-type="bibr" rid="B95">Zhang et al., 2021a</xref>; <xref ref-type="bibr" rid="B97">Zhang et al., 2021c</xref>; <xref ref-type="bibr" rid="B66">Pan et al., 2021</xref>). Their formulations are written as follows:<disp-formula id="e1">
<mml:math id="m5">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
<disp-formula id="e2">
<mml:math id="m6">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
<disp-formula id="e3">
<mml:math id="m7">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
<disp-formula id="e4">
<mml:math id="m8">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m9">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
<disp-formula id="e6">
<mml:math id="m10">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where TP represents the number of truly positive samples, FP represents the number of false-positive samples, TN represents the number of truly negative samples, and FN represents the number of false-negative samples. Among these six measurements, we selected F1-measure as the key one, whereas others were provided for reference.</p>
</sec>
<sec id="s2-6">
<title>2.6 Gene Ontology Enrichment Analysis on Optimal Genes</title>
<p>Some rules can be extracted <italic>via</italic> the Johnson reducer and RIPPER algorithms, which involved several features (genes), called rule genes, in the following text. We performed Gene Ontology (GO) enrichment analysis using R package <italic>topGO</italic> (<ext-link ext-link-type="uri" xlink:href="http://bioconductor.org/packages/release/bioc/html/topGO.html">http://bioconductor.org/packages/release/bioc/html/topGO.html</ext-link>, v.2.24.0) on these rule genes. The genes of interest were set as rule genes, and the gene background was set as all the available genes. The <italic>p</italic>-value threshold was set at 0.001.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Results</title>
<p>T2D is one type of DM and makes up most DM cases. In this study, we investigated potential pathogenic factors of T2D at the single-cell level by analyzing a single-cell RNA sequencing dataset. Such dataset contained 1,600 single cells, including 949 cells from T2D patients and 651 cells from normal controls. It was analyzed by some powerful machine learning algorithms, including MCFS (<xref ref-type="bibr" rid="B24">Draminski et al., 2008</xref>), SVM (<xref ref-type="bibr" rid="B18">Cortes and Vapnik, 1995</xref>), KNN (<xref ref-type="bibr" rid="B19">Cover and Hart, 1967</xref>), RF (<xref ref-type="bibr" rid="B6">Breiman, 2001</xref>), and RIPPER (<xref ref-type="bibr" rid="B16">Cohen, 1995</xref>). The entire procedure is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. On one hand, we obtained some T2D-associated genes, which can be novel biomarkers of T2D. On the other hand, some interesting rules were constructed, which can uncover different expression patterns in T2D patients and normal controls. This section gives the detailed results of these procedures.</p>
</sec>
<sec id="s4">
<title>3.1 Results of the Monte Carlo Feature Selection Method</title>
<p>The MCFS method was directly applied to the RNA sequencing data to analyze the importance of all features (genes). Each gene was assigned a RI score. A total of 26,978 genes were assigned RI scores larger than zero. These genes and their RI scores are provided in <xref ref-type="sec" rid="s12">Supplementary Table S1</xref>. Because the RI scores of the rest genes were zero, meaning their associations for the identification of T2D samples were very weak, they were discarded. A feature list was generated by sorting the remaining 26,978 genes in the decreasing order of their RI scores, which is also provided in <xref ref-type="sec" rid="s12">Supplementary Table S1</xref>.</p>
<p>In addition to the feature list, the MCFS method can output some informative features. For investigating RNA sequencing data, 235 informative features were extracted by the MCFS method, which were the top 235 genes listed in <xref ref-type="sec" rid="s12">Supplementary Table S1</xref>.</p>
</sec>
<sec id="s5">
<title>3.2 Results of the Incremental Feature Selection Method</title>
<p>To further extract optimal features, the IFS method combined with one classification algorithm was employed. Here, we tried three classification algorithms: SVM, KNN, and RF. Some main parameters of each algorithm were tuned. In detail, for SVM, four kernels were attempted, including linear, polynomial, RBF, and sigmoid kernels. The parameter <italic>k</italic> for KNN was set to 1, 5, and 10, and the parameter, number of decision trees (I), for RF was set to 20, 40, 60, 80, and 100. Because the feature list contained a huge number of features, we only considered the top 5,000 features in this study to save time. Several feature subsets were constructed using step 5.</p>
<p>When the classification algorithm was KNN, several KNN classifiers with a certain parameter <italic>k</italic> were constructed on all feature subsets. All these classifiers were evaluated by 10-fold cross-validation. The obtained six measurements are listed in <xref ref-type="sec" rid="s12">Supplementary Table S2</xref>. For an easy observation, we plot a curve for KNN with a certain parameter <italic>k</italic>, as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, in which the F1-measure was set to the y-axis and the number of features was set to the x-axis. We can see that when <italic>k</italic> &#x3d; 1, 5, and 10, the highest F1-measure was 0.885, 0.886, and 0.880, respectively. Thus, the KNN classifier with <italic>k</italic> &#x3d; 5 provided the best performance. Such classifier used the top 665 features (genes) in the feature list. These features were the optimal features for KNN. The other five measurements are illustrated in <xref ref-type="fig" rid="F3">Figure 3</xref>. Except MCC, all measurements exceeded 0.8, implying the good performance of such KNN classifiers. Furthermore, it can be observed from <xref ref-type="fig" rid="F2">Figure 2</xref> that the IFS curves of KNN with different parameters <italic>k</italic> had a common feature. The curve followed a sharp decreasing trend before about top 600 features were used. The top features in the list were highly related to class labels (T2D patients and non-diabetic patients in this study), and a simple scheme based on these features, as KNN used, can correctly predict the cells of T2D patients and non-diabetic patients. However, when features with low ranks, which had low relevance to class labels, were added, KNN cannot exclude interference information contained in these features as KNN has no training procedures, inducing the quick descent of its performance. In this study, the set containing about top 600 features was a pivotal point for KNN. After this point, the performance of KNN followed a sharp decreasing trend.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Performance of KNN integrated in IFS using different numbers of features. The y-axis is F1-measure, and the x-axis is the number of participated features. <italic>k</italic> is the parameter of KNN, indicating the number of nearest neighbors that are used to make prediction. KNN can yield the best F1-measure of 0.886 when <italic>k</italic> &#x3d; 5 and the top 665 features are used.</p>
</caption>
<graphic xlink:href="fbioe-10-890901-g002.tif"/>
</fig>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Bar chart to show five measurements of three optimal classifiers based on different classification algorithms.</p>
</caption>
<graphic xlink:href="fbioe-10-890901-g003.tif"/>
</fig>
<p>We also tried another classification algorithm, RF. The same IFS procedure was conducted on this algorithm. The obtained measurements are listed in <xref ref-type="sec" rid="s12">Supplementary Table S3</xref>. Likewise, a curve was plotted for RF with a certain number of decision trees, as shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. It can be observed that when <italic>I</italic> &#x3d; 20, 40, 60, 80, and 100, the highest F1-measure was 0.903, 0.904, 0.905, 0.904, and 0.907. The RF classifier with <italic>I</italic> &#x3d; 100 provided the highest performance. The top 305 features in the list were adopted in this classifier and were termed as optimal features for RF. Evidently, such an RF classifier was superior to the best KNN classifiers mentioned earlier. Furthermore, the other five measurements of this RF classifier are shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. All measurements were higher than 0.8, suggesting the better performance of this classifier than the aforementioned KNN classifier.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Performance of RF integrated in IFS using different numbers of features. The y-axis is F1-measure, and the x-axis is the number of participated features. <italic>I</italic> is the parameter of RF, indicating the number of decision trees. RF can yield the best F1-measure of 0.907 when <italic>I</italic> &#x3d; 100 and the top 305 features are used.</p>
</caption>
<graphic xlink:href="fbioe-10-890901-g004.tif"/>
</fig>
<p>Finally, we conducted the same IFS procedure for SVM. The measurements are listed in <xref ref-type="sec" rid="s12">Supplementary Table S4</xref>. Similarly, for each SVM with a certain kernel, a curve was plotted, as shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. With four different kernels, SVM yielded the highest F1-measure of 0.936, 0.894, 0.909, and 0.687. The SVM with a linear kernel provided the best performance. Also, such performances were based on the top 745 features in the list. Accordingly, they were called the optimal features for SVM. Furthermore, the performance of this SVM classifier was better than that of the aforementioned KNN and RF classifiers. The same conclusion can be obtained according to the five measurements of such SVM classifiers, illustrated in <xref ref-type="fig" rid="F3">Figure 3</xref>. Due to the best performance of the SVM with its optimal 745 genes, these genes were quite important for investigating T2D at the single-cell level. The top seven genes are listed in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Performance of SVM integrated in IFS using different numbers of features. The y-axis is F1-measure, and the x-axis is the number of participated features. SVM can yield the best F1-measure of 0.936 when the kernel is a linear function and the top 745 features are used.</p>
</caption>
<graphic xlink:href="fbioe-10-890901-g005.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Top seven genes among the optimal genes for SVM.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Rank</th>
<th align="center">Gene ID</th>
<th align="center">Gene symbol</th>
<th align="center">RI</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="center">100128906</td>
<td align="left">LOC100128906</td>
<td align="char" char=".">0.1140</td>
</tr>
<tr>
<td align="left">2</td>
<td align="center">100873254</td>
<td align="left">MTND4P24</td>
<td align="char" char=".">0.1046</td>
</tr>
<tr>
<td align="left">3</td>
<td align="center">100271063</td>
<td align="left">RPS14P1</td>
<td align="char" char=".">0.1032</td>
</tr>
<tr>
<td align="left">4</td>
<td align="center">100652939</td>
<td align="left">MTND2P28</td>
<td align="char" char=".">0.0979</td>
</tr>
<tr>
<td align="left">5</td>
<td align="center">285045</td>
<td align="left">LINC00486</td>
<td align="char" char=".">0.0959</td>
</tr>
<tr>
<td align="left">6</td>
<td align="center">729898</td>
<td align="left">ZBTB8OSP2</td>
<td align="char" char=".">0.0954</td>
</tr>
<tr>
<td align="left">7</td>
<td align="center">391524</td>
<td align="left">THRAP3P1</td>
<td align="char" char=".">0.0862</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>With the earlier IFS results with different classification algorithms using various parameters, the SVM with linear kernel and top 745 features provided the best performance of F1-measure 0.936. The ACC and MCC of such classifier were 0.925 and 0.846, respectively. Other three measurements, SN, SP, and precision were 0.925, 0.925, and 0.947, respectively. These measurements suggested the excellent performance of this classifier, and it can be an efficient tool to identify cells of T2D patients.</p>
<sec id="s5-1">
<title>3.3 Classification Rules</title>
<p>Although we can construct efficient classifiers to identify cells of T2D patients through three classification algorithms, these classifiers were absolute black-box algorithms, which prevented us from uncovering the essential differences between cells of T2D patients and non-diabetic donors. As mentioned in <xref ref-type="sec" rid="s2-4">Section 2.4</xref>, rule learning algorithms were employed.</p>
<p>According to the output of the MCFS program, 235 features were selected as informative features. To test the utility of the classification rules yielded by Johnson reducer and RIPPER algorithms, we performed the 10-fold cross-validations three times, obtaining the F1-measure of 0.910, which was lower than that of the optimal SVM classifier but higher than that of the optimal KNN and RF classifiers. The SN was 0.898, SP was 0.891, ACC was 0.895, MCC was 0.784, and precision was 0.923. Although such performance was lower than that of the optimal SVM classifier, the RIPPER algorithm can construct a group of rules, which made the classification procedure completely open and provided more insights. Thus, the Johnson reducer and RIPPER algorithms were applied to all samples, producing nine different classification rules, as listed in <xref ref-type="table" rid="T2">Table 2</xref>. These rules are able to accurately screen patients with T2D from non-diabetic population. Although these rules were mainly for non-diabetes, based on the aforementioned evaluation results (SP &#x3d; 0.891), it was believed that these rules were statistically shown to cover almost all possible non-diabetes samples. Thus, investigation on these rules can also figure out the characteristics of T2D patients in an opposite aspect.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Nine classification rules for diabetes generated by the RIPPER algorithm.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Rule</th>
<th align="center">Criteria</th>
<th align="center">Patient</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="4" align="left">Rule 1</td>
<td align="left">Gene Id 100128906 (LOC100128906) &#x2265; 2.7722</td>
<td rowspan="4" align="left">Non-diabetes</td>
</tr>
<tr>
<td align="left">Gene Id 326307 (RPL3P4) &#x2264; 15.2306</td>
</tr>
<tr>
<td align="left">Gene Id 8781 (PSPHP1) &#x2265; 0.0965</td>
</tr>
<tr>
<td align="left">Gene Id 100873065 (PTCHD1-AS) &#x2264; 0.1036</td>
</tr>
<tr>
<td rowspan="4" align="left">Rule 2</td>
<td align="left">Gene Id 100462954 (MICOS10P3) &#x2265; 2.0984</td>
<td rowspan="4" align="left">Non-diabetes</td>
</tr>
<tr>
<td align="left">Gene Id 1487 (CTBP1) &#x2264; 17.3460</td>
</tr>
<tr>
<td align="left">Gene Id 326307 (RPL3P4) &#x2264; 6.2868</td>
</tr>
<tr>
<td align="left">Gene Id 100873254 (MTND4P24) &#x2265; 3.0364</td>
</tr>
<tr>
<td rowspan="5" align="left">Rule 3</td>
<td align="left">Gene Id 100128906 (LOC100128906) &#x2265; 49.6340</td>
<td rowspan="4" align="left">Non-diabetes</td>
</tr>
<tr>
<td align="left">Gene Id 143244 (EIF5AL1) &#x2265; 1.0987</td>
</tr>
<tr>
<td align="left">Gene Id 486 (FXYD2) &#x2264; 152.8666</td>
</tr>
<tr>
<td align="left">Gene Id 326307 (RPL3P4) &#x2264; 11.3894</td>
</tr>
<tr>
<td align="left">Gene Id 6126 (RPL9P7) &#x2264; 103.5050</td>
</tr>
<tr>
<td rowspan="6" align="left">Rule 4</td>
<td align="left">Gene Id 100128906 (LOC100128906) &#x2265; 3.0256</td>
<td rowspan="6" align="center">Non-diabetes</td>
</tr>
<tr>
<td align="left">Gene Id 326307 (RPL3P4) &#x2264; 22.4381</td>
</tr>
<tr>
<td align="left">Gene Id 100128906 (LOC100128906) &#x2265; 225.8732</td>
</tr>
<tr>
<td align="left">Gene Id 388147 (RPL9P9) &#x2264; 50.3934</td>
</tr>
<tr>
<td align="left">Gene Id 100271332 (RPL36AP21) &#x2265; 1.7952</td>
</tr>
<tr>
<td align="left">Gene Id 222901 (RPL23P8) &#x2264; 2.6067</td>
</tr>
<tr>
<td rowspan="3" align="left">Rule 5</td>
<td align="left">Gene Id 100652939 (MTND2P28) &#x2265; 450.8125</td>
<td rowspan="3" align="center">Non-diabetes</td>
</tr>
<tr>
<td align="left">Gene Id 4574 (MT-TS1) &#x2264; 445.4115</td>
</tr>
<tr>
<td align="left">Gene Id 1487 (CTBP1) &#x2264; 37.6438</td>
</tr>
<tr>
<td rowspan="5" align="left">Rule 6</td>
<td align="left">Gene Id 285045 (LINC00486) &#x2264; 0.0930</td>
<td rowspan="5" align="center">Non-diabetes</td>
</tr>
<tr>
<td align="left">Gene Id 100873254 (MTND4P24) &#x2264; 28.2479</td>
</tr>
<tr>
<td align="left">Gene Id 653147 (RPL26P30) &#x2265; 5.1856</td>
</tr>
<tr>
<td align="left">Gene Id 285900 (RPL6P20) &#x2265; 0.4760</td>
</tr>
<tr>
<td align="left">Gene Id 643932 (RPS3AP20) &#x2265; 5.5063</td>
</tr>
<tr>
<td rowspan="3" align="left">Rule 7</td>
<td align="left">Gene Id 100128906 (LOC100128906) &#x2265; 3.0256</td>
<td rowspan="3" align="center">Non-diabetes</td>
</tr>
<tr>
<td align="left">Gene Id 440737 (RPL35P1) &#x2265; 4.118</td>
</tr>
<tr>
<td align="left">Gene Id 100271003 (RPL34P18) &#x2265; 9.0166</td>
</tr>
<tr>
<td rowspan="4" align="left">Rule 8</td>
<td align="left">Gene Id 100128906 (LOC100128906) &#x2265; 109.2232</td>
<td rowspan="4" align="center">Non-diabetes</td>
</tr>
<tr>
<td align="left">Gene Id 100873254 (MTND4P24) &#x2264; 28.3353</td>
</tr>
<tr>
<td align="left">Gene Id 644972 (RPS3AP26) &#x2265; 53.5552</td>
</tr>
<tr>
<td align="left">Gene Id 644604 (EEF1A1P12) &#x2264; 7.9556</td>
</tr>
<tr>
<td align="left">Rule 9</td>
<td align="left">Others</td>
<td align="left">Diabetes</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s5-2">
<title>3.4 Comparison of Classifiers With Informative Features</title>
<p>The MCFS method can directly output some informative features. These features can capture essential information of the dataset. Here, as mentioned in <xref ref-type="sec" rid="s5-1">Section 3.3</xref>, 235 features were selected as informative features. We can directly use them to construct classifiers with different classification algorithms. These classifiers were also evaluated by 10-fold cross-validation. The main measurement, F1-measure, of these classifiers is listed in <xref ref-type="table" rid="T3">Table 3</xref>. For KNN, F1-measure varied between 0.839 and 0.849. The F1-measure of RF changed between 0.889 and 0.897. Also, SVM provided the F1-measure varying between 0.631 and 0.886. Compared with the F1-measure yielded by the optimal classifier based on the corresponding classification algorithm, the classifier using informative features generated a lower F1-measure, suggesting that such a classifier was inferior to the optimal classifier. The employment of the IFS method can help us construct more efficient classifiers.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Performance of classifiers using informative features yielded by the MCFS method.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Classification algorithm</th>
<th align="center">F1-measure</th>
<th align="center">Decrement<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">KNN (k &#x3d; 1)</td>
<td align="char" char=".">0.849</td>
<td align="char" char=".">0.036</td>
</tr>
<tr>
<td align="left">KNN (k &#x3d; 5)</td>
<td align="char" char=".">0.839</td>
<td align="char" char=".">0.047</td>
</tr>
<tr>
<td align="left">KNN (k &#x3d; 10)</td>
<td align="char" char=".">0.847</td>
<td align="char" char=".">0.033</td>
</tr>
<tr>
<td align="left">RF (I &#x3d; 20)</td>
<td align="char" char=".">0.889</td>
<td align="char" char=".">0.014</td>
</tr>
<tr>
<td align="left">RF (I &#x3d; 40)</td>
<td align="char" char=".">0.891</td>
<td align="char" char=".">0.013</td>
</tr>
<tr>
<td align="left">RF (I &#x3d; 60)</td>
<td align="char" char=".">0.894</td>
<td align="char" char=".">0.011</td>
</tr>
<tr>
<td align="left">RF (I &#x3d; 80)</td>
<td align="char" char=".">0.894</td>
<td align="char" char=".">0.010</td>
</tr>
<tr>
<td align="left">RF (I &#x3d; 100)</td>
<td align="char" char=".">0.897</td>
<td align="char" char=".">0.010</td>
</tr>
<tr>
<td align="left">SVM (linear kernel)</td>
<td align="char" char=".">0.882</td>
<td align="char" char=".">0.054</td>
</tr>
<tr>
<td align="left">SVM (polynomial kernel)</td>
<td align="char" char=".">0.859</td>
<td align="char" char=".">0.035</td>
</tr>
<tr>
<td align="left">SVM (RBF kernel)</td>
<td align="char" char=".">0.886</td>
<td align="char" char=".">0.023</td>
</tr>
<tr>
<td align="left">SVM (sigmoid kernel)</td>
<td align="char" char=".">0.631</td>
<td align="char" char=".">0.056</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn1">
<label>a</label>
<p>Numbers listed in this column indicate the difference of F1-measure yielded by the optimal classifier and that listed in the second column of this table.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec id="s6">
<title>4 Discussion</title>
<p>As we have described earlier, we applied our newly presented computational framework to the expression profiling data of more than 1,600 single pancreatic islet cells, constituting 949 diabetic cells and 651 non-diabetic cells (<xref ref-type="bibr" rid="B92">Xin et al., 2016</xref>). Based on such a bioinformatics approach, we not only screened out a group of discriminative genes that have distinctive expression patterns in diabetic or non-diabetic cells but also set up a series of quantitative rules for the recognition of pathogenic cells at the single-cell level. According to recent literature reports, several identified genes and established rules could be validated by existing experimental datasets, indicating the efficacy and accuracy of our analysis. The detailed functional analysis and evaluation of each predicted genes with high informative rank and their optimal rules in the expression pattern have been summarized and introduced in the following sections.</p>
<sec id="s6-1">
<title>4.1 Analysis of Optimal Type 2 Diabetes-Associated Genes</title>
<p>Because the optimal SVM classifier provided the best performance, which used top 745 features (genes), we focused on these 745 genes. However, it is impossible to analyze them one by one. Here, only top seven genes were analyzed, which are listed in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<p>The first predicted gene, <italic>WDR45-like pseudogene</italic> (100128906), is the pseudogene of gene <italic>WDR45</italic>. According to recent publications, it encodes a functional lncRNA associated with the regulation of <italic>WDR45</italic> (<xref ref-type="bibr" rid="B85">Tsuyuki et al., 2014</xref>; <xref ref-type="bibr" rid="B47">Lebovitz et al., 2015</xref>). <italic>WDR4</italic>5 has been functionally related to autophagy (<xref ref-type="bibr" rid="B47">Lebovitz et al., 2015</xref>). Considering that abnormal autophagy has been well known to contribute to the pathogenesis of T2D (<xref ref-type="bibr" rid="B48">Lee, 2014</xref>), it is reasonable to speculate that the expression level of <italic>WDR45</italic> and its upstream regulator (i.e., our predicted gene <italic>LOC100128906</italic>) may have quite different expressions in diabetic pancreatic islets cells compared to normal cells.</p>
<p>The next identified gene is <italic>MTND4P24</italic> (100873254), which is shown to have quite different expression levels in diabetic and normal tissues containing multiple cell subtypes. As an lncRNA-encoding pseudogene, the expression level of such a gene is able to reflect the regulatory ability of lncRNAs on its target gene, <italic>MT-ND4</italic> (<xref ref-type="bibr" rid="B81">Torrell et al., 2013</xref>; <xref ref-type="bibr" rid="B62">Mella et al., 2016</xref>). Recent publications also confirmed that the expression level of the target gene <italic>MT-ND4</italic> is functionally related to cellular insulin sensitivity in rat models (<xref ref-type="bibr" rid="B36">Houstek et al., 2012</xref>). Therefore, as one regulator of <italic>MT-ND</italic>4&#x2019;s expression, the expression pattern of <italic>MTND4P24</italic> may involve in the pathogenic insulin sensitivity decreasing in type 2 diabetic cells. Similarly, a homolog of <italic>MTND4P24</italic> and <italic>MTND2P28</italic> (100652939) has also been predicted to have different expression levels in multiple cell subtypes from pathogenic or normal pancreatic islets. Considering its similar regulatory mechanisms and the biological function of MTND2, it is also quite convincing to regard such a gene as a potential distinctive standard for diabetic and non-diabetic cells (<xref ref-type="bibr" rid="B60">Mathews et al., 2005</xref>).</p>
<p>The predicted gene, <italic>RPS14P1</italic> (100271063), is also a pseudogene, contributing to the regulation of ribosomal protein S14&#x2019;s expression (<xref ref-type="bibr" rid="B3">Aubert et al., 1992</xref>). Meanwhile, the function of ribosomal protein S14 is widely reported to participate in p53-dependent cell-cycle arrest by interacting with <italic>MDM2</italic> (<xref ref-type="bibr" rid="B102">Zhou et al., 2013</xref>), which is abnormally activated during the pathogenesis of diabetes (<xref ref-type="bibr" rid="B33">Golubnitschaja et al., 2006</xref>; <xref ref-type="bibr" rid="B32">Garufi et al., 2017</xref>). Thus, it is a reasonable assumption that ribosomal protein S14 together with <italic>RPS14P1</italic> has different expression levels in normal and diabetic cells.</p>
<p>Apart from such predicted pseudogenes, we also identified some functional lncRNAs that may have different expression patterns in normal and diabetic cells. LINC00486 (285045) is a predicted lncRNA that contributes to the distinction of normal and diabetic cells. According to recent publications, various functional lncRNAs (<xref ref-type="bibr" rid="B54">Liu et al., 2014</xref>; <xref ref-type="bibr" rid="B70">Pullen and Rutter, 2014</xref>), including LINC00486, have been confirmed to contribute to the initiation and progression of T2D (<xref ref-type="bibr" rid="B70">Pullen and Rutter, 2014</xref>).</p>
<p>The following predicted gene, named <italic>ZBTB8OSP2</italic> (729898), is a pseudogene and has been reported to contribute to anti-saccade response and eating disorders (<xref ref-type="bibr" rid="B17">Cornelis et al., 2014</xref>; <xref ref-type="bibr" rid="B7">Broer and van Duijn, 2015</xref>). As a transcriptional regulator for <italic>ZBTB8</italic>, such genes may indirectly contribute to a specific complication of T2D, the refractory diabetes insipidus, especially in adolescent male patients (<xref ref-type="bibr" rid="B74">Soto et al., 2014</xref>). Therefore, we can infer that such genes together with their downstream binding targets may have respective specific expression patterns in normal and diabetic cells.</p>
<p>The next predicted gene is <italic>THRAP3P1</italic> (391524), the pseudogene of <italic>THRAP3</italic>. The post-transcriptional regulatory target of <italic>THRAP3</italic> has been confirmed to dock on phosphoserine 273 of PPAR-gamma and further contribute to the pathogenic programming of diabetic genes, inducing insulin resistance (<xref ref-type="bibr" rid="B14">Choi et al., 2014</xref>). Therefore, to accomplish the regulatory role, such gene has a high expression level in normal cells compared to diabetic cells.</p>
</sec>
<sec id="s6-2">
<title>4.2 Specific Role of Pseudogenes in Type 2 Diabetes-Associated Genes</title>
<p>As we have discussed earlier, we identified multiple pseudogenes associated with T2D. Pseudogenes are nonfunctional segments with similar or reverse sequences of actual coding genes. The biological functions of pseudogenes are still unclear. It has only been speculated that pseudogenes participate in the post-transcriptional regulation <italic>via</italic> generating siRNAs, piRNAs, microRNAs, or other small RNAs (<xref ref-type="bibr" rid="B35">Guo et al., 2009</xref>). Although pseudogenes cannot generate protein products, the regulatory effects of such group of genes may still be significant under physical and pathological conditions (<xref ref-type="bibr" rid="B78">Tay et al., 2014</xref>). For transcriptomics analyses, especially for single-cell transcriptomics analyses, multiple pseudogenes have been identified as candidate biomarkers for different systematic diseases in studies specifically focusing on pseudogenes&#x2019; effects (<xref ref-type="bibr" rid="B42">Kalyana-Sundaram et al., 2012</xref>; <xref ref-type="bibr" rid="B68">Poliseno et al., 2015</xref>). For most previous studies, the pseudogenes were removed in the data preprocessing. Therefore, most previous studies have not identified a lot of pseudogenes as potential candidate biomarkers for diabetes. In our study, we did not filter out the pseudogenes and for the first time confirmed that pseudogenes with potential transcriptomic regulatory effects may further contribute to the regulation of specific diseases <italic>via</italic> regulating the biological functions of their respective recognized protein-coding genes.</p>
</sec>
<sec id="s6-3">
<title>4.3 Comparison With Previously Reported Type 2 Diabetes Biomarkers</title>
<p>Here, in this study from other perspective of view, we applied several machine learning algorithms to identify new potential biomarkers for T2D patients. Multiple previous publications have already identified a group of T2D biomarkers such as HbA1c, advanced glycation end-products (AGEs), and pigment epithelial-derived factor (PEDF) (<xref ref-type="bibr" rid="B56">Lyons and Basu, 2012</xref>). Also, for the publication from which we retrieved the single-cell sequencing data, unique biomarkers like <italic>LINC00486</italic>, <italic>ZNF445</italic>, and <italic>SYBU</italic> have also been identified for T2D (<xref ref-type="bibr" rid="B92">Xin et al., 2016</xref>). Compared with these prediction results, first, we identified a group of confirmed biomarkers like <italic>LINC00486</italic>, validating the efficacy and accuracy of our results. Second, we identified a group of new biomarkers like <italic>MTND4P24</italic> and <italic>THRAP3P1</italic>. Although such genes have been shown to be functionally correlated with T2D, previous studies have not identified such genes as potential biomarkers of T2D. There are two major advantages of our studies compared to previous studies, which may lead us to find novel biomarkers:<list list-type="simple">
<list-item>
<p>1) First, compared with previous studies, we used the single-cell level data with the gene expression profiling of different cells and not just an averaged comprehensive value for each patient. Therefore, we can identify potential biomarkers that are missing due to the averaging procedures.</p>
</list-item>
<list-item>
<p>2) Second, due to the sample size and cell type distribution, it is not proper to use feature selection and machine learning models for distinguishing each cell type independently. An integration of all the cell types may lead to a more reasonable result with effective biomarkers with clinical application potentials.</p>
</list-item>
</list>
</p>
<p>Such advantages explained why we identified novel protein biomarkers to distinguish T2D patients from normal controls. As we have discussed earlier, some identified biomarkers have been functionally correlated with T2D, implying that it is reasonable to regard such genes/transcripts as potential biomarkers for T2D.</p>
</sec>
<sec id="s6-4">
<title>4.4 Analysis of Optimal Type 2 Diabetes-Associated Rules</title>
<p>We also screened out a group of functional quantitative rules of the gene expression pattern to distinguish non-diabetic cells from diabetic ones with more interpretability, which are listed in <xref ref-type="table" rid="T2">Table 2</xref>. Many qualitative rules can be validated according to the gene expression level in existing databases and recent reports on gene expression trends, which support the efficacy and accuracy of the rules. The detailed analysis of each expression rule is widely discussed as follows:</p>
<p>The first rule (rule1) involved four genes including <italic>LOC100128906</italic> [(100128906), <italic>RPL3P4</italic> (326307), <italic>PSPHP1</italic> 8781], and <italic>PTCHD1</italic> (100873065). As mentioned earlier, gene <italic>LOC100128906</italic> has been reported to have quite different transcriptomics patterns between normal and diabetic cells, inhibiting autophagy (<xref ref-type="bibr" rid="B47">Lebovitz et al., 2015</xref>). As the antagonistic gene of diabetes-associated autophagy, such genes are reasonable to have high expression in normal cells compared to diabetic cells. As for gene <italic>RPL3P4</italic>, the regulatory target of such pseudogene, <italic>RPL3</italic> has been reported to have a quite low expression level in diabetic cells compared to normal cells (<xref ref-type="bibr" rid="B83">Tsai et al., 1994</xref>), corresponding with this rule. As for <italic>PSPHP1</italic> (8781), it has been shown to be associated with the macrophage-related inflammation processes (<xref ref-type="bibr" rid="B87">Walker et al., 2015</xref>). Considering that during the initiation and progression of diabetes, regional and systematic inflammation have been widely observed (<xref ref-type="bibr" rid="B23">Donath et al., 2003</xref>; <xref ref-type="bibr" rid="B55">Lontchi-Yimagou et al., 2013</xref>), it is reasonable to predict such genes as quantitative parameters for the distinction of non-diabetes and diabetes. As for <italic>PTCHD1</italic>, although no direct evidence confirms its contribution on diabetes, it has been confirmed that such a gene is associated with the eye and ear complications of diabetes (<xref ref-type="bibr" rid="B30">Gambin et al., 2017</xref>), consistent with this rule.</p>
<p>As for the second rule (rule2), four genes were involved including <italic>MICOS10P3</italic> (100462954), <italic>CTBP1</italic> (1487), <italic>RPL3P4</italic> (326307), and <italic>MTND4P24</italic> (100873254). Few publications have reported the biological contribution of <italic>MICOS10P3</italic>; therefore, it is hard to interpret such gene&#x2019;s contribution on T2D. As for gene <italic>CTBP1</italic>, it has been reported to participate in the abnormal phosphorylation processes (<xref ref-type="bibr" rid="B44">Kim et al., 2013</xref>) and shows a quite high expression level in diabetic cells compared to normal controls. As for gene <italic>RPL3P4</italic>, the regulatory target of such a pseudogene, <italic>RPL3</italic> has been reported to have a quite low expression level in diabetic cells compared to normal cells (<xref ref-type="bibr" rid="B83">Tsai et al., 1994</xref>), corresponding with such a rule. <italic>MTND4P24</italic> and its homolog, <italic>MTND5P11</italic>, have been confirmed to regulate a group of functional mitochondrial-encoded NADH ubiquinone oxidoreductase. According to recent publications, during the pathogenesis of diabetes, <italic>MT-ND4</italic> has a quite low-expression pattern and on the contrary, <italic>MT-ND5</italic> has a relevantly higher expression level, corresponding with the prediction expression level of their agonists individually (<xref ref-type="bibr" rid="B26">Elango et al., 2014</xref>; <xref ref-type="bibr" rid="B86">Urbanova et al., 2017</xref>).</p>
<p>In the third rule (rule3), apart from genes we have discussed earlier, the gene <italic>EIF5AL1</italic> (143244) has also been predicted to have a higher expression pattern in normal cells but not in diabetic cells. Considering the abnormal endocrine stress responses of diabetic cells (<xref ref-type="bibr" rid="B73">Siddiqui et al., 2015</xref>), the lower expression level of <italic>EIF5AL1</italic> may also contribute to the identification of diabetic cells. <italic>FXYD2</italic> (486) has been shown to contribute to the pathogenesis of diabetes (<xref ref-type="bibr" rid="B20">Ding et al., 2019</xref>). Another specific gene in rule3 is the homolog of <italic>RPL3P4</italic>, <italic>RPL9P7</italic>, which may also participate in the regulation of the pathogenesis of T2D with similar expression patterns to <italic>RPL3P4</italic>.</p>
<p>From the fourth to eighth rules, most of the involved genes occurred in the top three rules or were the top T2D-associated genes just with different combination patterns. Specific genes, like <italic>RPL9P9</italic> (388147) and <italic>RPL36AP21</italic> (100271332) for rule4, <italic>MT-TS1</italic> (4574) for rule5, <italic>RPL26P30</italic> (653147) and <italic>RPL6P20</italic> (285900) for rule6, <italic>RPL35P1</italic> (440737) for rule7, and <italic>RPS3AP26</italic> (644972) for rule8, have been identified in our quantitative rules. As we can see from such typical rule associating biomarkers, most of the genes are ribosome-associated genes like <italic>RPL3P4</italic> (326307) as discussed earlier. Although no direct evidence confirmed the associations between such genes and T2D, it is still reasonable to speculate that such genes may play an irreplaceable role in the identification of T2D. As for <italic>MT-TS1</italic>, such genes have already been reported as potential biomarkers for T2D (<xref ref-type="bibr" rid="B58">Mannino and Sesti, 2012</xref>), corresponding with our prediction.</p>
</sec>
<sec id="s6-5">
<title>4.5 Potential Applications of Identified Type 2 Diabetes-Associated Genes and Rules</title>
<p>There are two potential applications for identified T2D-associated genes: 1) potential biomarkers for T2D diagnosis and monitoring; 2) potential drug target for T2D therapy.</p>
<p>For the identified T2D-associated genes, considering that such genes are identified from pancreatic tissues, they can reflect the original tissue alterations during T2D initiation and progression. Therefore, such genes can be used as biomarkers for direct pancreatic islet biopsy examinations. Apart from that, the candidate genes as potential drug targets can also be manually regulated to prevent the initiation and progression of T2D. Using high-throughput drug screening, antibodies or chemicals specifically targeting the candidate genes can be identified and developed as potential target drugs for T2D.</p>
<p>For the quantitative T2D-associated rules, although we have already identified a group of genes associated with T2D, it is still quite difficult to diagnose T2D. With specific quantitative rules, the identification of T2D patients can be more accurate and efficient. Also, the rules can also be summarized as clinical guidelines for T2D diagnosis using pancreatic tissue single-cell sequencing techniques.</p>
</sec>
<sec id="s6-6">
<title>4.6 Functional Interpretation of Significant Rule Genes</title>
<p>As listed in <xref ref-type="table" rid="T2">Table 2</xref>, we identified quantitative rules associated with T2D. The GO enrichment analyses on rule genes were conducted. <xref ref-type="table" rid="T4">Table 4</xref> lists the enriched GO terms of these rule genes. It was indicated that most rules are shown to be associated with ribosome-associated biological processes. According to recent publications, ribosome-associated biological processes have been widely shown to be associated with the pathogenesis of T2D. In 2019, in a metabolic study on pancreatic tissues, ribosome-associated genes have been shown to participate in the ERK/hnRNPK/DDX3X pathway in pancreatic islet cells and further regulated the initiation and progression of T2D (<xref ref-type="bibr" rid="B34">Good et al., 2019</xref>), consistent with our results. Apart from that, in 2020, DIMT1, as a regulator of ribosomal biogenesis has been shown to participate in the physical biological processes of pancreatic tissue, further validating our results.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Significant Gene Ontology enrichment analysis result on rule genes.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">GO ID</th>
<th align="center">Term</th>
<th align="center">
<italic>p</italic>-value</th>
<th align="center">Cluster</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">GO:1903408</td>
<td align="left">Positive regulation of sodium: potassium-exchanging ATPase activity</td>
<td align="center">5.30E-04</td>
<td align="center">BP</td>
</tr>
<tr>
<td align="left">GO:0045901</td>
<td align="left">Positive regulation of translational elongation</td>
<td align="center">7.00E-04</td>
<td align="center">BP</td>
</tr>
<tr>
<td align="left">GO:0045905</td>
<td align="left">Positive regulation of translational termination</td>
<td align="center">7.00E-04</td>
<td align="center">BP</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s6-7">
<title>4.7 Limitations of Current Analyses</title>
<p>In this study, for the first time, we adopted several machine learning algorithms to identify disease-specific biomarkers using the mixed single-cell sequencing data. Such analyses may not only identify biomarkers from the single-cell level, getting rid of the bias generated by the averaged transcriptomics using the bulk sequencing method, but also overcome the sample size restriction of traditional single-cell analysis. Compared with traditional single-cell analysis, we did not focus on the classification of different cell subtypes but just the patients and control subjects, improving the analysis accuracy. However, there still remain three major limitations of current analyses on pancreatic single-cell sequencing data:<list list-type="simple">
<list-item>
<p>1) First, the dataset we used is still a relatively small dataset, with around 20 subjects. A larger single-cell sequencing dataset may improve the efficacy and accuracy of our results.</p>
</list-item>
<list-item>
<p>2) Second, the number of cells in each group is not balanced in the raw dataset. Although in the original publications the authors have claimed that the sampling procedure does not affect the distribution of cell subgroups in each subject, a more balanced dataset may perform better.</p>
</list-item>
<list-item>
<p>3) Single-cell sequencing always misses a lot of genes at low-expression levels which cannot be detected at the single-cell level but can be identified in bulk sequencing. Our analyses may also lose the gene expression profiling and analysis on such low-expression genes.</p>
</list-item>
</list>
</p>
</sec>
</sec>
</body>
<back>
<sec id="s7">
<title>Data Availability Statement</title>
<p>Publicly available datasets were analyzed in this study. These data can be found at: <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE81608">https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc&#x3d;GSE81608</ext-link>.</p>
</sec>
<sec id="s8">
<title>Author Contributions</title>
<p>Y-DC designed the study. ZL and XP performed the experiments. ZL analyzed the results. ZL and XP wrote the manuscript. All authors contributed to the research and reviewed the manuscript.</p>
</sec>
<sec id="s9">
<title>Funding</title>
<p>This research was funded by the Strategic Priority Research Program of Chinese Academy of Sciences (XDA26040304 and XDB38050200), the National Key R&#x26;D Program of China (2018YFC0910403), and the Fund of the Key Laboratory of Tissue Microenvironment and Tumor of Chinese Academy of Sciences (202002).</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fbioe.2022.890901/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fbioe.2022.890901/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table2.XLSX" id="SM1" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table3.XLSX" id="SM2" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table4.XLSX" id="SM3" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table1.XLSX" id="SM4" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<collab>American Diabetes Association</collab> (<year>2014</year>). <article-title>Diagnosis and Classification of Diabetes Mellitus</article-title>. <source>Diabetes Care</source> <volume>37</volume> (<issue>Suppl. 1</issue>), <fpage>S81</fpage>&#x2013;<lpage>S90</lpage>. <pub-id pub-id-type="doi">10.2337/dc14-S081</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andersen</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Pedersen</surname>
<given-names>C.-E. T.</given-names>
</name>
<name>
<surname>Moltke</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Hansen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Albrechtsen</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Grarup</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Genetics of Type 2 Diabetes: the Power of Isolated Populations</article-title>. <source>Curr. Diab Rep.</source> <volume>16</volume>, <fpage>65</fpage>. <pub-id pub-id-type="doi">10.1007/s11892-016-0757-z</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aubert</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Bisanz-Seyer</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Herzog</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>1992</year>). <article-title>Mitochondrial Rps14 Is a Transcribed and Edited Pseudogene in <italic>Arabidopsis thaliana</italic>
</article-title>. <source>Plant Mol. Biol.</source> <volume>20</volume>, <fpage>1169</fpage>&#x2013;<lpage>1174</lpage>. <pub-id pub-id-type="doi">10.1007/bf00028903</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boden</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Role of Fatty Acids in the Pathogenesis of Insulin Resistance and NIDDM</article-title>. <source>Diabetes</source> <volume>46</volume>, <fpage>3</fpage>&#x2013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.2337/diabetes.46.1.3</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Borg</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gotts&#xe4;ter</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Landin-Olsson</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fernlund</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Sundkvist</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>High Levels of Antigen-specific Islet Antibodies Predict Future&#x3b2; -Cell Failure in Patients with Onset of Diabetes in Adult Age1</article-title>. <source>J. Clin. Endocrinol. Metabolism</source> <volume>86</volume>, <fpage>3032</fpage>&#x2013;<lpage>3038</lpage>. <pub-id pub-id-type="doi">10.1210/jcem.86.7.7658</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breiman</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Random Forests</article-title>. <source>Mach. Learn.</source> <volume>45</volume>, <fpage>5</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1023/a:1010933404324</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Broer</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Van Duijn</surname>
<given-names>C. M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>GWAS and Meta-Analysis in Aging/Longevity</article-title>. <source>Adv. Exp. Med. Biol.</source> <volume>847</volume>, <fpage>107</fpage>&#x2013;<lpage>125</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-4939-2404-2_5</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Brunk</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Pazzani</surname>
<given-names>M. J.</given-names>
</name>
</person-group> (<year>1991</year>). &#x201c;<article-title>An Investigation of Noise-Tolerant Relational Concept Learning Algorithms</article-title>,&#x201d; in <conf-name>Proceedings of the Eighth International Conference</conf-name>, <conf-loc>Evanston, Illinois</conf-loc>, <conf-date>June, 1991</conf-date>, <fpage>389</fpage>&#x2013;<lpage>393</lpage>. <pub-id pub-id-type="doi">10.1016/b978-1-55860-200-7.50080-5</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Casanova</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Saldana</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chew</surname>
<given-names>E. Y.</given-names>
</name>
<name>
<surname>Danis</surname>
<given-names>R. P.</given-names>
</name>
<name>
<surname>Greven</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Ambrosius</surname>
<given-names>W. T.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Application of Random Forests Methods to Diabetic Retinopathy Classification Analyses</article-title>. <source>PLoS One</source> <volume>9</volume>, <fpage>e98587</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0098587</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chakraborty</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Doss</surname>
<given-names>C. G. P.</given-names>
</name>
<name>
<surname>Bandyopadhyay</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Agoramoorthy</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Influence of miRNA in Insulin Signaling Pathway and Insulin Resistance: Micro-molecules with a Major Role in Type-2 Diabetes</article-title>. <source>WIREs RNA</source> <volume>5</volume>, <fpage>697</fpage>&#x2013;<lpage>712</lpage>. <pub-id pub-id-type="doi">10.1002/wrna.1240</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y. H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>Y. D.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Predicting RNA 5-methylcytosine Sites by Using Essential Sequence Features and Distributions</article-title>. <source>Biomed. Res. Int.</source> <volume>2022</volume>, <fpage>4035462</fpage>. <pub-id pub-id-type="doi">10.1155/2022/4035462</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.-H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>Z.-H.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Identify Key Sequence Features to Improve CRISPR sgRNA Efficacy</article-title>. <source>IEEE Access</source> <volume>5</volume>, <fpage>26582</fpage>&#x2013;<lpage>26590</lpage>. <pub-id pub-id-type="doi">10.1109/access.2017.2775703</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>iMPT-FDNPL: Identification of Membrane Protein Types with Functional Domains and a Natural Language Processing Approach</article-title>. <source>Comput. Math. Methods Med.</source> <volume>2021</volume>, <fpage>7681497</fpage>. <pub-id pub-id-type="doi">10.1155/2021/7681497</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Choi</surname>
<given-names>J. H.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>S.-S.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>E. S.</given-names>
</name>
<name>
<surname>Jedrychowski</surname>
<given-names>M. P.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y. R.</given-names>
</name>
<name>
<surname>Jang</surname>
<given-names>H.-J.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Thrap3 Docks on Phosphoserine 273 of PPAR&#x3b3; and Controls Diabetic Gene Programming</article-title>. <source>Genes Dev.</source> <volume>28</volume>, <fpage>2361</fpage>&#x2013;<lpage>2369</lpage>. <pub-id pub-id-type="doi">10.1101/gad.249367.114</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Clocquet</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Egan</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Stoffers</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Muller</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Wideman</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chin</surname>
<given-names>G. A.</given-names>
</name>
<etal/>
</person-group> (<year>2000</year>). <article-title>Impaired Insulin Secretion and Increased Insulin Sensitivity in Familial Maturity-Onset Diabetes of the Young 4 (Insulin Promoter Factor 1 Gene)</article-title>. <source>Diabetes</source> <volume>49</volume>, <fpage>1856</fpage>&#x2013;<lpage>1864</lpage>. <pub-id pub-id-type="doi">10.2337/diabetes.49.11.1856</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cohen</surname>
<given-names>W. W.</given-names>
</name>
</person-group> (<year>1995</year>). &#x201c;<article-title>Fast Effective Rule Induction</article-title>,&#x201d; in <conf-name>Proceedings of the Twelfth International Conference on Machine Learning</conf-name>, <conf-loc>Tahoe City, CA</conf-loc>, <conf-date>July 9&#x2013;July 12, 1995</conf-date>, <fpage>115</fpage>&#x2013;<lpage>123</lpage>. <pub-id pub-id-type="doi">10.1016/b978-1-55860-377-6.50023-2</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cornelis</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Rimm</surname>
<given-names>E. B.</given-names>
</name>
<name>
<surname>Curhan</surname>
<given-names>G. C.</given-names>
</name>
<name>
<surname>Kraft</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Hunter</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>F. B.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Obesity Susceptibility Loci and Uncontrolled Eating, Emotional Eating and Cognitive Restraint Behaviors in Men and Women</article-title>. <source>Obesity</source> <volume>22</volume>, <fpage>E135</fpage>&#x2013;<lpage>E141</lpage>. <pub-id pub-id-type="doi">10.1002/oby.20592</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cortes</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Vapnik</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>1995</year>). <article-title>Support-vector Networks</article-title>. <source>Mach. Learn</source> <volume>20</volume>, <fpage>273</fpage>&#x2013;<lpage>297</lpage>. <pub-id pub-id-type="doi">10.1007/bf00994018</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cover</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hart</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>1967</year>). <article-title>Nearest Neighbor Pattern Classification</article-title>. <source>IEEE Trans. Inf. Theory</source> <volume>13</volume>, <fpage>21</fpage>&#x2013;<lpage>27</lpage>. <pub-id pub-id-type="doi">10.1109/tit.1967.1053964</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xue</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Identification of Core Genes and Pathways in Type 2 Diabetes Mellitus by Bioinformatics Analysis</article-title>. <source>Mol. Med. Rep.</source> <volume>20</volume>, <fpage>2597</fpage>&#x2013;<lpage>2608</lpage>. <pub-id pub-id-type="doi">10.3892/mmr.2019.10522</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Predicting Heart Cell Types by Using Transcriptome Profiles and a Machine Learning Method</article-title>. <source>Life</source> <volume>12</volume>, <fpage>228</fpage>. <pub-id pub-id-type="doi">10.3390/life12020228</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<collab>Disease and Injury Incidence and Prevalence Collaborators</collab> (<year>2017</year>). <article-title>Global, Regional, and National Incidence, Prevalence, and Years Lived with Disability for 328 Diseases and Injuries for 195 Countries, 1990-2016: a Systematic Analysis for the Global Burden of Disease Study 2016</article-title>. <source>Lancet</source> <volume>390</volume>, <fpage>1211</fpage>&#x2013;<lpage>1259</lpage>. <pub-id pub-id-type="doi">10.1016/S0140-6736(17)32154-2</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Donath</surname>
<given-names>M. Y.</given-names>
</name>
<name>
<surname>St&#xf8;rling</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Maedler</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Mandrup-Poulsen</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Inflammatory Mediators and Islet beta-cell Failure: a Link between Type 1 and Type 2 Diabetes</article-title>. <source>J. Mol. Med.</source> <volume>81</volume>, <fpage>455</fpage>&#x2013;<lpage>470</lpage>. <pub-id pub-id-type="doi">10.1007/s00109-003-0450-y</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Draminski</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rada-Iglesias</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Enroth</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wadelius</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Koronacki</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Komorowski</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Monte Carlo Feature Selection for Supervised Classification</article-title>. <source>Bioinformatics</source> <volume>24</volume>, <fpage>110</fpage>&#x2013;<lpage>117</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btm486</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eckardt</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>G&#xf6;rgens</surname>
<given-names>S. W.</given-names>
</name>
<name>
<surname>Raschke</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Eckel</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Myokines in Insulin Resistance and Type 2 Diabetes</article-title>. <source>Diabetologia</source> <volume>57</volume>, <fpage>1087</fpage>&#x2013;<lpage>1099</lpage>. <pub-id pub-id-type="doi">10.1007/s00125-014-3224-x</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elango</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Venugopal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Thangaraj</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Viswanadha</surname>
<given-names>V. P.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Novel Mutations in ATPase 8, ND1 and ND5 Genes Associated with Peripheral Neuropathy of Diabetes</article-title>. <source>Diabetes Res. Clin. Pract.</source> <volume>103</volume>, <fpage>e49</fpage>&#x2013;<lpage>e52</lpage>. <pub-id pub-id-type="doi">10.1016/j.diabres.2013.12.015</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ferrannini</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Insulin Resistance versus &#x3b2;-cell Dysfunction in the Pathogenesis of Type 2 Diabetes</article-title>. <source>Curr. Diab Rep.</source> <volume>9</volume>, <fpage>188</fpage>&#x2013;<lpage>189</lpage>. <pub-id pub-id-type="doi">10.1007/s11892-009-0031-8</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<collab>Global Burden of Disease Cancer Collaboration</collab>
<person-group person-group-type="author">
<name>
<surname>Fitzmaurice</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Allen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Barber</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Barregard</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Bhutta</surname>
<given-names>Z. A.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Global, Regional, and National Cancer Incidence, Mortality, Years of Life Lost, Years Lived with Disability, and Disability-Adjusted Life-Years for 32 Cancer Groups, 1990 to 2015: A Systematic Analysis for the Global Burden of Disease Study</article-title>. <source>JAMA Oncol.</source> <volume>3</volume>, <fpage>524</fpage>&#x2013;<lpage>548</lpage>. <pub-id pub-id-type="doi">10.1001/jamaoncol.2016.5688</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Forst</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Anastassiadis</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Diessel</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>L&#xf6;ffler</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pf&#xfc;tzner</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Effect of Linagliptin Compared with Glimepiride on Postprandial Glucose Metabolism, Islet Cell Function and Vascular Function Parameters in Patients with Type 2 Diabetes Mellitus Receiving Ongoing Metformin Treatment</article-title>. <source>Diabetes Metab. Res. Rev.</source> <volume>30</volume>, <fpage>582</fpage>&#x2013;<lpage>589</lpage>. <pub-id pub-id-type="doi">10.1002/dmrr.2525</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gambin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Rosenfeld</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Coban-Akdemir</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Identification of Novel Candidate Disease Genes from De Novo Exonic Copy Number Variants</article-title>. <source>Genome Med.</source> <volume>9</volume>, <fpage>83</fpage>. <pub-id pub-id-type="doi">10.1186/s13073-017-0472-7</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>H. X.</given-names>
</name>
<name>
<surname>Regier</surname>
<given-names>E. E.</given-names>
</name>
<name>
<surname>Close</surname>
<given-names>K. L.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>International Diabetes Federation World Diabetes Congress 2015</article-title>. <source>J. Diabetes</source> <volume>8</volume>, <fpage>300</fpage>. <pub-id pub-id-type="doi">10.1111/1753-0407.12377</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Garufi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pistritto</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Baldari</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Toietta</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cirone</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>D&#x2019;Orazi</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>p53-Dependent PUMA to DRAM Antagonistic Interplay as a Key Molecular Switch in Cell-Fate Decision in Normal/high Glucose Conditions</article-title>. <source>J. Exp. Clin. Cancer Res.</source> <volume>36</volume>, <fpage>126</fpage>. <pub-id pub-id-type="doi">10.1186/s13046-017-0596-z</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Golubnitschaja</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Moenkemann</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Trog</surname>
<given-names>D. B.</given-names>
</name>
<name>
<surname>Blom</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>De Vriese</surname>
<given-names>A. S.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Activation of Genes Inducing Cell-Cycle Arrest and of Increased DNA Repair in the Hearts of Rats with Early Streptozotocin-Induced Diabetes Mellitus</article-title>. <source>Med. Sci. Monit.</source> <volume>12</volume>, <fpage>BR68</fpage>&#x2013;<lpage>74</lpage>. </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Good</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Haemmerle</surname>
<given-names>M. W.</given-names>
</name>
<name>
<surname>Oguh</surname>
<given-names>A. U.</given-names>
</name>
<name>
<surname>Doliba</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Stoffers</surname>
<given-names>D. A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Metabolic Stress Activates an ERK/hnRNPK/DDX3X Pathway in Pancreatic &#x3b2; Cells</article-title>. <source>Mol. Metab.</source> <volume>26</volume>, <fpage>45</fpage>&#x2013;<lpage>56</lpage>. <pub-id pub-id-type="doi">10.1016/j.molmet.2019.05.009</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gerstein</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Small RNAs Originated from Pseudogenes: Cis- or Trans-acting?</article-title> <source>PLoS Comput. Biol.</source> <volume>5</volume>, <fpage>e1000449</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1000449</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hou&#x161;tek</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hejzlarov&#xe1;</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Vrback&#xfd;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Drahota</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Landa</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Z&#xed;dek</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Nonsynonymous Variants in Mt-Nd2, Mt-Nd4, and Mt-Nd5 Are Linked to Effects on Oxidative Phosphorylation and Insulin Sensitivity in Rat Conplastic Strains</article-title>. <source>Physiol. Genomics</source> <volume>44</volume>, <fpage>487</fpage>&#x2013;<lpage>494</lpage>. <pub-id pub-id-type="doi">10.1152/physiolgenomics.00156.2011</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jia</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zuo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>O-GlcNAcPRED-II: an Integrated Classification Algorithm for Identifying O-GlcNAcylation Sites Based on Fuzzy Undersampling and a K-Means PCA Oversampling Technique</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>2029</fpage>&#x2013;<lpage>2036</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty039</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jia</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Similarity-Based Machine Learning Model for Predicting the Metabolic Pathways of Compounds</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>130687</fpage>&#x2013;<lpage>130696</lpage>. <pub-id pub-id-type="doi">10.1109/access.2020.3009439</pub-id> </citation>
</ref>
<ref id="B39">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Johannes</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Widmer</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1994</year>). &#x201c;<article-title>Incremental Reduced Error Pruning</article-title>,&#x201d; in <conf-name>Proceedings of the Eleventh International Conference</conf-name>, <conf-loc>Rutgers University, New Brunswick, NJ</conf-loc>, <conf-date>July 10&#x2013;July 13, 1994</conf-date>, <fpage>70</fpage>&#x2013;<lpage>77</lpage>. <pub-id pub-id-type="doi">10.1016/b978-1-55860-335-6.50017-9</pub-id> </citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Johnson</surname>
<given-names>D. S.</given-names>
</name>
</person-group> (<year>1974</year>). <article-title>Approximation Algorithms for Combinatorial Problems</article-title>. <source>J. Comput. Syst. Sci.</source> <volume>9</volume>, <fpage>256</fpage>&#x2013;<lpage>278</lpage>. <pub-id pub-id-type="doi">10.1016/s0022-0000(74)80044-9</pub-id> </citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kahn</surname>
<given-names>S. E.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>The Relative Contributions of Insulin Resistance and Beta-Cell Dysfunction to the Pathophysiology of Type 2 Diabetes</article-title>. <source>Diabetologia</source> <volume>46</volume>, <fpage>3</fpage>&#x2013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1007/s00125-002-1009-0</pub-id> </citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kalyana-Sundaram</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kumar-Sinha</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Shankar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Robinson</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.-M.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Expressed Pseudogenes in the Transcriptional Landscape of Human Cancers</article-title>. <source>Cell</source> <volume>149</volume>, <fpage>1622</fpage>&#x2013;<lpage>1634</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2012.04.041</pub-id> </citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kandaswamy</surname>
<given-names>K. K.</given-names>
</name>
<name>
<surname>Chou</surname>
<given-names>K.-C.</given-names>
</name>
<name>
<surname>Martinetz</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>M&#xf6;ller</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Suganthan</surname>
<given-names>P. N.</given-names>
</name>
<name>
<surname>Sridharan</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>AFP-pred: A Random Forest Approach for Predicting Antifreeze Proteins from Sequence-Derived Properties</article-title>. <source>J. Theor. Biol.</source> <volume>270</volume>, <fpage>56</fpage>&#x2013;<lpage>62</lpage>. <pub-id pub-id-type="doi">10.1016/j.jtbi.2010.10.037</pub-id> </citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>J.-H.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>S.-Y.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>B.-H.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S.-M.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>H. S.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>G.-Y.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>AMP-activated Protein Kinase Phosphorylates CtBP1 and Down-Regulates its Activity</article-title>. <source>Biochem. Biophysical Res. Commun.</source> <volume>431</volume>, <fpage>8</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1016/j.bbrc.2012.12.117</pub-id> </citation>
</ref>
<ref id="B45">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kohavi</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>1995</year>). &#x201c;<article-title>A Study of Cross-Validation and Bootstrap for Accuracy Estimation and Model Selection</article-title>,&#x201d; in <conf-name>International Joint Conference on Artificial Intelligence</conf-name>, <conf-loc>Montreal Quebec Canada</conf-loc>, <conf-date>August 20&#x2013;August 25, 1995</conf-date> (<publisher-name>Lawrence Erlbaum Associates Ltd</publisher-name>), <fpage>1137</fpage>&#x2013;<lpage>1145</lpage>. </citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lawlor</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>George</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bolisetty</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kursawe</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Sivakamasundari</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Single-cell Transcriptomes Identify Human Islet Cell Signatures and Reveal Cell-type-specific Expression Changes in Type 2 Diabetes</article-title>. <source>Genome Res.</source> <volume>27</volume>, <fpage>208</fpage>&#x2013;<lpage>222</lpage>. <pub-id pub-id-type="doi">10.1101/gr.212720.116</pub-id> </citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lebovitz</surname>
<given-names>C. B.</given-names>
</name>
<name>
<surname>Robertson</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Goya</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Morin</surname>
<given-names>R. D.</given-names>
</name>
<name>
<surname>Marra</surname>
<given-names>M. A.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Cross-cancer Profiling of Molecular Alterations within the Human Autophagy Interaction Network</article-title>. <source>Autophagy</source> <volume>11</volume>, <fpage>1668</fpage>&#x2013;<lpage>1687</lpage>. <pub-id pub-id-type="doi">10.1080/15548627.2015.1067362</pub-id> </citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>M.-S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Role of Islet &#x3b2; Cell Autophagy in the Pathogenesis of Diabetes</article-title>. <source>Trends Endocrinol. Metabolism</source> <volume>25</volume>, <fpage>620</fpage>&#x2013;<lpage>627</lpage>. <pub-id pub-id-type="doi">10.1016/j.tem.2014.08.005</pub-id> </citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Identification of Protein Functions in Mouse with a Label Space Partition Method</article-title>. <source>Mbe</source> <volume>19</volume>, <fpage>3820</fpage>&#x2013;<lpage>3842</lpage>. <pub-id pub-id-type="doi">10.3934/mbe.2022176</pub-id> </citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Exploring the Genomic Patterns in Human and Mouse Cerebellums via Single-Cell Sequencing and Machine Learning Method</article-title>. <source>Front. Genet.</source> <volume>13</volume>, <fpage>857851</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2022.857851</pub-id> </citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Prediction of Drug Side Effects with a Refined Negative Sample Selection Strategy</article-title>. <source>Comput. Math. Methods Med.</source> <volume>2020</volume>, <fpage>1573543</fpage>. <pub-id pub-id-type="doi">10.1155/2020/1573543</pub-id> </citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Identifying Protein Subcellular Location with Embedding Features Learned from Networks</article-title>. <source>Cp</source> <volume>18</volume>, <fpage>646</fpage>&#x2013;<lpage>660</lpage>. <pub-id pub-id-type="doi">10.2174/1570164617999201124142950</pub-id> </citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Setiono</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Incremental Feature Selection</article-title>. <source>Appl. Intell.</source> <volume>9</volume>, <fpage>217</fpage>&#x2013;<lpage>230</lpage>. <pub-id pub-id-type="doi">10.1023/a:1008363719778</pub-id> </citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>J.-Y.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.-M.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>Y.-C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.-Q.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.-J.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Pathogenic Role of lncRNA-MALAT1 in Endothelial Cell Dysfunction in Diabetes Mellitus</article-title>. <source>Cell Death Dis.</source> <volume>5</volume>, <fpage>e1506</fpage>. <pub-id pub-id-type="doi">10.1038/cddis.2014.466</pub-id> </citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lontchi-Yimagou</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Sobngwi</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Matsha</surname>
<given-names>T. E.</given-names>
</name>
<name>
<surname>Kengne</surname>
<given-names>A. P.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Diabetes Mellitus and Inflammation</article-title>. <source>Curr. Diab Rep.</source> <volume>13</volume>, <fpage>435</fpage>&#x2013;<lpage>444</lpage>. <pub-id pub-id-type="doi">10.1007/s11892-013-0375-y</pub-id> </citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lyons</surname>
<given-names>T. J.</given-names>
</name>
<name>
<surname>Basu</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Biomarkers in Diabetes: Hemoglobin A1c, Vascular and Tissue Markers</article-title>. <source>Transl. Res.</source> <volume>159</volume>, <fpage>303</fpage>&#x2013;<lpage>312</lpage>. <pub-id pub-id-type="doi">10.1016/j.trsl.2012.01.009</pub-id> </citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Single-cell Gene Expression Analysis Reveals &#x3b2;-cell Dysfunction and Deficit Mechanisms in Type 2 Diabetes</article-title>. <source>BMC Bioinforma.</source> <volume>19</volume>, <fpage>515</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-018-2519-1</pub-id> </citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mannino</surname>
<given-names>G. C.</given-names>
</name>
<name>
<surname>Sesti</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Individualized Therapy for Type 2 Diabetes: Clinical Implications Of Pharmacogenetic Data</article-title>. <source>Mol. Diagn Ther.</source> <volume>16</volume>, <fpage>285</fpage>&#x2013;<lpage>302</lpage>. <pub-id pub-id-type="doi">10.1007/s40291-012-0002-7</pub-id> </citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marques</surname>
<given-names>Y. B.</given-names>
</name>
<name>
<surname>De Paiva Oliveira</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ribeiro Vasconcelos</surname>
<given-names>A. T.</given-names>
</name>
<name>
<surname>Cerqueira</surname>
<given-names>F. R.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Mirnacle: Machine Learning with SMOTE and Random Forest for Improving Selectivity in Pre-miRNA Ab Initio Prediction</article-title>. <source>BMC Bioinforma.</source> <volume>17</volume>, <fpage>474</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-016-1343-8</pub-id> </citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mathews</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Leiter</surname>
<given-names>E. H.</given-names>
</name>
<name>
<surname>Spirina</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Bykhovskaya</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gusdon</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Ringquist</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2005</year>). <article-title>mt-Nd2 Allele of the ALR/Lt Mouse Confers Resistance against Both Chemically Induced and Autoimmune Diabetes</article-title>. <source>Diabetologia</source> <volume>48</volume>, <fpage>261</fpage>&#x2013;<lpage>267</lpage>. <pub-id pub-id-type="doi">10.1007/s00125-004-1644-8</pub-id> </citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Matthews</surname>
<given-names>B. W.</given-names>
</name>
</person-group> (<year>1975</year>). <article-title>Comparison of the Predicted and Observed Secondary Structure of T4 Phage Lysozyme</article-title>. <source>Biochimica Biophysica Acta (BBA) - Protein Struct.</source> <volume>405</volume>, <fpage>442</fpage>&#x2013;<lpage>451</lpage>. <pub-id pub-id-type="doi">10.1016/0005-2795(75)90109-9</pub-id> </citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mella</surname>
<given-names>M. T.</given-names>
</name>
<name>
<surname>Kohari</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pe&#xf1;a</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ferrara</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Stone</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Mitochondrial Gene Expression Profiles Are Associated with Intrahepatic Cholestasis of Pregnancy</article-title>. <source>Placenta</source> <volume>45</volume>, <fpage>16</fpage>&#x2013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1016/j.placenta.2016.07.002</pub-id> </citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mirza</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Berthelsen</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Seemann</surname>
<given-names>S. E.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Frederiksen</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Vilien</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Transcriptomic Landscape of lncRNAs in Inflammatory Bowel Disease</article-title>. <source>Genome Med.</source> <volume>7</volume>, <fpage>39</fpage>. <pub-id pub-id-type="doi">10.1186/s13073-015-0162-2</pub-id> </citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Onesime</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Genomic Island Prediction via Chi-Square Test and Random Forest Algorithm</article-title>. <source>Comput. Math. Methods Med.</source> <volume>2021</volume>, <fpage>9969751</fpage>. <pub-id pub-id-type="doi">10.1155/2021/9969751</pub-id> </citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pan</surname>
<given-names>X.-Y.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>H.-B.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Robust Prediction of B-Factor Profile from Sequence Using Two-Stage SVR Based on Random Forest Feature Selection</article-title>. <source>Ppl</source> <volume>16</volume>, <fpage>1447</fpage>&#x2013;<lpage>1454</lpage>. <pub-id pub-id-type="doi">10.2174/092986609789839250</pub-id> </citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Identification of Protein Subcellular Localization with Network and Functional Embeddings</article-title>. <source>Front. Genet.</source> <volume>11</volume>, <fpage>626500</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2020.626500</pub-id> </citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pandey</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chawla</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Guchhait</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Type-2 Diabetes: Current Understanding and Future Perspectives</article-title>. <source>IUBMB Life</source> <volume>67</volume>, <fpage>506</fpage>&#x2013;<lpage>513</lpage>. <pub-id pub-id-type="doi">10.1002/iub.1396</pub-id> </citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Poliseno</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Marranci</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pandolfi</surname>
<given-names>P. P.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Pseudogenes in Human Cancer</article-title>. <source>Front. Med.</source> <volume>2</volume>, <fpage>68</fpage>. <pub-id pub-id-type="doi">10.3389/fmed.2015.00068</pub-id> </citation>
</ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prentki</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nolan</surname>
<given-names>C. J.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Islet Cell Failure in Type 2 Diabetes</article-title>. <source>J. Clin. Investigation</source> <volume>116</volume>, <fpage>1802</fpage>&#x2013;<lpage>1812</lpage>. <pub-id pub-id-type="doi">10.1172/jci29103</pub-id> </citation>
</ref>
<ref id="B70">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pullen</surname>
<given-names>T. J.</given-names>
</name>
<name>
<surname>Rutter</surname>
<given-names>G. A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Roles of lncRNAs in Pancreatic Beta Cell Identity and Diabetes Susceptibility</article-title>. <source>Front. Genet.</source> <volume>5</volume>, <fpage>193</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2014.00193</pub-id> </citation>
</ref>
<ref id="B71">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Quinlan</surname>
<given-names>J. R.</given-names>
</name>
</person-group> (<year>1990</year>). <article-title>Learning Logical Definitions from Relations</article-title>. <source>Mach. Learn</source> <volume>5</volume>, <fpage>239</fpage>&#x2013;<lpage>266</lpage>. <pub-id pub-id-type="doi">10.1007/bf00117105</pub-id> </citation>
</ref>
<ref id="B72">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Segerstolpe</surname>
<given-names>&#xc5;.</given-names>
</name>
<name>
<surname>Palasantza</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Eliasson</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Andersson</surname>
<given-names>E.-M.</given-names>
</name>
<name>
<surname>Andr&#xe9;asson</surname>
<given-names>A.-C.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Single-Cell Transcriptome Profiling of Human Pancreatic Islets in Health and Type 2 Diabetes</article-title>. <source>Cell Metab.</source> <volume>24</volume>, <fpage>593</fpage>&#x2013;<lpage>607</lpage>. <pub-id pub-id-type="doi">10.1016/j.cmet.2016.08.020</pub-id> </citation>
</ref>
<ref id="B73">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Siddiqui</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Madhu</surname>
<given-names>S. V.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Desai</surname>
<given-names>N. G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Endocrine Stress Responses and Risk of Type 2 Diabetes Mellitus</article-title>. <source>Stress</source> <volume>18</volume>, <fpage>498</fpage>&#x2013;<lpage>506</lpage>. <pub-id pub-id-type="doi">10.3109/10253890.2015.1067677</pub-id> </citation>
</ref>
<ref id="B74">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Soto</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Cheruvu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bialo</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Quintos</surname>
<given-names>J. B.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Refractory Diabetes Insipidus Leading to Diagnosis of Type 2 Diabetes Mellitus and Non-ketotic Hyperglycemia in an Adolescent Male</article-title>. <source>R. I. Med. J. (2013)</source> <volume>97</volume>, <fpage>34</fpage>&#x2013;<lpage>35</lpage>. </citation>
</ref>
<ref id="B75">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stancakova</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Laakso</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Genetics of Type 2 Diabetes</article-title>. <source>Endocr. Dev.</source> <volume>31</volume>, <fpage>203</fpage>&#x2013;<lpage>220</lpage>. <pub-id pub-id-type="doi">10.2337/dc10-1013</pub-id> </citation>
</ref>
<ref id="B76">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>iATC-NFMLP: Identifying Classes of Anatomical Therapeutic Chemicals Based on Drug Networks, Fingerprints and Multilayer Perceptron</article-title>. <source>Curr. Bioinforma.</source> <volume>17</volume>. <pub-id pub-id-type="doi">10.2174/1574893617666220318093000</pub-id> </citation>
</ref>
<ref id="B77">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Epidemiological Perspectives of Diabetes</article-title>. <source>Cell Biochem. Biophys.</source> <volume>73</volume>, <fpage>181</fpage>&#x2013;<lpage>185</lpage>. <pub-id pub-id-type="doi">10.1007/s12013-015-0598-4</pub-id> </citation>
</ref>
<ref id="B78">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tay</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Rinn</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Pandolfi</surname>
<given-names>P. P.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>The Multilayered Complexity of ceRNA Crosstalk and Competition</article-title>. <source>Nature</source> <volume>505</volume>, <fpage>344</fpage>&#x2013;<lpage>352</lpage>. <pub-id pub-id-type="doi">10.1038/nature12986</pub-id> </citation>
</ref>
<ref id="B79">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Taylor</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Type 2 Diabetes: Etiology And Reversibility</article-title>. <source>Diabetes Care</source> <volume>36</volume>, <fpage>1047</fpage>&#x2013;<lpage>1055</lpage>. <pub-id pub-id-type="doi">10.2337/dc12-1805</pub-id> </citation>
</ref>
<ref id="B80">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thorens</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Neural Regulation of Pancreatic Islet Cell Mass and Function</article-title>. <source>Diabetes Obes. Metab.</source> <volume>16</volume> (<issue>Suppl. 1</issue>), <fpage>87</fpage>&#x2013;<lpage>95</lpage>. <pub-id pub-id-type="doi">10.1111/dom.12346</pub-id> </citation>
</ref>
<ref id="B81">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Torrell</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Monta&#xf1;a</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Abasolo</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Roig</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Gaviria</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Vilella</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Mitochondrial DNA (mtDNA) in Brain Samples from Patients with Major Psychiatric Disorders: Gene Expression Profiles, mtDNA Content and Presence of the mtDNA Common Deletion</article-title>. <source>Am. J. Med. Genet.</source> <volume>162</volume>, <fpage>213</fpage>&#x2013;<lpage>223</lpage>. <pub-id pub-id-type="doi">10.1002/ajmg.b.32134</pub-id> </citation>
</ref>
<ref id="B82">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Trujillo</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Nuffer</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>GLP-1 Receptor Agonists for Type 2 Diabetes Mellitus: Recent Developments and Emerging Agents</article-title>. <source>Pharmacotherapy</source> <volume>34</volume>, <fpage>1174</fpage>&#x2013;<lpage>1186</lpage>. <pub-id pub-id-type="doi">10.1002/phar.1507</pub-id> </citation>
</ref>
<ref id="B83">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tsai</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cowan</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>D. G.</given-names>
</name>
<name>
<surname>Brannon</surname>
<given-names>P. M.</given-names>
</name>
</person-group> (<year>1994</year>). <article-title>Regulation of Pancreatic Amylase and Lipase Gene Expression by Diet and Insulin in Diabetic Rats</article-title>. <source>Am. J. Physiology-Gastrointestinal Liver Physiology</source> <volume>267</volume>, <fpage>G575</fpage>&#x2013;<lpage>G583</lpage>. <pub-id pub-id-type="doi">10.1152/ajpgi.1994.267.4.g575</pub-id> </citation>
</ref>
<ref id="B84">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tseng</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Landolph</surname>
<given-names>J. R.</given-names>
<suffix>Jr.</suffix>
</name>
</person-group> (<year>2012</year>). <article-title>Diabetes and Cancer: Epidemiological, Clinical, and Experimental Perspectives</article-title>. <source>Exp. Diabetes Res.</source> <volume>2012</volume>, <fpage>101802</fpage>. <pub-id pub-id-type="doi">10.1155/2012/101802</pub-id> </citation>
</ref>
<ref id="B85">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tsuyuki</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Takabayashi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kawazu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kudo</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Watanabe</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nagata</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Detection ofWIPI1mRNA as an Indicator of Autophagosome Formation</article-title>. <source>Autophagy</source> <volume>10</volume>, <fpage>497</fpage>&#x2013;<lpage>513</lpage>. <pub-id pub-id-type="doi">10.4161/auto.27419</pub-id> </citation>
</ref>
<ref id="B86">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Urbanov&#xe1;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mr&#xe1;z</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>&#x10e;urovcov&#xe1;</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Trachta</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Klou&#x10d;kov&#xe1;</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kav&#xe1;lkov&#xe1;</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>The Effect of Very-Low-Calorie Diet on Mitochondrial Dysfunction in Subcutaneous Adipose Tissue and Peripheral Monocytes of Obese Subjects with Type 2 Diabetes Mellitus</article-title>. <source>Physiol. Res.</source> <volume>66</volume>, <fpage>811</fpage>&#x2013;<lpage>822</lpage>. <pub-id pub-id-type="doi">10.33549/physiolres.933469</pub-id> </citation>
</ref>
<ref id="B87">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Walker</surname>
<given-names>W. E.</given-names>
</name>
<name>
<surname>Kurscheid</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Joshi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lopez</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Goh</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Increased Levels of Macrophage Inflammatory Proteins Result in Resistance to R5-Tropic HIV-1 in a Subset of Elite Controllers</article-title>. <source>J. Virol.</source> <volume>89</volume>, <fpage>5502</fpage>&#x2013;<lpage>5514</lpage>. <pub-id pub-id-type="doi">10.1128/jvi.00118-15</pub-id> </citation>
</ref>
<ref id="B88">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Using Recursive Feature Selection with Random Forest to Improve Protein Structural Class Prediction for Low-Similarity Sequences</article-title>. <source>Comput. Math. Methods Med.</source> <volume>2021</volume>, <fpage>5529389</fpage>. <pub-id pub-id-type="doi">10.1155/2021/5529389</pub-id> </citation>
</ref>
<ref id="B89">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Luan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nagai</surname>
<given-names>L. A. E.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Exploring Sequence-Based Features for the Improved Prediction of DNA N4-Methylcytosine Sites in Multiple Species</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>1326</fpage>&#x2013;<lpage>1333</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty824</pub-id> </citation>
</ref>
<ref id="B90">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Westermark</surname>
<given-names>G. T.</given-names>
</name>
<name>
<surname>Westermark</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Importance of Aggregated Islet Amyloid Polypeptide for the Progressive Beta-Cell Failure in Type 2 Diabetes and in Transplanted Human Islets</article-title>. <source>Exp. Diabetes Res.</source> <volume>2008</volume>, <fpage>528354</fpage>. <pub-id pub-id-type="doi">10.1155/2008/528354</pub-id> </citation>
</ref>
<ref id="B91">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Similarity-based Method with Multiple-Feature Sampling for Predicting Drug Side Effects</article-title>. <source>Comput. Math. Methods Med.</source> <volume>2022</volume>, <fpage>1</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1155/2022/9547317</pub-id> </citation>
</ref>
<ref id="B92">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Okamoto</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ni</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Adler</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>RNA Sequencing of Single Human Islet Cells Reveals Type 2 Diabetes Genes</article-title>. <source>Cell Metab.</source> <volume>24</volume>, <fpage>608</fpage>&#x2013;<lpage>615</lpage>. <pub-id pub-id-type="doi">10.1016/j.cmet.2016.08.018</pub-id> </citation>
</ref>
<ref id="B93">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yabe</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Seino</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fukushima</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Seino</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>&#x3b2; Cell Dysfunction versus Insulin Resistance in the Pathogenesis of Type 2 Diabetes in East Asians</article-title>. <source>Curr. Diab Rep.</source> <volume>15</volume>, <fpage>602</fpage>. <pub-id pub-id-type="doi">10.1007/s11892-015-0602-9</pub-id> </citation>
</ref>
<ref id="B94">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Identification of Drug-Disease Associations by Using Multiple Drug and Disease Networks</article-title>. <source>Cbio</source> <volume>17</volume>, <fpage>48</fpage>&#x2013;<lpage>59</lpage>. <pub-id pub-id-type="doi">10.2174/1574893616666210825115406</pub-id> </citation>
</ref>
<ref id="B95">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.-H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2021a</year>). <article-title>Identifying Transcriptomic Signatures and Rules for SARS-CoV-2 Infection</article-title>. <source>Front. Cell Dev. Biol.</source> <volume>8</volume>, <fpage>627302</fpage>. <pub-id pub-id-type="doi">10.3389/fcell.2020.627302</pub-id> </citation>
</ref>
<ref id="B96">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.-H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2021b</year>). <article-title>Detecting the Multiomics Signatures of Factor-specific Inflammatory Effects on Airway Smooth Muscles</article-title>. <source>Front. Genet.</source> <volume>11</volume>, <fpage>599970</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2020.599970</pub-id> </citation>
</ref>
<ref id="B97">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.-H.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>Y.-D.</given-names>
</name>
</person-group> (<year>2021c</year>). <article-title>Determining Protein-Protein Functional Associations by Functional Rules Based on Gene Ontology and KEGG Pathway</article-title>. <source>Biochimica Biophysica Acta (BBA) - Proteins Proteomics</source> <volume>1869</volume>, <fpage>140621</fpage>. <pub-id pub-id-type="doi">10.1016/j.bbapap.2021.140621</pub-id> </citation>
</ref>
<ref id="B98">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Z.-H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Predicting Drug Side Effects with Compact Integration of Heterogeneous Networks</article-title>. <source>Cbio</source> <volume>14</volume>, <fpage>709</fpage>&#x2013;<lpage>720</lpage>. <pub-id pub-id-type="doi">10.2174/1574893614666190220114644</pub-id> </citation>
</ref>
<ref id="B99">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A Similarity-Based Method for Prediction of Drug Side Effects with Heterogeneous Information</article-title>. <source>Math. Biosci.</source> <volume>306</volume>, <fpage>136</fpage>&#x2013;<lpage>144</lpage>. <pub-id pub-id-type="doi">10.1016/j.mbs.2018.09.010</pub-id> </citation>
</ref>
<ref id="B100">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>J.-P.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020b</year>). <article-title>iATC-FRAKEL: a Simple Multi-Label Web Server for Recognizing Anatomical Therapeutic Chemical Classes of Drugs with Their Fingerprints Only</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>3568</fpage>&#x2013;<lpage>3569</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa166</pub-id> </citation>
</ref>
<ref id="B101">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Z. H.</given-names>
</name>
</person-group> (<year>2020a</year>). <article-title>iATC-NRAKEL: An Efficient Multi-Label Classifier for Recognizing Anatomical Therapeutic Chemical Classes of Drugs</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>1391</fpage>&#x2013;<lpage>1396</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz757</pub-id> </citation>
</ref>
<ref id="B102">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Hao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Ribosomal Protein S14 Unties the MDM2-P53 Loop upon Ribosomal Stress</article-title>. <source>Oncogene</source> <volume>32</volume>, <fpage>388</fpage>&#x2013;<lpage>396</lpage>. <pub-id pub-id-type="doi">10.1038/onc.2012.63</pub-id> </citation>
</ref>
<ref id="B103">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>iMPTCE-Hnetwork: A Multilabel Classifier for Identifying Metabolic Pathway Types of Chemicals and Enzymes with a Heterogeneous Network</article-title>. <source>Comput. Math. Methods Med.</source> <volume>2021</volume>, <fpage>6683051</fpage>. <pub-id pub-id-type="doi">10.1155/2021/6683051</pub-id> </citation>
</ref>
<ref id="B104">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zick</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Insulin Resistance: a Phosphorylation-Based Uncoupling of Insulin Signaling</article-title>. <source>Trends Cell Biol.</source> <volume>11</volume>, <fpage>437</fpage>&#x2013;<lpage>441</lpage>. <pub-id pub-id-type="doi">10.1016/s0962-8924(01)81297-6</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>