<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article article-type="methods-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">890672</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.890672</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>An Interaction-Based Method for Refining Results From Gene Set Enrichment Analysis</article-title>
<alt-title alt-title-type="left-running-head">Wang et al.</alt-title>
<alt-title alt-title-type="right-running-head">Method to Refine GSEA Results</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Yishen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hong</surname>
<given-names>Yiwen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mao</surname>
<given-names>Shudi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jiang</surname>
<given-names>Yukang</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cui</surname>
<given-names>Yamei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Pan</surname>
<given-names>Jianying</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Luo</surname>
<given-names>Yan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/630591/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>State Key Laboratory of Ophthalmology</institution>, <institution>Zhongshan Ophthalmic Center</institution>, <institution>Sun Yat-Sen University</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Statistical Science</institution>, <institution>School of Mathematics</institution>, <institution>Sun Yat-Sen University</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/619688/overview">Maurice HT. Ling</ext-link>, Temasek Polytechnic, Singapore</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1436391/overview">Adison Choonkit Wong</ext-link>, Singapore Institute of Technology, Singapore</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1696467/overview">Jing Wui Yeoh</ext-link>, National University of Singapore, Singapore</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Yan Luo, <email>luoyan2@mail.sysu.edu.cn</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Computational Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>30</day>
<month>05</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>890672</elocation-id>
<history>
<date date-type="received">
<day>06</day>
<month>03</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>05</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Wang, Hong, Mao, Jiang, Cui, Pan and Luo.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Wang, Hong, Mao, Jiang, Cui, Pan and Luo</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>
<bold>Purpose:</bold> To demonstrate an interaction-based method for the refinement of Gene Set Enrichment Analysis (GSEA) results.</p>
<p>
<bold>Method:</bold> Intravitreal injection of miR-124-3p antagomir was used to knockdown the expression of miR-124-3p in mouse retina at postnatal day 3 (P3). Whole retinal RNA was extracted for mRNA transcriptome sequencing at P9. After preprocessing the dataset, GSEA was performed, and the leading-edge subsets were obtained. The Apriori algorithm was used to identify the frequent genes or gene sets from the union of the leading-edge subsets. A new statistic <inline-formula id="inf1">
<mml:math id="m1">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> was introduced to evaluate the frequent genes or gene sets. Reverse transcription quantitative PCR (RT-qPCR) was performed to validate the expression trend of candidate genes after the knockdown of miR-124-3p.</p>
<p>
<bold>Results:</bold> A total of 115,140 assembled transcript sequences were obtained from the clean data. With GSEA, the NOD-like receptor signaling pathway, C-type-like lectin receptor signaling pathway, phagosome, necroptosis, JAK-STAT signaling pathway, Toll-like receptor signaling pathway, leukocyte transendothelial migration, chemokine signaling pathway, NF-kappa B signaling pathway and RIG-I-like signaling pathway were identified as the top 10 enriched pathways, and their leading-edge subsets were obtained. After being refined by the Apriori algorithm and sorted by the value of the modulus of <inline-formula id="inf2">
<mml:math id="m2">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula>
<bold>,</bold> Prkcd, Irf9, Stat3, Cxcl12, Stat1, Stat2, Isg15, Eif2ak2, Il6st, Pdgfra, Socs4 and Csf2ra had the significant number of interactions and the greatest value of <inline-formula id="inf3">
<mml:math id="m3">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> to downstream genes among all frequent transactions. Results of RT-qPCR validation for the expression of candidate genes after the knockdown of miR-124-3p showed a similar trend to the RNA-Seq results.</p>
<p>
<bold>Conclusion:</bold> This study indicated that using the Apriori algorithm and defining the statistic <inline-formula id="inf4">
<mml:math id="m4">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> was a novel way to refine the GSEA results. We hope to convey the intricacies from the computational results to the low-throughput experiments, and to plan experimental investigations specifically.</p>
</abstract>
<kwd-group>
<kwd>GSEA</kwd>
<kwd>RNA-Seq</kwd>
<kwd>MiR-124-3p</kwd>
<kwd>apriori algorithm</kwd>
<kwd>miRNA</kwd>
</kwd-group>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">Natural Science Foundation of Guangdong Province<named-content content-type="fundref-id">10.13039/501100003453</named-content>
</contract-sponsor>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>MicroRNAs (miRNAs) are a class of noncoding RNAs that play key roles in regulating gene expression and are involved in a variety of biological processes during retinal development (<xref ref-type="bibr" rid="B6">Damiani et al., 2008</xref>). In most cases, miRNAs inhibit the expression or promote the degradation of messenger RNA (mRNA) by interacting with the specific sequences located in the 3 UTR of their target mRNA (<xref ref-type="bibr" rid="B9">Ha and Kim, 2014</xref>). According to this feature, each miRNA can target hundreds of mRNAs. The development of transcriptomics technologies such as RNA sequencing (RNA-Seq) can provide a broad account of RNA transcripts and consequently insight into changes in the mRNA expression of downstream genes after experimental intervention on the miRNA (<xref ref-type="bibr" rid="B24">Wang et al., 2009</xref>).</p>
<p>Due to the large amount of sequencing data generated by RNA-Seq, appropriate bioinformatics methods are needed to handle the data. After preprocessing the data, traditional quantitative analysis for mRNA expression analysis focuses on identifying differentially expressed upregulated and downregulated genes between two individual groups. Traditional strategies usually set an arbitrary cutoff in terms of expression fold-change (e.g., fold-change &#x2265; 1.5 considered significant) to filter the critical genes. Conservative and relaxed cutoff values may cause false negative and false positive results, respectively, making the results less objective and reproducible. To overcome the analytical challenges of focusing on a single gene, gene set enrichment analysis (GSEA) helps to gain further insight into the distribution of genes preannotated in biological categories by incorporating the entirety of gene expression. However, both single and global gene analysis often generate a large number of candidate genes (<xref ref-type="bibr" rid="B1">Ackermann and Strimmer, 2009</xref>). The lack of golden standard datasets also makes the assessment of gene set analysis methods rudimentary (<xref ref-type="bibr" rid="B15">Maleki et al., 2020</xref>). Hence, instead of extracting more genes from datasets, the goal of this study was to attempt reducing the dimensionality of analysis results and refining the intricacies from the high-throughput results to the low-throughput experiments.</p>
<p>Several studies have characterized miRNA expression in the developing mammalian retina, and miR-124-3p has been shown to be one of the most abundantly expressed miRNAs (<xref ref-type="bibr" rid="B12">Karali et al., 2007</xref>; <xref ref-type="bibr" rid="B10">Hackler et al., 2010</xref>; <xref ref-type="bibr" rid="B11">Karali et al., 2010</xref>; <xref ref-type="bibr" rid="B13">Karali et al., 2016</xref>). Although the miR-124a-3p knockdown mouse exhibited neuronal dysfunction and dysmaturation by disinhibition of Lhx2, its downstream responses in mouse retinal development after birth were still relatively deficient at the transcriptome level. In our previous study, miR-124-3p exhibited a significantly increasing trend after birth, and its related pathways predicted by bioinformatics analysis were associated with biological processes that may play crucial roles in mouse retinal development (<xref ref-type="bibr" rid="B23">Wang et al., 2020</xref>). To gain insight into its downstream processes, RNA-Seq was used to obtain the expression profile of mRNA transcripts after the knockdown of miR-124-3p.</p>
<p>In GSEA, given a set of genes sorted depending on given conditions (e.g., mRNA expression level of downstream genes) and a biological category, a running sum statistic is computed iteratively from top to bottom of the sorted set to evaluate whether it is enriched in the given biological category. When processed, the running sum will increase whenever a gene belonging to the given biological category is found and otherwise decrease. Therefore, the running sum will be relatively high if the gene set falls at either the top (overexpressed) or bottom (underexpressed) and is likely to be subsequently related to the given biological category.</p>
<p>Based on the properties mentioned above, GSEA is an appropriate tool to obtain a precise description of the downstream effects of miRNAs. Since a miRNA and its target mRNAs demonstrate negative correlations because of degradation (<xref ref-type="bibr" rid="B17">Ritchie et al., 2009</xref>; <xref ref-type="bibr" rid="B22">Wang and Li, 2009</xref>), when the expression value of the miRNA is manipulated to decrease, the upregulated and downregulated mRNA expression can be assumed to be its direct effects and indirect effects, which are likely to be enriched at the top and bottom in GSEA, respectively, and vice versa.</p>
<p>Through the analysis of a downstream mRNA dataset from RNA-Seq after miR-124-3p knockdown, we demonstrate how the original GSEA method was extended and the results from GSEA were refined, which could be used as a comprehensive protocol for downstream analysis in the loss- and gain-intervention to a specific miRNA. Some parameters, such as the leading-edge subset, were modified to better describe the characteristics of the bottom (underexpressed) mRNAs based on the classical GSEA approach by <xref ref-type="bibr" rid="B18">Subramanian et al. (2005)</xref>. The union of the leading-edge subsets in the enriched KEGG pathways was selected. For traditional GSEA, the number of generated candidate genes is usually still too high. One or several key genes or pathways need to be identified for further functional experiments. Apriori algorithm and a new statistic <inline-formula id="inf5">
<mml:math id="m5">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> were introduced to correct this issue. It is hypothesized that genes play critical roles in the entire leading-edge subsets if they have the most interactions with other genes. Apriori algorithm could use prior Boolean association rules (gene-gene interaction) to mine these pivotal genes or gene sets. Afterward, the expression vector of candidate genes or gene sets were modified with their relationship as a new statistic <inline-formula id="inf6">
<mml:math id="m6">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula>. Finally, the candidate genes were refined, and key genes were identified.</p>
</sec>
<sec id="s2">
<title>2 Methods</title>
<sec id="s2-1">
<title>2.1 Knockdown of miR-124-3p and RNA Extraction</title>
<p>C57BL/6J mice were used to study the transcriptome of miR-124-3p during retinal development. Mice at postnatal day 3 (P3) were given 1&#xa0;&#x3bc;l of 0.6&#xa0;nmol/&#x3bc;l miR-124-3p antagomir in the left eye as the anti-miR-124 group and 0.6&#xa0;nmol/&#x3bc;l antagomir negative control in the right eye as the negative control (NC) group by intravitreal injection. Retinas from mice at P9 were harvested, and total RNA was isolated by TRIzol (Invitrogen; Thermo Fisher Scientific, Inc, Waltham, MA, United States) according to the manufacturer&#x2019;s instructions. Both groups consisted of 12&#x2013;15 mixed retina tissues, and one biological replicate was conducted. In addition, to increase the heterogeneity of the sample, the samples in each group were from at least two different litters.</p>
</sec>
<sec id="s2-2">
<title>2.2. RT&#x2013;qPCR Validation for the Knockdown of miR-124-3p and RNA-Seq</title>
<p>Reverse transcription quantitative PCR (RT&#x2013;qPCR) was performed to validate the knockdown rate of the anti-miR-124 group compared with the NC group. Total RNA from both groups was reverse transcribed using a PrimeScript RT reagent kit (Takara Bio, Inc, Otsu, Japan). Real-time PCR was subsequently performed on the resulting cDNA template with a TB Green&#x2122; Premix Ex Taq&#x2122; II kit (Takara Bio, Inc, Otsu, Japan) on a StepOnePlus&#x2122; Real-Time PCR System (Applied Biosystems; Thermo Fisher Scientific, Inc, Waltham, MA, United States). The 2<sup>&#x2212;&#x394;&#x394;Ct</sup> method was used to quantify miRNA expression levels with the u6 gene as an internal reference (<xref ref-type="bibr" rid="B14">Livak and Schmittgen, 2001</xref>). After significant knockdown of the expression level of miR-124-3p was confirmed by RT&#x2013;qPCR, RNA-Seq was carried out to detect the expression levels of mRNAs in both groups. The RNA-Seq data in the present study are deposited in the Gene Expression Omnibus (GEO) repository, accession number GSE200915 (<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE200915">https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc&#x3d;GSE200915</ext-link>).</p>
</sec>
<sec id="s2-3">
<title>2.3 GSEA</title>
<p>GSEA was carried out using Python 3.6 (Python Software Foundation. Python Language Reference, version 3.6, available at <ext-link ext-link-type="uri" xlink:href="https://www.python.org">https://www.python.org</ext-link>).</p>
<sec id="s2-3-1">
<title>2.3.1 Data Preprocessing</title>
<p>To normalize the length of the mRNA sequences and the sequencing depth of a sample, transcripts per million (TPM) were used to assess the expression level of mRNAs. Read counts for all the mRNA transcripts were demonstrated in <xref ref-type="sec" rid="s12">Data Sheet 1</xref>. The method for calculating TPM was described in detail elsewhere (<xref ref-type="bibr" rid="B21">Wagner et al., 2012</xref>; <xref ref-type="bibr" rid="B7">Dillies et al., 2013</xref>). Based on previous research, the frequency distribution of genes with different expression levels has a mode at TPM nearly equal to 0 and a long tail toward higher TPM values (<xref ref-type="bibr" rid="B20">Wagner et al., 2013</xref>; <xref ref-type="bibr" rid="B4">Bush et al., 2018</xref>; <xref ref-type="bibr" rid="B16">Monaco et al., 2019</xref>). Therefore, to determine an appropriate interval to filter genes with very low expression, a base 10 logarithmic scale was used to evaluate the frequency distribution of TPM for the genes in all groups.</p>
</sec>
<sec id="s2-3-2">
<title>2.3.2 Calculating the Expression Difference</title>
<p>In each group set, the expression difference for a gene between the two groups was defined as the log2-transformed fold-change value of TPM. To reduce the false positive rate of the results, only genes that had the same expression trend between the two biological replicates were considered reliable. Based on these methods, a list of candidate genes and their corresponding gene expression differences was obtained.</p>
</sec>
<sec id="s2-3-3">
<title>2.3.3 Obtaining the Enrichment Score and Its Related Parameters in Kyoto Encyclopedia of Genes and Genomes Pathways</title>
<p>The candidate genes were ranked from highest to lowest expression level, and the enrichment score (ES) for each Kyoto Encyclopedia of Genes and Genomes (KEGG) pathway was calculated based on the GSEA approach (<xref ref-type="bibr" rid="B18">Subramanian et al., 2005</xref>). KEGG gene annotation was derived from the KEGG database (<ext-link ext-link-type="uri" xlink:href="https://www.genome.jp/kegg-bin/download_htext?htext=ko00001.keg&amp;format=json&amp;filedir=">https://www.genome.jp/kegg-bin/download_htext?htext&#x3d;ko00001.keg&#x26;format&#x3d;json&#x26;filedir&#x3d;</ext-link>). After excluding three types of annotations (&#x201c;09160 Human Diseases&#x201d;, &#x201c;09180 Brite Hierarchies&#x201d;, &#x201c;09,190 Not Included in Pathway or Brite&#x201d;) in KEGG pathways that were not related to retinal development, GSEA was performed on 354 pathways whose expression patterns might have changed in the retina. Afterward, the <italic>p</italic> value for each KEGG pathway was estimated by comparing the absolute value of its maximum ES with randomly generated sets of absolute values of maximum ES. A <italic>p</italic> value &#x3c;0.05 was considered significant or enriched. To balance the accuracy of the estimation and the required computing power, the number of permutations used to generate the comparisons was set to 1,000. For an enriched pathway, a leading-edge subset is a set of genes that contributes to the maximum ES in the ranked list of genes. The subset will be found at the top if the maximum ES is positive (upregulated gene subset) or at the bottom if the maximum ES is negative (downregulated gene subset). Afterward, the union of the leading-edge subsets in the enriched KEGG pathways was selected.</p>
<sec id="s2-3-3-1">
<title>2.3.3.1 Calculating the Maximum ES</title>
<p>Since a total of <inline-formula id="inf7">
<mml:math id="m7">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula> candidate genes were ranked from highest to lowest expression level, we denoted <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2026;</mml:mo>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> as the vector of differences, where each element <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in <inline-formula id="inf10">
<mml:math id="m10">
<mml:mi mathvariant="bold-italic">&#x3b4;</mml:mi>
</mml:math>
</inline-formula> represented the expression difference of a single gene <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> between two groups. Accordingly, a corresponding gene vector <inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:mi mathvariant="bold-italic">g</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2026;</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> was defined, where each element <inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> was the gene name using NCBI Gene Symbol. To calculate the ES in a KEGG pathway, it is necessary to determine whether the gene <inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> belongs to this pathway. We established the following definition: if <inline-formula id="inf15">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> belonged to the annotations of this pathway, <inline-formula id="inf16">
<mml:math id="m16">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>; if <inline-formula id="inf17">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> did not belong to the annotations of this pathway, <inline-formula id="inf18">
<mml:math id="m18">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, and a new vector <inline-formula id="inf19">
<mml:math id="m19">
<mml:mrow>
<mml:mi mathvariant="bold-italic">r</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> was generated, where each element <inline-formula id="inf20">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> consisted of 0 and 1. To facilitate the iteration in the computer program, the calculation of ES was rewritten into the following form according to the method described by Subramanian <italic>et al</italic> (<xref ref-type="bibr" rid="B18">Subramanian et al., 2005</xref>).<disp-formula id="equ1">
<mml:math id="m21">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ2">
<mml:math id="m22">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x7c;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf21">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf22">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x7c;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf23">
<mml:math id="m25">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the value of ES and <inline-formula id="inf24">
<mml:math id="m26">
<mml:mi>p</mml:mi>
</mml:math>
</inline-formula> is a weighing factor with a range of <inline-formula id="inf25">
<mml:math id="m27">
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. A value of <inline-formula id="inf26">
<mml:math id="m28">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> was set as a constant in the study. By the method of mathematical induction, it was easy to prove that <inline-formula id="inf27">
<mml:math id="m29">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>In particular, <inline-formula id="inf28">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> equal to 0 indicated that none of the genes in <inline-formula id="inf29">
<mml:math id="m31">
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:math>
</inline-formula> belonged to the annotations in the KEGG pathway. To prevent a division by zero error (when <inline-formula id="inf30">
<mml:math id="m32">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>), we set any value of <inline-formula id="inf31">
<mml:math id="m33">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> equal to 0 under this condition. Thus, the maximum ES, <inline-formula id="inf32">
<mml:math id="m34">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mrow>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo>&#x7c;</mml:mo>
</mml:mrow>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>&#x7c;</mml:mo>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, was obtained, where <inline-formula id="inf33">
<mml:math id="m35">
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>&#x7c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> was the maximum element in the <inline-formula id="inf34">
<mml:math id="m36">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> function, and <inline-formula id="inf35">
<mml:math id="m37">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represented the signum function to indicate the positive and negative values of <inline-formula id="inf36">
<mml:math id="m38">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
<sec id="s2-3-3-2">
<title>2.3.3.2 Estimating the Significance of the Maximum ES</title>
<p>To estimate the significance of the <inline-formula id="inf37">
<mml:math id="m39">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> generated from <inline-formula id="inf38">
<mml:math id="m40">
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:math>
</inline-formula> in the KEGG pathway, a nominal <italic>p</italic> value was calculated using an empirical phenotype-based permutation method derived from <xref ref-type="sec" rid="s12">Subramanian et al. (2005)</xref>. Given the null hypothesis that the order of the elements in <inline-formula id="inf39">
<mml:math id="m41">
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:math>
</inline-formula> was random, the way to reject the null hypothesis was to calculate the probability of <inline-formula id="inf40">
<mml:math id="m42">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in its randomly generated distribution, which was derived from a set of randomly assigned <inline-formula id="inf41">
<mml:math id="m43">
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:math>
</inline-formula>. When the order of elements in <inline-formula id="inf42">
<mml:math id="m44">
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:math>
</inline-formula> was disrupted and the other parameters remained unchanged, the new corresponding value of <inline-formula id="inf43">
<mml:math id="m45">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> was related only to the random rearrangement of elements in <inline-formula id="inf44">
<mml:math id="m46">
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:math>
</inline-formula>. After repeating the above random rearrangement process <inline-formula id="inf45">
<mml:math id="m47">
<mml:mi>v</mml:mi>
</mml:math>
</inline-formula> times, a series of values of <inline-formula id="inf46">
<mml:math id="m48">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mi>&#x2032;</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> with a total number of <inline-formula id="inf47">
<mml:math id="m49">
<mml:mi>v</mml:mi>
</mml:math>
</inline-formula> were obtained based on the null hypothesis, and the absolute value was taken: (<inline-formula id="inf48">
<mml:math id="m50">
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>e</mml:mi>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>&#x2032;</mml:mi>
</mml:msubsup>
<mml:mo>&#x7c;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>e</mml:mi>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mi>&#x2032;</mml:mi>
</mml:msubsup>
<mml:mo>&#x7c;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>e</mml:mi>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mi>&#x2032;</mml:mi>
</mml:msubsup>
<mml:mo>&#x7c;</mml:mo>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. The empirical cumulative distribution function (CDF) <inline-formula id="inf49">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>F</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> for <inline-formula id="inf50">
<mml:math id="m52">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mi>&#x2032;</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> (derived from the null hypothesis) can be described as follows:<disp-formula id="equ3">
<mml:math id="m53">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>v</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>v</mml:mi>
</mml:munderover>
<mml:mi mathvariant="double-struck">I</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>e</mml:mi>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mi>&#x2032;</mml:mi>
</mml:msubsup>
<mml:mo>&#x7c;</mml:mo>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf51">
<mml:math id="m54">
<mml:mrow>
<mml:mi mathvariant="double-struck">I</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the indicator function of an event and <inline-formula id="inf52">
<mml:math id="m55">
<mml:mi>v</mml:mi>
</mml:math>
</inline-formula> corresponds to the number of permutations in Subramanian <italic>et al</italic> (<xref ref-type="bibr" rid="B18">Subramanian et al., 2005</xref>). According to the empirical CDF, the <italic>p</italic> value for an <inline-formula id="inf53">
<mml:math id="m56">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in a pathway was calculated as follows:<disp-formula id="equ4">
<mml:math id="m57">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x7c;</mml:mo>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
</sec>
<sec id="s2-3-3-3">
<title>2.3.3.3 Obtaining the Leading-Edge Subsets</title>
<p>In an enriched pathway, the leading-edge subset in <inline-formula id="inf54">
<mml:math id="m58">
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:math>
</inline-formula> appears prior to the peak score for a positive <inline-formula id="inf55">
<mml:math id="m59">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and appears subsequent to the peak score for a negative <inline-formula id="inf56">
<mml:math id="m60">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The leading-edge subset <inline-formula id="inf57">
<mml:math id="m61">
<mml:mi mathvariant="bold-italic">L</mml:mi>
</mml:math>
</inline-formula> can be described as follows:<disp-formula id="equ5">
<mml:math id="m62">
<mml:mrow>
<mml:mi mathvariant="bold-italic">L</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mtext>&#x7c;</mml:mtext>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>j</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mtext>&#x7c;</mml:mtext>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>j</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mo>&#x2205;</mml:mo>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf58">
<mml:math id="m63">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula> represents the position number of <inline-formula id="inf59">
<mml:math id="m64">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
</sec>
<sec id="s2-3-4">
<title>2.3.4 Refining the Results From Leading-Edge Subsets</title>
<p>Some genes or gene sets with interactions appear more frequently than others, indicating their pivotal roles in the enriched KEGG pathways. For instance, the interacting gene set Raf1&#x2013;Map2k1&#x2013;Erk plays an important role in a series of pathways, such as the ErbB signaling, FoxO signaling, and Ras signaling pathways. The Apriori algorithm designed for finding frequent item sets was introduced to identify these frequently appearing genes or interacting gene sets (<xref ref-type="bibr" rid="B2">Agrawal and Srikant, 1994</xref>). The goal of the algorithm is to identify genes (including a single gene or multiple genes with interactions) whose frequency of occurrence in the gene interactions is greater than a specified threshold. Prior knowledge of gene&#x2013;gene interactions was determined by the Reactome database (<xref ref-type="bibr" rid="B5">Croft et al., 2011</xref>) (<ext-link ext-link-type="uri" xlink:href="https://reactome.org/download/tools/ReatomeFIs/FIsInGene_122220_with_annotations.txt">https://reactome.org/download/tools/ReatomeFIs/FIsInGene_122220_with_annotations.txt</ext-link>). After the frequent gene sets were calculated by the Apriori algorithm, a new statistic <inline-formula id="inf60">
<mml:math id="m65">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> was defined to comprehensively evaluate the real weights of a frequent gene set. Finally, by sorting the modulus of the statistic <inline-formula id="inf61">
<mml:math id="m66">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula>, the significant genes were obtained for further experiments.</p>
<p>The number of iterations was set to 4, which meant that gene interactions were generated with a maximum item size of 4. As Apriori uses a &#x201c;bottom up&#x201d; approach, genes in the union of the leading-edge subsets were used as the first-level candidates. Afterward, the number of genes in a candidate was increased by one, and the candidates that had an infrequent pattern (defined by the threshold) or were not consistent with the gene interaction knowledge were pruned. According to the above method, frequent candidates were extended one gene at a time. The threshold was set to 3, indicating that a candidate would be removed if its frequency was less than 3 among all transactions. The algorithm terminated when no frequent candidates could be generated, and frequent gene sets were identified by screening. Since the frequent gene sets were calculated based on prior annotation, a new statistic <inline-formula id="inf62">
<mml:math id="m67">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> was defined to comprehensively evaluate the real weights of a frequent gene set. The statistic <inline-formula id="inf63">
<mml:math id="m68">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> of a frequent gene set is equal to the multiplication between the column vector of the expression level and its correlation matrix, and the modulus of the statistic <inline-formula id="inf64">
<mml:math id="m69">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> represents comprehensive expression based on prior annotation and the actual expression level. Finally, by sorting the modulus of the statistic <inline-formula id="inf65">
<mml:math id="m70">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula>, the significant genes were obtained for further experiments.</p>
<sec id="s2-3-4-1">
<title>2.3.4.1 Generating the Correlation Matrixes From a Gene Set</title>
<p>According to prior knowledge in the Reactome database, there are three categories of gene effects on another: upregulation, downregulation, or no effect, where the symbols &#x2018;<inline-formula id="inf66">
<mml:math id="m71">
<mml:mo>&#x2192;</mml:mo>
</mml:math>
</inline-formula>&#x2019; and &#x2018;&#x21e2;&#x2019; represent upregulation and downregulation, respectively. For a vector containing <inline-formula id="inf67">
<mml:math id="m72">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula> genes, its correlation matrix <inline-formula id="inf68">
<mml:math id="m73">
<mml:mi mathvariant="bold">M</mml:mi>
</mml:math>
</inline-formula> was a square matrix of order <inline-formula id="inf69">
<mml:math id="m74">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula>, whose elements were the interaction information among these genes. We defined its element <inline-formula id="inf70">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> as follows:<disp-formula id="equ6">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2192;</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>and</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x22ef;</mml:mo>
<mml:mi mathvariant="normal">&#x3e;</mml:mi>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>and</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>had</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>no</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>effect</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>on</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>and</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf71">
<mml:math id="m77">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula> and <inline-formula id="inf72">
<mml:math id="m78">
<mml:mi>j</mml:mi>
</mml:math>
</inline-formula> represent the index numbers of <inline-formula id="inf73">
<mml:math id="m79">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf74">
<mml:math id="m80">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in the tuple, respectively.</p>
<p>For example, for a 4-dimensional vector of a gene set <inline-formula id="inf75">
<mml:math id="m81">
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;D</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> with the following relationship:<disp-formula id="e1">
<mml:math id="m82">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>&#x2192;</mml:mo>
<mml:mtext>B</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mtext>,&#xa0;</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>&#x22ef;</mml:mo>
<mml:mi mathvariant="normal">&#x3e;</mml:mi>
<mml:mtext>C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mtext>,&#xa0;</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>&#x2192;</mml:mo>
<mml:mtext>D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mtext>,&#xa0;</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>B</mml:mtext>
<mml:mo>&#x22ef;</mml:mo>
<mml:mi mathvariant="normal">&#x3e;</mml:mi>
<mml:mtext>C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mtext>,&#xa0;</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>C</mml:mtext>
<mml:mo>&#x2192;</mml:mo>
<mml:mtext>D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>the correlation matrix was described as follows:<disp-formula id="equ7">
<mml:math id="m83">
<mml:mrow>
<mml:mi mathvariant="bold">M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
</sec>
<sec id="s2-3-4-2">
<title>2.3.4.2 Generating Interactions From the Leading-Edge Subsets</title>
<p>Genes from the union of the leading-edge subsets were used to iteratively generate gene transactions. After obtaining the union of the leading-edge subsets with <inline-formula id="inf76">
<mml:math id="m84">
<mml:mi>m</mml:mi>
</mml:math>
</inline-formula> genes, the vector <inline-formula id="inf77">
<mml:math id="m85">
<mml:mi mathvariant="bold-italic">l</mml:mi>
</mml:math>
</inline-formula> consisting of these genes and the correlation matrix <inline-formula id="inf78">
<mml:math id="m86">
<mml:mi mathvariant="bold">M</mml:mi>
</mml:math>
</inline-formula> reflecting their interactions were created. Elements in <inline-formula id="inf79">
<mml:math id="m87">
<mml:mi mathvariant="bold-italic">l</mml:mi>
</mml:math>
</inline-formula> were considered basic gene interactions in the first iteration. For an iteration vector, a gene would be appended to a gene transaction if it had an interaction (upregulation or downregulation) with the gene at the end of the transaction, and all possible new interactions were used as the basic gene interactions for the next iteration. In the study, gene interactions with a maximum of 4 items were generated.</p>
<p>For example, let the 4-dimensional vector <inline-formula id="inf80">
<mml:math id="m88">
<mml:mrow>
<mml:mi mathvariant="bold-italic">l</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;D</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> be the union of the leading-edge subsets, and let the interactions be as described above in (1). For 4 iterations, the sequences of gene interactions <inline-formula id="inf81">
<mml:math id="m89">
<mml:mrow>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>4</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> were generated as follows:<disp-formula id="e3">
<mml:math id="m90">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>A</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>B</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>C</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>D</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mi mathvariant="bold-italic">&#xa0;</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;B</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mi mathvariant="bold-italic">&#xa0;&#xa0;</mml:mi>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mi mathvariant="bold-italic">&#xa0;&#xa0;</mml:mi>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>4</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-3-4-3">
<title>2.3.4.3 Mining Frequent Gene Sets by the Apriori Algorithm</title>
<p>To detect frequent gene sets from the gene transactions, the value <inline-formula id="inf82">
<mml:math id="m91">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> was used to evaluate the frequency of a gene set. For the vector <inline-formula id="inf83">
<mml:math id="m92">
<mml:mi mathvariant="bold-italic">a</mml:mi>
</mml:math>
</inline-formula> of a gene set, the <inline-formula id="inf84">
<mml:math id="m93">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of <inline-formula id="inf85">
<mml:math id="m94">
<mml:mi mathvariant="bold-italic">a</mml:mi>
</mml:math>
</inline-formula> was defined as the total count of <inline-formula id="inf86">
<mml:math id="m95">
<mml:mi mathvariant="bold-italic">a</mml:mi>
</mml:math>
</inline-formula> in all transactions.</p>
<p>For example, the <inline-formula id="inf87">
<mml:math id="m96">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of <inline-formula id="inf88">
<mml:math id="m97">
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> in interactions <inline-formula id="inf89">
<mml:math id="m98">
<mml:mi mathvariant="bold">I</mml:mi>
</mml:math>
</inline-formula> in (3) was 2 (<inline-formula id="inf90">
<mml:math id="m99">
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> in <inline-formula id="inf91">
<mml:math id="m100">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf92">
<mml:math id="m101">
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> in <inline-formula id="inf93">
<mml:math id="m102">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>).</p>
<p>Given a threshold <inline-formula id="inf94">
<mml:math id="m103">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> for <inline-formula id="inf95">
<mml:math id="m104">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (a threshold <inline-formula id="inf96">
<mml:math id="m105">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> was set in this study), gene interactions <inline-formula id="inf97">
<mml:math id="m106">
<mml:mrow>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2026;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> generated from <inline-formula id="inf98">
<mml:math id="m107">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula> iterations and setting both the frequent gene set and the infrequent gene set, <inline-formula id="inf99">
<mml:math id="m108">
<mml:mi mathvariant="bold">F</mml:mi>
</mml:math>
</inline-formula> and <inline-formula id="inf100">
<mml:math id="m109">
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold">F</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>, to <inline-formula id="inf101">
<mml:math id="m110">
<mml:mo>&#x2205;</mml:mo>
</mml:math>
</inline-formula> initially, the algorithm in the study was described in two steps below:</p>
<p>
<statement content-type="step" id="Step_1">
<label>Step 1</label>
<p>Start with <inline-formula id="inf102">
<mml:math id="m111">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. Traverse all interactions in <inline-formula id="inf103">
<mml:math id="m112">
<mml:mi mathvariant="bold">I</mml:mi>
</mml:math>
</inline-formula>. Append the interactions whose <inline-formula id="inf104">
<mml:math id="m113">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf105">
<mml:math id="m114">
<mml:mi mathvariant="bold">F</mml:mi>
</mml:math>
</inline-formula> and the interactions whose <inline-formula id="inf106">
<mml:math id="m115">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf107">
<mml:math id="m116">
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold">F</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>, respectively. Based on the Apriori principle, if a transaction is found to be infrequent, then all its super interactions are also infrequent. Remove the transaction in <inline-formula id="inf108">
<mml:math id="m117">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold">i</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> (if it exists) when a transaction in <inline-formula id="inf109">
<mml:math id="m118">
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold">F</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> belongs to the transaction in <inline-formula id="inf110">
<mml:math id="m119">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold">i</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</statement>
</p>
<p>
<statement content-type="step" id="Step_2">
<label>Step 2</label>
<p>Repeat Step 1 until no further successful extensions are found. <inline-formula id="inf111">
<mml:math id="m120">
<mml:mi mathvariant="bold">F</mml:mi>
</mml:math>
</inline-formula> is the returned result.</p>
<p>For example, for the gene interactions <inline-formula id="inf112">
<mml:math id="m121">
<mml:mi mathvariant="bold">I</mml:mi>
</mml:math>
</inline-formula> described above in (3) and a given threshold <inline-formula id="inf113">
<mml:math id="m122">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, the process of the Apriori algorithm was as follows:<disp-formula id="equ8">
<mml:math id="m123">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>:</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>A</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>B</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>C</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>D</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>A</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>B</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>C</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:mn>8</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>D</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mi mathvariant="bold-italic">&#xa0;&#xa0;</mml:mi>
<mml:mi mathvariant="bold">F</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>A</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>B</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>C</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>D</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mi mathvariant="bold-italic">&#xa0;&#xa0;</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold">F</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2205;</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>In <inline-formula id="inf114">
<mml:math id="m124">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, no interactions in <inline-formula id="inf115">
<mml:math id="m125">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> were removed.<disp-formula id="equ9">
<mml:math id="m126">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>:</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;B</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;B</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mi mathvariant="bold-italic">&#xa0;&#xa0;&#xa0;</mml:mi>
<mml:mi mathvariant="bold">F</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>A</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>B</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>C</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>D</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>B</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mi mathvariant="bold-italic">&#xa0;&#xa0;</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold">F</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>In <inline-formula id="inf116">
<mml:math id="m127">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf117">
<mml:math id="m128">
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf118">
<mml:math id="m129">
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> were super interactions in <inline-formula id="inf119">
<mml:math id="m130">
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold">F</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> and were removed in <inline-formula id="inf120">
<mml:math id="m131">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.<disp-formula id="equ10">
<mml:math id="m132">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>3</mml:mn>
<mml:mo>:</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mi mathvariant="bold-italic">&#xa0;&#xa0;</mml:mi>
<mml:mi mathvariant="bold">&#xa0;F</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>A</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>B</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>C</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>D</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>B</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mi mathvariant="bold-italic">&#xa0;</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ11">
<mml:math id="m133">
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#xa0;</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold">F</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>C</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>In <inline-formula id="inf121">
<mml:math id="m134">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf122">
<mml:math id="m135">
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;D</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> was a super transaction in <inline-formula id="inf123">
<mml:math id="m136">
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold">F</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> and was removed in <inline-formula id="inf124">
<mml:math id="m137">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">I</mml:mi>
<mml:mn>4</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Iteration was terminated, and the frequent gene set <inline-formula id="inf125">
<mml:math id="m138">
<mml:mi mathvariant="bold">F</mml:mi>
</mml:math>
</inline-formula> was obtained.</p>
</statement>
</p>
</sec>
<sec id="s2-3-4-4">
<title>2.3.4.4 Calculating the Weights of the Frequent Gene Set</title>
<p>After the frequent gene set was obtained by the Apriori algorithm, a new statistic <inline-formula id="inf126">
<mml:math id="m139">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> was introduced to evaluate the real &#x2018;weights&#x2019; of the frequent gene set consisting of expression levels obtained from RNA-Seq. Given an element vector <inline-formula id="inf127">
<mml:math id="m140">
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:math>
</inline-formula> from the frequent gene set, <inline-formula id="inf128">
<mml:math id="m141">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> was defined as the multiplication of its expression vector <inline-formula id="inf129">
<mml:math id="m142">
<mml:mi mathvariant="bold-italic">&#x3b4;</mml:mi>
</mml:math>
</inline-formula> on its correlation matrix <inline-formula id="inf130">
<mml:math id="m143">
<mml:mi mathvariant="bold">M</mml:mi>
</mml:math>
</inline-formula>:<disp-formula id="equ12">
<mml:math id="m144">
<mml:mrow>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="bold-italic">&#x3b4;</mml:mi>
<mml:mi mathvariant="bold">M</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>For example, given an element vector <inline-formula id="inf131">
<mml:math id="m145">
<mml:mrow>
<mml:mi mathvariant="bold-italic">g</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>B</mml:mtext>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> from the frequent gene set <inline-formula id="inf132">
<mml:math id="m146">
<mml:mi mathvariant="bold">F</mml:mi>
</mml:math>
</inline-formula>, its expression vector <inline-formula id="inf133">
<mml:math id="m147">
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and correlation matrix <inline-formula id="inf134">
<mml:math id="m148">
<mml:mrow>
<mml:mi mathvariant="bold">M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> [derived from the relationship (A <inline-formula id="inf135">
<mml:math id="m149">
<mml:mo>&#x2192;</mml:mo>
</mml:math>
</inline-formula> B)] were:<disp-formula id="equ13">
<mml:math id="m150">
<mml:mrow>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mi>a</mml:mi>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>The modulus of <inline-formula id="inf136">
<mml:math id="m151">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> should be:<disp-formula id="equ14">
<mml:math id="m152">
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msup>
<mml:mi>a</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Since gene <inline-formula id="inf137">
<mml:math id="m153">
<mml:mtext>B</mml:mtext>
</mml:math>
</inline-formula> was upregulated by gene <inline-formula id="inf138">
<mml:math id="m154">
<mml:mtext>A</mml:mtext>
</mml:math>
</inline-formula> according to the prior knowledge, theoretically, if the expression value <inline-formula id="inf139">
<mml:math id="m155">
<mml:mi>a</mml:mi>
</mml:math>
</inline-formula> of gene <inline-formula id="inf140">
<mml:math id="m156">
<mml:mtext>A</mml:mtext>
</mml:math>
</inline-formula> was increased <inline-formula id="inf141">
<mml:math id="m157">
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, its expression value <inline-formula id="inf142">
<mml:math id="m158">
<mml:mi>b</mml:mi>
</mml:math>
</inline-formula> should also be increased <inline-formula id="inf143">
<mml:math id="m159">
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. However, under the same condition where the expression value <inline-formula id="inf144">
<mml:math id="m160">
<mml:mi>a</mml:mi>
</mml:math>
</inline-formula> was increased <inline-formula id="inf145">
<mml:math id="m161">
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, if the expression value <inline-formula id="inf146">
<mml:math id="m162">
<mml:mi>b</mml:mi>
</mml:math>
</inline-formula> was decreased <inline-formula id="inf147">
<mml:math id="m163">
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, or if the increase in the expression value <inline-formula id="inf148">
<mml:math id="m164">
<mml:mi>b</mml:mi>
</mml:math>
</inline-formula> was not as large as in the first case, <inline-formula id="inf149">
<mml:math id="m165">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula> would be smaller in the second case than in the first case, indicating that the weights in the second case were smaller than those in the first case. Specifically, for a single gene <inline-formula id="inf150">
<mml:math id="m166">
<mml:mrow>
<mml:mtext>&#xa0;S</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf151">
<mml:math id="m167">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula> is the absolute value of its expression level:<disp-formula id="equ15">
<mml:math id="m168">
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>&#x7c;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Any frequent gene set could be described by <inline-formula id="inf152">
<mml:math id="m169">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> no matter how complicated the interactions among genes, and its <inline-formula id="inf153">
<mml:math id="m170">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula> represented the comprehensive weights related to their interactions and expression levels.</p>
<p>For instance, given the 4-dimensional vector of gene set <inline-formula id="inf154">
<mml:math id="m171">
<mml:mrow>
<mml:mi mathvariant="bold-italic">l</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;D</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, its expression vector <inline-formula id="inf155">
<mml:math id="m172">
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and its correlation matrix <inline-formula id="inf156">
<mml:math id="m173">
<mml:mi mathvariant="bold">M</mml:mi>
</mml:math>
</inline-formula> given above, its comprehensive weights were described as follows:<disp-formula id="equ16">
<mml:math id="m174">
<mml:mrow>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mi>a</mml:mi>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ17">
<mml:math id="m175">
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msup>
<mml:mi>a</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>According to this method, the weights associated with any frequent gene set will be referred to as its value of <inline-formula id="inf157">
<mml:math id="m176">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula>, and the frequent gene set with the highest weights could then be identified by ranking the value of <inline-formula id="inf158">
<mml:math id="m177">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula>. Since the value of <inline-formula id="inf159">
<mml:math id="m178">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula> will increase with the increase in the number of items, the value of <inline-formula id="inf160">
<mml:math id="m179">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula> must be compared among frequent gene sets with the same number of genes.</p>
</sec>
</sec>
</sec>
<sec id="s2-4">
<title>2.4 RT-qPCR Validation for the Candidate Genes</title>
<p>RT-qPCR was performed to validate the candidate genes. Total RNA from retinas was isolated and procedures for reverse transcription and Real-Time PCR were described previously earlier. Gene expression levels were quantified using the 2<sup>&#x2212;&#x394;&#x394;Ct</sup> method and normalized to GAPDH levels (<xref ref-type="bibr" rid="B14">Livak and Schmittgen, 2001</xref>). Graphical representation of the results was performed using GraphPad Prism v8.3.0 (GraphPad Software, Inc.).</p>
</sec>
</sec>
<sec id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Data Preparation and Normalization</title>
<p>A total of 115,140 assembled transcript sequences were obtained from the clean data, among which 75,959 were distinct mRNA sequences. After TPM normalization, the expression levels of genes were observed on a 10-base logarithmic scale. The overall TPM values showed a one-tailed distribution, which was consistent with previous literature reports (<xref ref-type="bibr" rid="B20">Wagner et al., 2013</xref>; <xref ref-type="bibr" rid="B4">Bush et al., 2018</xref>; <xref ref-type="bibr" rid="B16">Monaco et al., 2019</xref>). <xref ref-type="fig" rid="F1">Figure 1</xref> shows that several genes had TPM values of 10<sup>&#x2212;4</sup>, an amount of which was so small that it could not be clearly observed on the histogram. In addition, the TPM distribution on the order of 10<sup>&#x2212;7</sup> was irregular and inconsistent with the one-tailed distribution of the previous orders of magnitude. TPM data below 10<sup>&#x2212;6</sup> were excluded from further analysis. The expression differences of filtered genes were determined based on the log2-transformed fold-change TPM values. After miR-124-3p was knocked down in the mouse retina, most genes in the anti-miR-124 group were upregulated compared to the NC group, while a small portion were downregulated.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The upper panel showed the overall distribution of mRNA TPM values, and the lower panel showed the log2-transformed fold-change in filtered genes.</p>
</caption>
<graphic xlink:href="fgene-13-890672-g001.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 Identified Enriched KEGG Pathways and Their Leading-Edge Subsets</title>
<p>Based on GSEA of 354 pathways, the ranked gene list was enriched at the top in most of the pathways and enriched at the bottom in a small number of pathways. After <italic>p</italic> values were calculated based on 1,000 permutations, pathways related to retinal development with <italic>p</italic> values &#x3c;0.05 were selected (<xref ref-type="fig" rid="F2">Figure 2</xref>). The results indicated that only pathways with a positive maximum ES showed significant enrichment in upregulated genes, consistent with the negative regulation of mRNA by miRNA. The top 10 enriched pathways and the expression levels of their leading-edge subsets are shown in <xref ref-type="fig" rid="F3">Figure 3</xref>: the NOD-like receptor signaling pathway, C-type-like lectin receptor signaling pathway, phagosome, necroptosis, JAK-STAT signaling pathway, Toll-like receptor signaling pathway, leukocyte transendothelial migration, chemokine signaling pathway, NF-kappa B signaling pathway and RIG-I-like signaling pathway.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The left heatmap demonstrated ES values for all the 354 pathways in GSEA. The right heatmap illustrated randomly generated maximum ES values for the corresponding pathways.</p>
</caption>
<graphic xlink:href="fgene-13-890672-g002.tif"/>
</fig>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The left panel indicated the top 10 enriched pathways. The right panel indicated the expression level of the union of their leading-edge subsets.</p>
</caption>
<graphic xlink:href="fgene-13-890672-g003.tif"/>
</fig>
</sec>
<sec id="s3-3">
<title>3.3 The Weights of Frequent Gene Sets From the Leading-Edge Subsets</title>
<p>The Apriori algorithm was performed on the union of the leading-edge subsets to identify the frequent gene set. After the first iteration, 60 single genes were identified as frequent. Based on this, frequent gene sets with 2 and 3 genes were further identified, and no frequent gene sets were found with more than 4 genes. Values of <inline-formula id="inf161">
<mml:math id="m180">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula> for frequent gene sets with 2 and 3 genes were calculated. In <xref ref-type="fig" rid="F4">Figure 4</xref>, each colored line represents an interaction. The wider the height of the color line is, the greater the value of <inline-formula id="inf162">
<mml:math id="m181">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula>. The background color of the gene represents the interactions between the gene and its downstream genes. The darker the background color of the gene is, the higher the number of interactions. Upstream of the frequent interactions with two genes, Prkcd, Igf1, Irf9, Cxcl12, Trim25, Stat2, Stat3 and Stat1 had a considerable <inline-formula id="inf163">
<mml:math id="m182">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula> value and interactions with the downstream genes. For frequent interactions with 3 genes in the first set, genes with a considerable <inline-formula id="inf164">
<mml:math id="m183">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula> value and interactions were Isg15, Prkcd, Eif2ak2, Cxcl12, Il6st, Pdgfra, Socs4, Stat2, Stat3, Csf2ra, Irf9 and Stat1. In summary, Prkcd, Irf9, Stat3, Cxcl12, Stat1 and Stat2 had the greatest interactions and the greatest value of <inline-formula id="inf165">
<mml:math id="m184">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula> to downstream genes among all frequent transactions, indicating that they played a pivotal role in the downstream genes of miR-124-3p. The frequent gene sets and their weights are shown in <xref ref-type="sec" rid="s12">Data Sheet 2</xref>. RT-qPCR was performed to validate the expression of candidate genes. These genes exhibited a similar trend to the RNA-Seq results (<xref ref-type="fig" rid="F5">Figure 5</xref>). Primer sequences were described in <xref ref-type="sec" rid="s12">Data Sheet 3</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>The figure illustrated <inline-formula id="inf166">
<mml:math id="m185">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> values and their interactions in frequent gene sets. The upper panel showed the expression patterns of frequent gene sets with two items, while the lower panel showed the expression patterns of frequent gene sets with 3 items.</p>
</caption>
<graphic xlink:href="fgene-13-890672-g004.tif"/>
</fig>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Results of RT-qPCR showed the expression of candidate genes bewteen NC and anti-124 group, which exhibited a similar trend to the RNA-Seq results. Each RT-qPCR experiment was repeated three times with independently isolated RNA samples. Results were presented as mean &#xb1; standard deviation (anti-miR-124 group, miR-124-3p knockdown group; NC group, negative control group; &#x2a;<italic>p</italic> &#x3c; 0.05, &#x2a;&#x2a;<italic>p</italic> &#x3c; 0.01)</p>
</caption>
<graphic xlink:href="fgene-13-890672-g005.tif"/>
</fig>
</sec>
</sec>
<sec id="s4">
<title>4 Discussion</title>
<p>After gain and loss intervention of the upstream miRNA in a functional experiment, transcriptomics technologies such as RNA-Seq make it possible to obtain a large amount of downstream gene expression data in a single experiment. To handle high-throughput data, although there are many other methods based on the entire gene set analysis, such as CePa (<xref ref-type="bibr" rid="B8">Gu et al., 2012</xref>) and SPIA (<xref ref-type="bibr" rid="B19">Tarca et al., 2009</xref>), GSEA is still one of the most widely used and outperformed methods by focusing on the entire gene set rather than a single threshold (<xref ref-type="bibr" rid="B3">Bayerlov&#xe1; et al., 2015</xref>). Using GSEA, we analyzed the ranked gene list after knockdown of miR-124-3p to determine if they were enriched in given phenotypes, that is, KEGG pathways related to retinal development in our study. In the results, most genes in the anti-miR-124 group were upregulated compared to the NC group, and only pathways in upregulated genes showed significant enrichment, which was consistent with prior theories. A series of KEGG pathways with significant enrichment could be obtained as well as their leading-edge subsets of their associated genes by GSEA. However, for a low-throughput functional experiment, only one or several downstream genes and pathways were selected. The number of generated candidate genes is usually still too high and needs further refinement.</p>
<p>Since not all genes play an equally important role in a pathway, we hypothesized that the genes that have the most interactions with other genes in the entire leading-edge subsets are likely to be the most important. To obtain those genes or gene sets, interactions based on KEGG pathway topology were generated, and the Apriori algorithm was performed. The Apriori algorithm is designed for mining the frequent item set. Although it has exponential time complexity, interactions were generated based on the correlation matrix of <inline-formula id="inf167">
<mml:math id="m186">
<mml:mi mathvariant="bold-italic">a</mml:mi>
</mml:math>
</inline-formula>, and the nonexistent gene interactions (with an element equal to 0) were excluded, which reduced the amount of computation from bottom to top.</p>
<p>This greatly decreased the calculation time of the Apriori algorithm, making the calculation time acceptable. Moreover, we introduced a new statistic <inline-formula id="inf168">
<mml:math id="m187">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> to evaluate the connection between the prior knowledge and its actual expression. The larger the value of <inline-formula id="inf169">
<mml:math id="m188">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> in a gene or gene set, the better the agreement between the KEGG pathway topology and its actual expression. Accordingly, we further refined the result of leading-edge subsets.</p>
<p>The statistic <inline-formula id="inf170">
<mml:math id="m189">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> was a corrected expression vector, considering the relationship and gene expressions. For genes having interactions, if the expression levels did not conform to the prior interactions among them, the value of <inline-formula id="inf171">
<mml:math id="m190">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula> might still be small even if the expression levels of these genes were large. This made the modulus of <inline-formula id="inf172">
<mml:math id="m191">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula> a good representative of the comprehensive value of the gene set. The comparison of relative biological importance among gene sets could be achieved by comparing their <inline-formula id="inf173">
<mml:math id="m192">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula> values. By sorting the value of <inline-formula id="inf174">
<mml:math id="m193">
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:math>
</inline-formula>, critical genes or gene sets were identified.</p>
</sec>
<sec id="s5">
<title>5 Conclusion</title>
<p>In this study, through mRNA sequencing after miR-124-3p knockdown, GSEA was performed to identify significant KEGG pathways and their leading-edge subsets. We demonstrated that using the Apriori algorithm and defining the statistic <inline-formula id="inf175">
<mml:math id="m194">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula> was a novel way to refine the GSEA results (<xref ref-type="fig" rid="F6">Figure 6</xref>). The actual expression level and its correlation of a gene set was combined and represented by a single number, the modulus of <inline-formula id="inf176">
<mml:math id="m195">
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:math>
</inline-formula>. According to this information, a series of key downstream genes and gene sets were identified.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>The flowchart of the proposed method.</p>
</caption>
<graphic xlink:href="fgene-13-890672-g006.tif"/>
</fig>
</sec>
</body>
<back>
<sec id="s6">
<title>Data Availability Statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found in the article/<xref ref-type="sec" rid="s12">Supplementary Material</xref>.</p>
</sec>
<sec id="s7">
<title>Ethics Statement</title>
<p>The animal study was reviewed and approved by The Ethics Committee of the Sun Yat-Sen University Zhongshan Ophthalmic Center.</p>
</sec>
<sec id="s8">
<title>Author Contributions</title>
<p>YW wrote the code, performed the bioinformatics analysis and contributed to writing manuscript. YH, SM, and YC contributed to the experiments <italic>in vivo</italic>. YJ and JP assisted with the bioinformatics analysis. YL supervised the study. All authors read and approved the final manuscript.</p>
</sec>
<sec id="s9">
<title>Funding</title>
<p>This work was supported by the National Natural Science Foundation of China to YL (81770971); and the Natural Science Foundation of Guangdong Province, China to YL (2020A1515010617).</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2022.890672/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2022.890672/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet3.xlsx" id="SM1" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="DataSheet1.CSV" id="SM2" mimetype="application/CSV" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="DataSheet2.xlsx" id="SM3" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ackermann</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Strimmer</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>A General Modular Framework for Gene Set Enrichment Analysis</article-title>. <source>BMC Bioinforma.</source> <volume>10</volume>, <fpage>47</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-10-47</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Agrawal</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Srikant</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>1994</year>). &#x201c;<article-title>Fast Algorithms for Mining Association Rules</article-title>,&#x201d; in <conf-name>Proceedings of the 20th International Conference on Very Large Data Bases</conf-name> (<publisher-loc>United States</publisher-loc>: <publisher-name>Morgan Kaufmann Publishers Inc.</publisher-name>), <fpage>487</fpage>&#x2013;<lpage>499</lpage>. </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bayerlov&#xe1;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jung</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kramer</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Klemm</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Bleckmann</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bei&#xdf;barth</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Comparative Study on Gene Set and Pathway Topology-Based Enrichment Methods</article-title>. <source>BMC Bioinforma.</source> <volume>16</volume>, <fpage>334</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-015-0751-5</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bush</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Freem</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>MacCallum</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>O&#x2019;Dell</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Afrasiabi</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Combination of Novel and Public RNA-Seq Datasets to Generate an mRNA Expression Atlas for the Domestic Chicken</article-title>. <source>BMC genomics</source> <volume>19</volume>, <fpage>594</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-018-4972-7</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Croft</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>O&#x27;Kelly</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Haw</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gillespie</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Matthews</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Reactome: a Database of Reactions, Pathways and Biological Processes</article-title>. <source>Nucleic acids Res.</source> <volume>39</volume>, <fpage>D691</fpage>&#x2013;<lpage>D697</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkq1018</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Damiani</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Alexander</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>O&#x27;Rourke</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>McManus</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jadhav</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Cepko</surname>
<given-names>C. L.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Dicer Inactivation Leads to Progressive Functional and Structural Degeneration of the Mouse Retina</article-title>. <source>J. Neurosci.</source> <volume>28</volume>, <fpage>4878</fpage>&#x2013;<lpage>4887</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.0828-08.2008</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dillies</surname>
<given-names>M.-A.</given-names>
</name>
<name>
<surname>Rau</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Aubert</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hennequet-Antier</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Jeanmougin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Servant</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>A Comprehensive Evaluation of Normalization Methods for Illumina High-Throughput RNA Sequencing Data Analysis</article-title>. <source>Briefings Bioinforma.</source> <volume>14</volume>, <fpage>671</fpage>&#x2013;<lpage>683</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbs046</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Centrality-based Pathway Enrichment: a Systematic Approach for Finding Significant Pathways Dominated by Key Genes</article-title>. <source>BMC Syst. Biol.</source> <volume>6</volume>, <fpage>56</fpage>. <pub-id pub-id-type="doi">10.1186/1752-0509-6-56</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ha</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>V. N.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Regulation of microRNA Biogenesis</article-title>. <source>Nat. Rev. Mol. Cell Biol.</source> <volume>15</volume>, <fpage>509</fpage>&#x2013;<lpage>524</lpage>. <pub-id pub-id-type="doi">10.1038/nrm3838</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hackler</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Swaroop</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Qian</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zack</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>MicroRNA Profile of the Developing Mouse Retina</article-title>. <source>Invest. Ophthalmol. Vis. Sci.</source> <volume>51</volume>, <fpage>1823</fpage>. <pub-id pub-id-type="doi">10.1167/iovs.09-4657</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karali</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Peluso</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Gennarino</surname>
<given-names>V. A.</given-names>
</name>
<name>
<surname>Bilio</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Verde</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Lago</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>miRNeye: a microRNA Expression Atlas of the Mouse Eye</article-title>. <source>BMC genomics</source> <volume>11</volume>, <fpage>715</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-11-715</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karali</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Peluso</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Marigo</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Banfi</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Identification and Characterization of microRNAs Expressed in the Mouse Eye</article-title>. <source>Invest. Ophthalmol. Vis. Sci.</source> <volume>48</volume>, <fpage>509</fpage>&#x2013;<lpage>515</lpage>. <pub-id pub-id-type="doi">10.1167/iovs.06-0866</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karali</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Persico</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mutarelli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Carissimo</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pizzo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Singh Marwah</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>High-resolution Analysis of the Human Retina miRNome Reveals isomiR Variations and Novel microRNAs</article-title>. <source>Nucleic Acids Res.</source> <volume>44</volume>, <fpage>1525</fpage>&#x2013;<lpage>1540</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw039</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Livak</surname>
<given-names>K. J.</given-names>
</name>
<name>
<surname>Schmittgen</surname>
<given-names>T. D.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Analysis of Relative Gene Expression Data Using Real-Time Quantitative PCR and the 2&#x2212;&#x394;&#x394;CT Method</article-title>. <source>Methods</source> <volume>25</volume>, <fpage>402</fpage>&#x2013;<lpage>408</lpage>. <pub-id pub-id-type="doi">10.1006/meth.2001.1262</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maleki</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ovens</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hogan</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Kusalik</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Gene Set Analysis: Challenges, Opportunities, and Future Research</article-title>. <source>Front. Genet.</source> <volume>11</volume>. <pub-id pub-id-type="doi">10.3389/fgene.2020.00654</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Monaco</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Mustafah</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hwang</surname>
<given-names>Y. Y.</given-names>
</name>
<name>
<surname>Carr&#xe9;</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>RNA-seq Signatures Normalized by mRNA Abundance Allow Absolute Deconvolution of Human Immune Cell Types</article-title>. <source>Cell Rep.</source> <volume>26</volume>, <fpage>1627</fpage>&#x2013;<lpage>1640</lpage>. <comment>e1627</comment>. <pub-id pub-id-type="doi">10.1016/j.celrep.2019.01.041</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ritchie</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Rajasekhar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Flamant</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rasko</surname>
<given-names>J. E. J.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Conserved Expression Patterns Predict microRNA Targets</article-title>. <source>PLoS Comput. Biol.</source> <volume>5</volume>, <fpage>e1000513</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1000513</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Subramanian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tamayo</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mootha</surname>
<given-names>V. K.</given-names>
</name>
<name>
<surname>Mukherjee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ebert</surname>
<given-names>B. L.</given-names>
</name>
<name>
<surname>Gillette</surname>
<given-names>M. A.</given-names>
</name>
<etal/>
</person-group> (<year>2005</year>). <article-title>Gene Set Enrichment Analysis: A Knowledge-Based Approach for Interpreting Genome-wide Expression Profiles</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A.</source> <volume>102</volume>, <fpage>15545</fpage>&#x2013;<lpage>15550</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.0506580102</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tarca</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Draghici</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Khatri</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Hassan</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Mittal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J.-s.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>A Novel Signaling Pathway Impact Analysis</article-title>. <source>Bioinforma. Oxf. Engl.</source> <volume>25</volume>, <fpage>75</fpage>&#x2013;<lpage>82</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btn577</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wagner</surname>
<given-names>G. P.</given-names>
</name>
<name>
<surname>Kin</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lynch</surname>
<given-names>V. J.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>A Model Based Criterion for Gene Expression Calls Using RNA-Seq Data</article-title>. <source>Theory Biosci.</source> <volume>132</volume>, <fpage>159</fpage>&#x2013;<lpage>164</lpage>. <pub-id pub-id-type="doi">10.1007/s12064-013-0178-3</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wagner</surname>
<given-names>G. P.</given-names>
</name>
<name>
<surname>Kin</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lynch</surname>
<given-names>V. J.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Measurement of mRNA Abundance Using RNA-Seq Data: RPKM Measure Is Inconsistent Among Samples</article-title>. <source>Theory Biosci.</source> <volume>131</volume>, <fpage>281</fpage>&#x2013;<lpage>285</lpage>. <pub-id pub-id-type="doi">10.1007/s12064-012-0162-3</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.-P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>K.-B.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Correlation of Expression Profiles between microRNAs and mRNA Targets Using NCI-60 Data</article-title>. <source>BMC genomics</source> <volume>10</volume>, <fpage>218</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-10-218</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Identification of Key miRNAs and Genes for Mouse Retinal Development Using a Linear Model</article-title>. <source>Mol. Med. Rep.</source> <volume>22</volume>, <fpage>494</fpage>&#x2013;<lpage>506</lpage>. <pub-id pub-id-type="doi">10.3892/mmr.2020.11082</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gerstein</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Snyder</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>RNA-seq: a Revolutionary Tool for Transcriptomics</article-title>. <source>Nat. Rev. Genet.</source> <volume>10</volume>, <fpage>57</fpage>&#x2013;<lpage>63</lpage>. <pub-id pub-id-type="doi">10.1038/nrg2484</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>