<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1511456</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2024.1511456</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Correlating gene expression levels with transcription factor binding sites facilitates identification of key transcription factors from transcriptome data</article-title>
<alt-title alt-title-type="left-running-head">Huang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2024.1511456">10.3389/fgene.2024.1511456</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Huang</surname>
<given-names>Tinghua</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2852442/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Niu</surname>
<given-names>Siqi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Fanghong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Binyu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Jianwu</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Liu</surname>
<given-names>Guoping</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1834894/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Yao</surname>
<given-names>Min</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2868994/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>College of Animal Science and Technology</institution>, <institution>Yangtze University</institution>, <addr-line>Jingzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>College of Agriculture</institution>, <institution>Yangtze University</institution>, <addr-line>Jingzhou</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/301928/overview">Gajendra P. S. Raghava</ext-link>, Indraprastha Institute of Information Technology Delhi, India</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1670944/overview">Rajesh Kumar</ext-link>, National Institutes of Health (NIH), United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/937849/overview">Anjali Lathwal</ext-link>, Indira Gandhi Delhi Technical University for Women, India</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Min Yao, <email>minyao@yangtzeu.edu.cn</email>; Guoping Liu, <email>guoping.liu@yangtzeu.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>29</day>
<month>11</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1511456</elocation-id>
<history>
<date date-type="received">
<day>15</day>
<month>10</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>11</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Huang, Niu, Zhang, Wang, Wang, Liu and Yao.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Huang, Niu, Zhang, Wang, Wang, Liu and Yao</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Identification of key transcription factors from transcriptome data by correlating gene expression levels with transcription factor binding sites is important for transcriptome data analysis. In a typical scenario, we always set a threshold to filter the top ranked differentially expressed genes and top ranked transcription factor binding sites. However, correlation analysis of filtered data can often result in spurious correlations. In this study, we tested four methods for creating the gene expression inputs (ranked gene list) in the correlation analysis: star coordinate map transformation (START), expression differential score (ED), preferential expression measure (PEM), and the specificity measure (SPM). Then, Kendall&#x2019;s tau correlation statistical algorithms implementing the standard (STD), LINEAR, MIX-LINEAR, DENSITY-CURVE, and MIXED-DENSITY-CURVE weighting methods were used to identify key transcription factors. ED was identified as the optimal method for creating a ranked gene list from filtered expression data, which can address the &#x201c;unable to detect negative correlation&#x201d; fallacy presented by other methods. The MIXED-DENSITY-CURVE was the most sensitive for identifying transcription factors from the gene set and list in which only the top proportion was correlated. Ultimately, 644 transcription factor candidates were identified from the transcriptome data of 1,206 cell lines, six of which were validated by wet lab experiments. The Jinzer and Flaver software implementing these methods can be obtained from <ext-link ext-link-type="uri" xlink:href="http://www.thua45/cn/flaver">http://www.thua45/cn/flaver</ext-link> under a free academic license.</p>
</abstract>
<kwd-group>
<kwd>transcription factor</kwd>
<kwd>transcriptome data</kwd>
<kwd>correlation analysis</kwd>
<kwd>Kendall&#x2019;s tau</kwd>
<kwd>Jinzer</kwd>
<kwd>Flaver</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>We started the analysis by creating the gene expression inputs (ranked gene list) for the significance of differential expression in a specific cell line or tissue sample. Identifying cell-specific expressed genes and the key transcription factors (TF) that regulate these genes from transcriptome data continue to pose challenges. At present, there are five representative state-of-the-art indexes for estimating and identifying the significance of sample-specific gene expression. Let x<sub>i</sub> be the expression level of gene x in sample i, s<sub>i</sub> summarize the expression levels of all genes in sample i, and n be the number of samples, which are calculated as follows:</p>
<p>The sample specificity expressed index &#x3c4; defined as <xref ref-type="disp-formula" rid="e1">formula 1</xref> was proposed by <xref ref-type="bibr" rid="B33">Yanai et al. (2005)</xref>.<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3c4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:munder>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>As presented by <xref ref-type="bibr" rid="B36">Yu et al. (2006)</xref>, the extent of tissue or cell line-specific expression is defined for each gene as EE (expression enrichment), and was calculated according to <xref ref-type="disp-formula" rid="e2">formula 2</xref>.<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:mtext>EE</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x200b;</mml:mo>
<mml:mo>&#x2a;</mml:mo>
<mml:mo>&#x200b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>The H<sub>g</sub> score measures the sample-specificity with entropy (<xref ref-type="bibr" rid="B26">Schug et al., 2005</xref>), calculated as shown in <xref ref-type="disp-formula" rid="e3">formula 3</xref>.<disp-formula id="e3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mi mathvariant="normal">g</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mfrac>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x200b;</mml:mo>
<mml:mo>&#x2a;</mml:mo>
<mml:mo>&#x200b;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>log</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mfrac>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>The specificity measure (SPM) from the TiSGeD database (<xref ref-type="bibr" rid="B32">Xiao et al., 2010</xref>) was calculated using <xref ref-type="disp-formula" rid="e4">formula 4</xref>.<disp-formula id="e4">
<mml:math id="m4">
<mml:mrow>
<mml:mtext>SPM</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msubsup>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>The preferential expression measure (PEM), which estimates how different the expression of the gene is relative to an expected expression (<xref ref-type="bibr" rid="B14">Huminiecki et al., 2003</xref>), can be calculated using <xref ref-type="disp-formula" rid="e5">formula 5</xref>.<disp-formula id="e5">
<mml:math id="m5">
<mml:mrow>
<mml:mtext>PEM</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>log</mml:mi>
<mml:mn>10</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x200b;</mml:mo>
<mml:mo>&#x2a;</mml:mo>
<mml:mo>&#x200b;</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>In a previous study, Kryuchkova-Mostacci et al. conducted a detailed evaluation and review of these indexes (<xref ref-type="bibr" rid="B17">Kryuchkova-Mostacci and Robinson-Rechavi, 2017</xref>). All these indexes focus on identifying genes that are specifically expressed in a certain sample or cell line, particularly those expressed dominantly.</p>
<p>A variety of factors including transcription factor binding, DNA accessibility changes, and sequence specificity determined the transcriptional levels of genes. Current transcription factor mining methods test the significance of a set of genes regulated by a transcription factor (gene set) in a list of differentially expressed genes (gene list) identified by gene expression profiling. Representative software includes GOStats (<xref ref-type="bibr" rid="B6">Falcon and Gentleman, 2007</xref>), ClusterProfiler (<xref ref-type="bibr" rid="B35">Yu et al., 2012</xref>), GSEA (<xref ref-type="bibr" rid="B28">Subramanian et al., 2005</xref>), WebGestalt (<xref ref-type="bibr" rid="B31">Wang et al., 2017</xref>), and Flaver (<xref ref-type="bibr" rid="B34">Yao et al., 2024</xref>), among others (<xref ref-type="bibr" rid="B21">Maleki et al., 2020</xref>). GOStats software and its alternatives implement Fisher&#x2019;s exact test (<xref ref-type="bibr" rid="B34">Yao et al., 2024</xref>; <xref ref-type="bibr" rid="B16">Keenan et al., 2019</xref>; <xref ref-type="bibr" rid="B20">Magnusson and Lubovac-Pilav, 2021</xref>; <xref ref-type="bibr" rid="B18">Lachmann et al., 2010</xref>), GSEA software implements the Kolmogorov Smirnov statistics (<xref ref-type="bibr" rid="B28">Subramanian et al., 2005</xref>; <xref ref-type="bibr" rid="B21">Maleki et al., 2020</xref>), and ClusterProfiler and WebGestalt implement both statistics (<xref ref-type="bibr" rid="B35">Yu et al., 2012</xref>; <xref ref-type="bibr" rid="B31">Wang et al., 2017</xref>). Among these, GOStats (<xref ref-type="bibr" rid="B6">Falcon and Gentleman, 2007</xref>) uses gene sets and gene lists without rank attributes, whereas ClusterProfiler and GSEA software use gene sets without rank attributes and gene lists with rank attributes. The HOMER (<xref ref-type="bibr" rid="B9">Heinz et al., 2010</xref>) and MEME (<xref ref-type="bibr" rid="B4">Bailey et al., 2009</xref>) software implemented comprehensive novel motif discovery and motif scanning algorithms that were designed for regulatory element analysis in genomics applications. ARACNE (<xref ref-type="bibr" rid="B22">Margolin et al., 2006</xref>) was an algorithm for the reconstruction of gene regulatory networks in a mammalian cellular context, and its partner VIPER (<xref ref-type="bibr" rid="B2">Alvarez et al., 2016</xref>) which implemented the aREA algorithm, testing for a global shift in the positions of each regulon genes when projected on the rank-sorted gene expression signature which is similar to GSEA&#x2019;s concepts. The recently developed Flaver software (<xref ref-type="bibr" rid="B34">Yao et al., 2024</xref>) employs weighted Kendall&#x2019;s tau rank correlation statistics, which support both ranked gene lists and ranked gene set inputs. The rank attributes of the genes in the gene list can be a measure of the degree of gene expression difference. The rank attributes of genes in the gene set may be the true or false possibilities for transcription factor binding sites.</p>
<p>In this study, a two-step pipeline was developed to analyze transcriptome data. A ranked gene list was created using four different methods: star coordinate map transformation (START), expression differential score (ED), preferential expression measure (PEM), and the specificity measure (SPM). TF-derived gene sets with rank information were created using the Jinzer software with updated promoter sequences and algorithms. Flaver software was developed to identify significant correlations between ranked gene sets and ranked gene lists. Finally, we applied the pipeline to transcriptome data from 1,206 human cell lines and obtained promising results, demonstrating its desirability as a gene set discovery tool.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Creation of a ranked gene list from transcriptome data and four indexes</title>
<p>The transcriptome data used in this study were obtained from the Protein Atlas database (<xref ref-type="bibr" rid="B29">Uhl&#xe9;n et al., 2015</xref>), including 1,206 human cell lines (quantile normalized). For the calculation of the four indexes, the SPM and PEM scores were specified using <xref ref-type="disp-formula" rid="e4">formula 4</xref>, <xref ref-type="disp-formula" rid="e5">5</xref> without modification. The index &#x3c4; and H<sub>g</sub> score only estimated the overall sample-specific expression per gene. The EE score was the same as the inverse logarithm form of PEM. These three indexes were not included in this study. The expression differential (ED) score was calculated using <xref ref-type="disp-formula" rid="e6">formula 6</xref>. The star coordinate transformation method is as follows:<disp-formula id="e6">
<mml:math id="m6">
<mml:mrow>
<mml:mtext>ED</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x200b;</mml:mo>
<mml:mo>&#x2a;</mml:mo>
<mml:mo>&#x200b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>The star coordinate (START) method is based on parallel coordinates (<xref ref-type="bibr" rid="B8">Heinrich and Weiskopf, 2009</xref>) and star coordinates (<xref ref-type="bibr" rid="B15">Kandogan, 2000</xref>). This study adopted a similar strategy: first, gene expression data were mapped to a two-dimensional space, and then the degree of differential gene expression was calculated (<xref ref-type="sec" rid="s12">Supplementary Figure S1</xref>). The arrangement order of each coordinate axis was determined by hierarchical clustering method. The START method defines a two-dimensional vector A &#x3d; [A1, A2, A3, &#x2026; , An] that starts at the origin (Ox, Oy) and extends along the coordinate axis. Let the initial value of the data points be D &#x3d; (D1, D2, D3, &#x2026; , Dn) and set to the lengths of the two-dimensional vectors Ai. The Cartesian coordinates of the two-dimensional vector Ai corresponding to Di are (Di &#x2a; cos &#x19f;, Di &#x2a; sin &#x19f;), where &#x19f; is the angle of coordinates. The final Cartesian coordinate value of data point D can be obtained from the sum of Ai, as specified in <xref ref-type="disp-formula" rid="e7">formula 7</xref>:<disp-formula id="e7">
<mml:math id="m7">
<mml:mrow>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">O</mml:mi>
<mml:mi mathvariant="normal">x</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi mathvariant="normal">D</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
<mml:mo>&#x2219;</mml:mo>
<mml:mi>cos</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi mathvariant="normal">&#x3b8;</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">O</mml:mi>
<mml:mi mathvariant="normal">y</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi mathvariant="normal">D</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
<mml:mo>&#x2219;</mml:mo>
<mml:mi>sin</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi mathvariant="normal">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
<p>The distance from point V, which is the vertical intersection point for point P to the coordinate axis of the target sample to the origin site O is a perfect measure of the degree of gene expression difference (Supplemental document <xref ref-type="fig" rid="F1">Figure 1A</xref>). Let &#x3b4; be the angle of OP, the length of VO can be calculated by <xref ref-type="disp-formula" rid="e8">formula 8</xref>:<disp-formula id="e8">
<mml:math id="m8">
<mml:mrow>
<mml:mtext>VO</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mroot>
<mml:mrow>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x200b;</mml:mo>
<mml:mo>&#x2a;</mml:mo>
<mml:mo>&#x200b;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:mroot>
<mml:mo>&#x200b;</mml:mo>
<mml:mo>&#x2a;</mml:mo>
<mml:mo>&#x200b;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>cos</mml:mi>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3b8;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">&#x3b4;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The distribution of the Jindex score for the transcription factor binding sites identified by Jinzer software. Left: A full-spectrum distribution. Right: The partial distribution for Jindex &#x3e;1.0. The plot shows the distributions for all binding sites of 15 randomly selected transcription factors in the human genome.</p>
</caption>
<graphic xlink:href="fgene-15-1511456-g001.tif"/>
</fig>
</sec>
<sec id="s2-2">
<title>2.2 Development of an improved version of transcription factor binding site prediction software</title>
<p>In our previous study, we developed a genome-wide transcription factor binding site prediction software, Grit (<xref ref-type="bibr" rid="B12">Huang et al., 2022</xref>), based on comparative genomics and the mixed Student&#x2019;s t-test. In this study, we revised the calculation method for the raw binding site score (RS), specified in <xref ref-type="disp-formula" rid="e9">formula 9</xref>, <xref ref-type="disp-formula" rid="e10">10</xref>. This revised score was denoted as &#x201c;Jindex&#x201d;. The reason for this revision is that, according to Markov Chain theory, the relationship between the probabilities of individual parts should be multiplicative rather than additive.<disp-formula id="e9">
<mml:math id="m9">
<mml:mrow>
<mml:mtext>RS</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>ln</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
</mml:msub>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x220f;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi mathvariant="normal">w</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">q</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="normal">k</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
<disp-formula id="e10">
<mml:math id="m10">
<mml:mrow>
<mml:mtext>Jindex</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>Max</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="&#x7c;">
<mml:mrow>
<mml:mi>ln</mml:mi>
<mml:mroot>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x220f;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi mathvariant="normal">w</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">q</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="normal">k</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mi>w</mml:mi>
</mml:mroot>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>;</mml:mo>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mn>1</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi mathvariant="normal">l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">w</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
<p>The Jindex calculation represents the maximum of repeated averaging of log likelihood ratios (LLRs). The averaged LLR indicated the possibility for a motif being present at one particular location in a sequence, where <italic>w</italic> was the width of the motif, <italic>L</italic> denoted the location being considered, <italic>L</italic>
<sub>
<italic>k</italic>
</sub> was the nucleotide at position <italic>k</italic> within this location, <italic>p(L</italic>
<sub>
<italic>k</italic>
</sub>
<italic>)</italic> is the background probability of observing nucleotide <italic>L</italic>
<sub>
<italic>k</italic>
</sub> estimated from the frequency of <italic>L</italic>
<sub>
<italic>k</italic>
</sub> in that sequence, and <italic>q(k, L</italic>
<sub>
<italic>k</italic>
</sub>
<italic>)</italic> is the probability of observing nucleotide <italic>L</italic>
<sub>
<italic>k</italic>
</sub> estimated from the frequency of the <italic>Kth</italic> location in the motif. The Jindex for a motif present in a sequence with length <italic>l</italic> was the maximum of the average of LLRs taken over all locations of <italic>s</italic>, where <italic>M</italic>
<sub>
<italic>s</italic>
</sub> was the number of locations in the sequence calculated as <italic>l</italic>&#x2013;<italic>w</italic> &#x2b; 1. Jindex value &#x3e;1.0 indicated that the probability of observing the motif in the sequence is higher than the background probability.</p>
<p>A total of 1,575 position weight matrixs (PWMs) for transcription factors were obtained from public sources (<xref ref-type="bibr" rid="B30">Vorontsov et al., 2024</xref>; <xref ref-type="bibr" rid="B25">Rauluseviciute et al., 2024</xref>). The newest promoter sequence set (550-bp set) was obtained from Jinzer&#x2019;s website (<xref ref-type="bibr" rid="B34">Yao et al., 2024</xref>). Jinzer software was run with PWMs and the promoter sequence set identified candidate transcription factor binding site (TFBS) with Jindex &#x2265;1.0. A gene set was created by assigning target genes to TFs if the gene had at least one TFBS. Each TF&#x2013;target gene pair was assigned a rank value, which was the Jindex of Jinzer&#x2019;s output. However, it should be aware that the analysis presented in the manuscript focused on the &#x2212;1 &#x223c; &#x2212;500bp upstream sequence which will lose distal transcription factor binding site located out of this region.</p>
</sec>
<sec id="s2-3">
<title>2.3 Identification of key transcription factors from transcriptome data</title>
<p>Flaver software (<xref ref-type="bibr" rid="B34">Yao et al., 2024</xref>), which implements weighted Kendall&#x2019;s tau statistics, was used to identify the key transcription factors within each cell line. This method allows for testing the significance of the correlation between the rank orders of genes in the gene set and their corresponding rank orders within the gene list. Let Si and Li, i &#x3d; 1, &#x2026; , n be the ranks of the gene set or list, respectively. Furthermore, let (i, Ri), i &#x3d; 1, &#x2026; , n be paired ranks, where Ri is a rank entity of L, whose corresponding S has rank i among Sj, j &#x3d; 1, &#x2026; , n. Kendall&#x2019;s tau has the form of <xref ref-type="disp-formula" rid="e11">formula 11</xref> and the limiting distribution (LD), following the U-statistics of <xref ref-type="disp-formula" rid="e12">formula 12</xref> approximated to N (0, 1). This method was implemented in Faver software as an STD (-w 0 option):<disp-formula id="e11">
<mml:math id="m11">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3c4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2219;</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mi mathvariant="normal">j</mml:mi>
</mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:msubsup>
<mml:mtext>sgn</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mtext>sgn</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>
<disp-formula id="e12">
<mml:math id="m12">
<mml:mrow>
<mml:mtext>LD</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>3</mml:mn>
<mml:msqrt>
<mml:mfrac>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:msqrt>
<mml:mo>&#x2219;</mml:mo>
<mml:mi mathvariant="normal">&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
</p>
<p>Shieh (1998) discovered a weighted version of the rank correlation, the weighted Kendall&#x2019;s tau, which has the form of <xref ref-type="disp-formula" rid="e13">formula 13</xref>:<disp-formula id="e13">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">&#x3c4;</mml:mi>
<mml:mi mathvariant="normal">w</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi mathvariant="normal">v</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi mathvariant="normal">v</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2219;</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mi mathvariant="normal">j</mml:mi>
</mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi mathvariant="normal">v</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi mathvariant="normal">v</mml:mi>
<mml:mi mathvariant="normal">j</mml:mi>
</mml:msub>
<mml:mtext>sgn</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mtext>sgn</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
</p>
<p>The sgn(x) &#x3d; &#x2212;1, 0 or 1, if x &#x3c;, &#x3d; or &#x3e;0, and v<sub>i</sub> represents the weighting function bounded to [1, n] and ranges from (0, 1). The limiting distribution (LD) can be derived using <xref ref-type="disp-formula" rid="e14">formula 14</xref>:<disp-formula id="e14">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mtext>LD</mml:mtext>
<mml:mi mathvariant="normal">w</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:msqrt>
<mml:msub>
<mml:mi mathvariant="normal">&#x3c4;</mml:mi>
<mml:mi mathvariant="normal">w</mml:mi>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:munder>
<mml:mi>lim</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>&#x221e;</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi mathvariant="normal">v</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msqrt>
<mml:mrow>
<mml:munder>
<mml:mi>lim</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>&#x221e;</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mi mathvariant="normal">v</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>
</p>
<p>Where when n&#x2192;&#x221e;, LD approximated to N(0, 1).</p>
<p>Four weighting functions were developed in this study, ranging from 0 to 1. The first two weighting functions were either based on the geometric mean of the gene ranks in gene set i<sub>s</sub> and gene list (i<sub>l</sub>) or separately (<xref ref-type="disp-formula" rid="e15">formulas 15</xref>, <xref ref-type="disp-formula" rid="e16">16</xref>) to determine how the genes in the top-ranked TF&#x2013;target gene pairs in the gene set correlated with the top-ranked differentially expressed gene in the gene list. The remaining two weighting functions were either based on the geometric mean of the density of genes in gene set d<sub>s</sub> and gene list (d<sub>l</sub>) or separately (<xref ref-type="disp-formula" rid="e17">formulas 17</xref>, <xref ref-type="disp-formula" rid="e18">18</xref>). The source code was deposited in GitHub and is available under a free academic license (<ext-link ext-link-type="uri" xlink:href="https://github.com/thua45/flaver">https://github.com/thua45/flaver</ext-link>). The v<sub>1</sub> to v<sub>4</sub> methods were implemented in Flaver software as MIX-LINEAR, LINEAR, MIXED-DENSITY-CURVE, and DENSITY-CURVE, and can be specified by options w 1, 2, 7, and 8, respectively.<disp-formula id="e15">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">v</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mfenced open="|" close="" separators="&#x7c;">
<mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="|" close="" separators="&#x7c;">
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2219;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mfenced open="|" close="" separators="&#x7c;">
<mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="|" close="" separators="&#x7c;">
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>
<disp-formula id="e16">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">v</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">x</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mtext>&#x2009;or&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
<disp-formula id="e17">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">v</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2219;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>
<disp-formula id="e18">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">v</mml:mi>
<mml:mn>4</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi mathvariant="normal">x</mml:mi>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi mathvariant="normal">x</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mtext>&#x2009;or&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:mrow>
</mml:math>
<label>(18)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-4">
<title>2.4 Cell culture and flow cytometry assay</title>
<p>HL-60, MOLM-13, OCI-AML-3, and U-937 cell lines were purchased from Procell (Wuhan, China). HL-60, MOLM-13 and U937 cells were cultured in an RPMI 1640 medium supplemented with 10% FBS, 1% glutamine and 1% penicillin&#x2013;streptomycin (ThermoFisher Scientific, Dreieich, Germany). OCI-AML-3 cells were cultured in a complete medium consisting of 84% IMDM, 15% FBS, and 1% penicillin streptomycin (ThermoFisher Scientific, Dreieich, Germany). All these suspension cell lines were maintained in culture at a density below 1 &#xd7; 10<sup>6</sup> cells/mL and were used for seeding in 6-well plates with 1 &#xd7; 10<sup>6</sup> cells per well. All cell lines were grown in a humidified air incubator at 37&#xb0;C containing 5% CO<sub>2</sub>. The cells were treated with shRNA lentiviral vectors targeting ZNF460, SPI1, SPIB, ZNF384, ZNF784, and BATF3 (detailed information available in Supplemental document) following manufacturer&#x2019;s instructions, and cell proliferation was measured using CellTrace&#x2122; CFSE Cell Proliferation Kit (ThermoFisher Scientific, Cat NO. C34554).</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Overview of ranked gene sets</title>
<p>The Jinzer run took 2&#xa0;h on a 8-core Dell desktop computer and identified a total of 5.91 million significant TFBS (Jindex &#x2265;1.0). The number of target genes found in at least one TFBS for the PWMs varied from 1,548 to 7,150. TGIF1 and OVOL2 had the highest and lowest number of target genes, respectively. Compared to the 11,539 human H3K27ac Chip-Seq datasets collected from the Chip-Atlas database (<xref ref-type="bibr" rid="B37">Zou et al., 2024</xref>), 60.4% of the TFBS candidates were supported by experimental evidence. For example, NFKB1 transcription has 3,879 candidate target genes with at least one TFBS identified by Jinzer. Among these, 68.3% of the targets had a Chip-Seq pick covering the TFBS, in line with our prediction. The Jinzer gene set file was created by assigning a RANK score to each TFtarget gene pair, which was assigned from the Jindex value.</p>
<p>The distribution of Jindex for the transcription factor binding sites identified by Jinzer is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. The results indicated that the Jindex values of the binding sites for the TF-target gene pairs were approximately normally distributed (<xref ref-type="fig" rid="F1">Figure 1A</xref>). Because binding sites with higher Jindex values are more likely to be true binding sites, we set a threshold value for Jindex in practice; that is, binding sites with Jindex scores greater than the threshold value were considered candidate sites (<xref ref-type="fig" rid="F1">Figure 1B</xref>). These candidate loci can be converted into a ranked gene set and used as an input file for Flaver, denoted as &#x201c;Jinzer set&#x201d; in this study.</p>
</sec>
<sec id="s3-2">
<title>3.2 Creation of ranked gene list</title>
<p>Transcriptome data from 1,206 human cell lines were transformed into ranked cell-specific gene lists using the four indexes described in the Materials and Methods section. The ranks of the ED (<xref ref-type="fig" rid="F2">Figure 2A</xref>) and PEM (<xref ref-type="fig" rid="F2">Figure 2B</xref>) indexes were similar and approximately normally distributed. The major difference between the distributions of the ranks of the ED and PEM indexes was that the standard deviation of the ED index was significantly higher than that of the PEM index. The results of the START (<xref ref-type="fig" rid="F2">Figure 2C</xref>) were unique, informative, and interesting. The gene list created using the SPM index was severely skewed and distributed (<xref ref-type="fig" rid="F2">Figure 2D</xref>). We performed a gene search according to the gene&#x2019;s Gene Ontology (GO) annotation and found an average of 22.3% overlap between the GO gene sets annotated with sample-specific functions and the cell-specific gene lists identified by the four indexes. This suggested that many genes with known sample-related functions were also dominantly or recessively expressed in the corresponding transcriptome data and <italic>vice versa</italic>. The gene lists created by the four indexes, named the CELLINE list, were used as inputs for the Flaver software.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The distribution of the rank values of the gene list created by the DE, PEM, START, and SPM indexes. The DE and PEM indexes shown in plot <bold>(A, B)</bold>, respectively, were approximately normal distributed. The START and SPM indexes shown in plot <bold>(C, D)</bold> were severely skewed. The boxes show the 25% and 75% percentile, and the whiskers show the 5% and 95% percentile.</p>
</caption>
<graphic xlink:href="fgene-15-1511456-g002.tif"/>
</fig>
</sec>
<sec id="s3-3">
<title>3.3 Generate and test simulated dataset</title>
<p>R (version 4.2) was used to generate four synthetic gene sets and gene list data to simulate the gene expression and transcription factor binding site data to the real data distribution properties. The gene set and gene list in simulated dataset 1 were normally distributed random data with covariance (R) range of [-1, 1] (<xref ref-type="fig" rid="F2">Figure 2A</xref> shows the simulated data with R &#x3d; 0.8). For simulated dataset 2, a threshold condition of RANK &#x3e;0 was set for both the gene set and gene list, in addition to simulated dataset 1, which was used to simulate filtering high-scoring binding sites and upregulated genes when analyzing transcriptome data (<xref ref-type="fig" rid="F2">Figure 2B</xref> shows the simulated data with R &#x3d; 0.8). The gene set and gene list in simulation dataset 3 were also normally distributed random data with a covariance range in [-1, 1]. However, in contrast to simulation dataset 2, only a filter condition of RANK gene set greater than zero was set, and the gene list had a full-spectrum distribution (<xref ref-type="fig" rid="F2">Figure 2C</xref> shows the simulation data with R &#x3d; 0.8). Simulation dataset 4 introduces a covariance gradient in addition to simulation dataset 3. The gene correlation coefficient for genes with RANK &#x3e;1.5 was set to R, the gene correlation coefficient for genes with 1.0 &#x3c; RANK &#x2264;1.5 was set to R/2, and the gene correlation coefficient for genes with RANK &#x2264;1.0 was set to zero. Simulation dataset 4 was used to simulate situations in which genes with higher RANK had a higher degree of correlation, and genes with lower RANK had lower or no correlation (<xref ref-type="fig" rid="F2">Figure 2D</xref> shows simulated data with R &#x3d; 0.8).</p>
<p>The four weighting methods described in the Materials and Methods section were used to test the four sets of simulated data. For simulation data 1, the P-value and the covariance of the gene set and gene list showed a good response relationship, with a &#x201c;U-shaped&#x201d; line graph. As shown in <xref ref-type="fig" rid="F4">Figure 4A</xref>, on both sides of covariance &#x3d; 0, the P-value decreased rapidly with an increase in &#x7c;covariance&#x7c;, and no significant difference was observed among the four weighting methods. For simulation data 2, the overall shape of the line graph was different from that of simulation data 1; namely, it was &#x201c;J-shaped.&#x201d; Specifically, on the covariance (0, 1) side, the P-value decreased rapidly with an increase in &#x7c;covariance&#x7c;, which is consistent with simulation test data 1. By contrast, in the covariance (&#x2212;1, 0) side, no marked decrease was observed in the P-value with an increase in &#x7c;covariance&#x7c;, and the results of the four weighting methods were highly consistent. For simulated data 3, the trend between the obtained P-value and the covariance of the gene set and gene list was similar to that of simulation data 1; namely, it was also &#x201c;U-shaped,&#x201d; that is, the P-value decreased rapidly with an increase in &#x7c;covariance&#x7c; on both sides of covariance &#x3d; 0. The steepness of the curve was significantly lower than that of simulation data 1, and slight differences were observed between the four weighting methods. For simulated data 4, the curve between the P-value and the covariance of the gene set and gene list was also &#x201c;U-shaped.&#x201d; However, the steepness of the curve was flatter than that of simulated data 3, and the P-values obtained by the four weighting methods were significantly different. Specifically, the W7 method was the steepest, followed by the W1, W2, W8, and W0 methods.</p>
</sec>
<sec id="s3-4">
<title>3.4 Identification of cell line-specific key transcription factors</title>
<p>The Flaver software was used to identify key transcription factors in the sample-specific gene lists. The real-world gene list, CELLINE list, described in the Materials and Methods section, and the Jinzer set created in the Jinzer software, were used as input files for Flaver. The MIXED-DENSITY-CURVE function, which has been proven to be the most sensitive method (W7), was used to run the Flaver. A total of 644 transcription factors were identified within a running time of 23&#xa0;h (FDR &#x3c;0.05). <xref ref-type="fig" rid="F5">Figure 5</xref> shows the clustering results according to the sign (Dir) &#x2a; -log (P-value) value of the transcription factors. Based on the Spearman&#x2019;s rank correlation coefficient distance, the 1,206 cell lines were divided into nine categories (<xref ref-type="fig" rid="F5">Figures 5A&#x2013;I</xref>), and the transcription factors were divided into 12 categories (<xref ref-type="fig" rid="F5">Figures 1&#x2013;5, 5&#x2013;12</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Simulated gene set and gene list datasets. The simulated data were generated using the &#x201c;mvrnorm&#x201d; function in R. Plot <bold>(A)</bold> shows the simulated dataset for a full-spectrum distribution with covariance (R) &#x3d; 0.8. Plot <bold>(B)</bold> shows the simulated dataset with RANK &#x3e;0 for both the gene list, gene set, and R &#x3d; 0.8. Plot <bold>(C)</bold> shows the simulated dataset with RANK &#x3e;0 for gene set, full-spectrum distribution for gene list, and R &#x3d; 0.8. The dataset in plot <bold>(D)</bold> is the same as that of plot <bold>(C)</bold> except R is the gradient correlation (i.e. R &#x3d; 0.8 for gene list &#x3e;1.5, R &#x3d; 0.4 for 1.0 &#x3c; gene list &#x2264;1.5, and R &#x3d; 0 for gene list &#x2264;1.0).</p>
</caption>
<graphic xlink:href="fgene-15-1511456-g003.tif"/>
</fig>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Testing simulated data using four different weighted Kendall&#x2019;s tau correlation statistical methods. Plot <bold>(A&#x2013;D)</bold> show the results of simulation data 1-4, respectively. Curves with different colors indicate the relationship between the P-values obtained by the different weighting methods and the covariance values (R). Dark blue, yellow, dark brown, light green, and light gray indicate the MIXED-DENSITY-CURVE (W7), DENSITY-CURVE (W1), MIX-LINEAR (W8), LINEAR (W2), and STD (W0) weighting methods, respectively.</p>
</caption>
<graphic xlink:href="fgene-15-1511456-g004.tif"/>
</fig>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Clustering analysis of Flaver results for genome-wide transcriptome data of 1,206 cell lines. Cell line samples are shown in the columns, while transcription factors were shown in the rows. Green dots indicate positive correlations between gene set and gene list, while blue dots indicate negative correlations. Color depth was calculated by -Log10 (P-value) according to the Flaver&#x2019;s statistics.</p>
</caption>
<graphic xlink:href="fgene-15-1511456-g005.tif"/>
</fig>
<p>The proportions of cell line types in each category are plotted in <xref ref-type="fig" rid="F6">Figure 6</xref>. Brain cancer samples were predominantly located in cluster D, while leukemia, lymphoma, and myeloma samples were mainly located in cluster H. Skin cancer samples were located in cluster A, breast cancer samples in class F, and colorectal cancer samples in clusters G and I. Head and neck cancer samples were mainly located in class F, while kidney cancer samples were distributed in cluster G. Uterine and ovarian cancer samples were distributed in cluster C. Lung cancer samples were found in clusters B, C, E, I, and D. Non-cancerous samples were mainly located in cluster D.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>The proportion of disease types for cell lines in clusters obtained by Flaver analysis. The proportion of cell types for sample in each cluster are shown as pie graphs from <bold>(A&#x2013;I)</bold>.</p>
</caption>
<graphic xlink:href="fgene-15-1511456-g006.tif"/>
</fig>
</sec>
<sec id="s3-5">
<title>3.5 Regulator module discriminated cancerous and non-cancerous cell lines</title>
<p>From the clustering analysis results (<xref ref-type="fig" rid="F5">Figure 5</xref>), we identified a set of transcription factors (row 9) that showed the same regulatory behavior in leukemia samples (column H). A significant positive correlation was observed between the RANK values of the gene set and the gene list for these transcription factors. This set of transcription factors could be used to distinguish leukemia cell lines from other cell types (<xref ref-type="fig" rid="F7">Figure 7A</xref>). Based on the comparison of the differences in -Sign (P-value) &#x2a; Log10(P-value) for these transcription factors between the leukemia cell lines and other types of cell lines, we selected six transcription factors with the highest differences between the leukemia and non-cancer groups for further experimental verification, namely ZNF460, SPI1, SPIB, ZNF384, ZNF784, and BATF3 (<xref ref-type="fig" rid="F7">Figure 7B</xref>).</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Comparative analysis of transcription factors between cancer and non-cancer groups. Plot <bold>(A)</bold> shows the normalized mean -Log10 (P-value) values for transcription factors between the cancer (orange) and non-cancer groups (light green). Plot <bold>(B)</bold> shows the difference between the cancer and non-cancer groups for each transcription factor, the top 6 of which were selected for wet-lab validation.</p>
</caption>
<graphic xlink:href="fgene-15-1511456-g007.tif"/>
</fig>
<p>Four cell lines (HL-60, MOLM-13, OCI-AML-3, and U-937) were used for experimental validation. The results which were presented in <xref ref-type="fig" rid="F8">Figure 8</xref> showed that the ZNF460 shRNA lentiviral vector significantly inhibited the proliferation of HL-60, MOLM-13, OCI-AML-3, and U-937 cells, and that the ZNF460, SPI1, ZNF784, and BATF3 shRNA lentiviral vectors exerted strong inhibitory effects. The inhibition rates of ZNF460, SPIB, ZNF784, and BATF3 in MOLM-13 cells were over 20%. Only one transcription factor (ZNF384) lentiviral vector did not reach significance in the OCI-AML-3 cell lines. The ZNF460 lentivirus vector inhibited the proliferation of 80% of the U-937 cells.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Flow cytometry analysis of the proliferation rate of leukemia cells treated with shRNA lentiviral vectors targeting key transcription factors. Plot <bold>(A&#x2013;D)</bold> show the flow cytometry results for HL-60, MOLM-13, OCI-AML-3, U-937, respectively. The nontreated control samples and samples treated with transcription factors of ZNF460, SPI1, SPIB, ZNF384, ZNF784, and BATF3 are colored in dark green to light green, respectively. Each experiment was run in triplicate.</p>
</caption>
<graphic xlink:href="fgene-15-1511456-g008.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>Accurately mining key transcription factors and elucidating their characteristics will help to determine the mechanism underlying the genetic regulation of the cell transcriptome (<xref ref-type="bibr" rid="B23">Meyer and Liu, 2020</xref>). This in turn requires the identification of the transcription factor binding sites and their activation patterns. By combining the activation and repression statuses of transcription factors and changes in the transcription levels of their target genes, we were able to identify the key transcription factors in the transcriptome (<xref ref-type="bibr" rid="B34">Yao et al., 2024</xref>). Under experimental conditions, high-throughput methods, such as RNA-seq, can be used to quantify transcription and non-transcription factors in the transcriptome simultaneously (<xref ref-type="bibr" rid="B27">Stark et al., 2019</xref>). Various well-developed transcription factor binding site prediction software and ChIP-Seq technologies have been used to identify a large number of candidate transcription factor binding sites (<xref ref-type="bibr" rid="B21">Maleki et al., 2020</xref>; <xref ref-type="bibr" rid="B19">MacQuarrie et al., 2011</xref>). This has in turn enabled the identification of key transcription factors by correlating their binding affinities to target genes with the expression levels of these genes. Recently, a comparison of four state-of-the-art key transcription factor mining tools found that the Flaver method, which is based on weighted Kendall&#x2019;s tau correlation coefficients, showed an improved performance (<xref ref-type="bibr" rid="B34">Yao et al., 2024</xref>).</p>
<p>At present, there are three correlation analysis methods: Pearson&#x2019;s correlation coefficient, Spearman&#x2019;s correlation coefficient, and Kendall&#x2019;s tau correlation coefficient. The prerequisite for applying the Pearson&#x2019;s correlation coefficient is that the input numerical value is quantitative data with a normal distribution (<xref ref-type="bibr" rid="B1">Akoglu, 2018</xref>). The Spearman&#x2019;s correlation coefficient can be used when the data do not follow a normal distribution (<xref ref-type="bibr" rid="B24">Pripp, 2018</xref>), while Kendall&#x2019;s tau correlation coefficient can be used when the input data are ordered categorical variables (<xref ref-type="bibr" rid="B3">Arndt et al., 1999</xref>). In a typical transcriptome profiling experiment, we set a threshold for the degree of difference in gene expression levels to obtain a gene list (<xref ref-type="bibr" rid="B10">Huang et al., 2018</xref>; <xref ref-type="bibr" rid="B11">Huang et al., 2020a</xref>). In transcription factor binding site analysis experiments, the same strategy is often used, namely setting a penalty threshold for binding sites and selecting data higher than the threshold to identify candidate binding sites with a higher reliability (<xref ref-type="bibr" rid="B12">Huang et al., 2022</xref>; <xref ref-type="bibr" rid="B7">Grant et al., 2011</xref>). This situation can be accurately simulated using the data shown in <xref ref-type="fig" rid="F3">Figure 3B</xref>. The test results of this study showed that an incomplete distribution profile of data similar to that shown in <xref ref-type="fig" rid="F3">Figure 3B</xref> can cause fallacies when applying Kendall&#x2019;s tau statistical analysis; that is, when both gene list and gene set are filtered with RANK &#x3e; threshold (zero in <xref ref-type="fig" rid="F3">Figure 3B</xref>), negatively correlated transcription factors cannot be detected. The reason for this fallacy is obvious: when the RANK threshold screening condition is set, the data in the RANK &#x3c; threshold part is missing. Assuming that the RANK value of gene set to be analyzed is positive, and negatively correlated with the gene list, the distribution of the corresponding gene list values should be located in the RANK &#x3c; threshold region. As mentioned above, if the RANK &#x3e; threshold screening condition is set, this part is missing, making the negatively correlated genes undetectable. In fact, this can be rescued if either the gene set or the gene list has a full-spectrum distribution; the simulation data in <xref ref-type="fig" rid="F3">Figures 3C, D</xref> illustrate the case when the simulated gene set has a semi-spectrum distribution and the gene list has a full-spectrum distribution. The results in <xref ref-type="fig" rid="F3">Figures 3C, D</xref> are similar to those shown in <xref ref-type="fig" rid="F3">Figure 3A</xref>, when both the gene set and gene list followed a full-spectrum normal distribution, and both positive and negative correlations could be detected without failure.</p>
<p>In this study, four methods were used to transform transcriptome data into a gene list. First, the star START method was used to successfully convert high-dimensional data into two-dimensional data. However, when applied to gene expression data, it did not resolve the issue of arranging the order of coordinate axes and estimating the degree of gene expression difference. As previously mentioned, the order of samples can be determined by the hierarchical clustering of gene expression data. The axes of each sample in the START system were arranged in the order determined by the cluster tree. Unique 2D scatter plots with self-organizing features can be constructed for specific transcriptome data. Then, the differential gene expression levels were measured by calculating the vertical distance between the data points and the coordinate axis in a two-dimensional scatter plot. The data points located near the axis with a long vertical distance from the origin were the dominantly expressed genes of the sample, whereas the data points located near the axis with a short vertical distance from the origin were the recessively expressed genes in the sample. However, the START method can only identify genes with RANK &#x3e; zero in the gene list. Combined with the analysis results of the four simulation datasets shown in <xref ref-type="fig" rid="F3">Figure 3</xref>, the gene list created by the START method suffered from the fallacy shown in <xref ref-type="fig" rid="F4">Figure 4B</xref> during Flaver analysis. The gene list transformed using the SPM method also belongs to the incomplete skewed distribution data type, such that it will also suffer the &#x201c;unable to detect negative correlation&#x201d; fallacy. The ED and Hg methods are essentially the same, and the RANK values for Hg method are the log form of the ED method. According to the results presented in <xref ref-type="fig" rid="F2">Figures 2A, B</xref>, the distribution of the gene list created using the ED and Hg method followed an approximately normal distribution. Therefore, theoretically, it conforms to the simulated data presented in <xref ref-type="fig" rid="F3">Figures 3C, D</xref>, and thus does not commit the &#x201c;unable to detect negative correlation&#x201d; fallacy shown in <xref ref-type="fig" rid="F3">Figure 3B</xref>.</p>
<p>A possible solution is presented by Flaver software, which emphasizing genes with a high RANK and de-emphasizing those with a low RANK by implementing weighted Kendall&#x2019;s tau statistics to measure the correlation, an innovative feature that is well demonstrated in this paper. The data in <xref ref-type="fig" rid="F3">Figure 3D</xref> simulated the gene set and gene list, which were correlated with gradient eco-efficiency; that is, the high RANK regions were highly correlated, the medium RANK regions were moderately correlated, and the low RANK regions were not correlated. The effects of the five weighting methods on this dataset were evident as they were significantly different from the uniformly correlated simulated data shown in <xref ref-type="fig" rid="F3">Figure 3A</xref>. The STD method tested in this paper is a standard implementation of Kendall&#x2019;s tau correlation coefficient, the MIX-LINEAR method is a mixed linear weighting formula based on RANK of both gene set and gene list, the LINEAR method is a linear weighting formula for gene list only, which have been discussed previously (<xref ref-type="bibr" rid="B34">Yao et al., 2024</xref>). The MIXED-DENSITY-CURVE method is a mixed density distribution curve weighting formula based on the RANK of both the gene set and the gene list, and the DENSITY-CURVE method is a density distribution curve weighting formula for the gene list only. From the evaluation results presented in <xref ref-type="fig" rid="F4">Figure 4D</xref>, the MIXED-DENSITY-CURVE method was identified as the most sensitive, resulting in a steeper U-shaped curve than any of the other four weighting methods.</p>
<p>As a result of transcriptome profiling experiments, most genes in the genome were found to be expressed at average levels (approximately normal distribution) in cells; genes with specifically high and specifically low levels of expression only accounted for a small number of genes (<xref ref-type="bibr" rid="B13">Huang et al., 2020b</xref>; <xref ref-type="bibr" rid="B5">Conesa et al., 2016</xref>). Providing equal weights to all genes in a weighted Kendall&#x2019;s tau correlation analysis may obscure the genes of greatest interest, namely those that are either highly and weakly expressed, resulting in inappropriate statistical inference (<xref ref-type="bibr" rid="B34">Yao et al., 2024</xref>). As shown in this study, the MIXED-DENSITY-CURVE method is a hybrid density distribution curve weighting formula based on the RANK of both the gene set and gene list, which is symmetrically distributed along the coordinate axis on both sides of the origin. Theoretically, the MIXED-DENSITY-CURVE method has advantages in emphasizing specifically highly expressed and specifically low expressed genes, as well as weakening non-differentially expressed genes. This is the essence of weighted Kendall&#x2019;s tau rank correlation, namely its ability to avoid the artificial statistical biases caused by imbalanced weights for upregulated or downregulated genes.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s12">Supplementary Material</xref>, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec sec-type="ethics-statement" id="s6">
<title>Ethics statement</title>
<p>Ethical approval was not required for the studies on humans in accordance with the local legislation and institutional requirements because only commercially available established cell lines were used.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>TH: Formal Analysis, Funding acquisition, Writing&#x2013;original draft, Methodology, Project administration, Software. SN: Formal Analysis, Writing&#x2013;original draft, Data curation. FZ: Data curation, Formal Analysis, Writing&#x2013;original draft. BW: Data curation, Formal Analysis, Writing&#x2013;original draft. JW: Data curation, Formal Analysis, Writing&#x2013;original draft. GL: Conceptualization, Project administration, Writing&#x2013;review and editing. MY: Conceptualization, Formal Analysis, Funding acquisition, Writing&#x2013;original draft.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This project was funded by the Science and Technology Project for Regional Innovation of Hubei Province [Grant No. 2024EHA010], Key Research and Development Program of Hubei Province [Grant No. 2023BBB045], National Natural Science Foundation of China [NSFC Grant No. 31902231], the College Students&#x2019; Innovation and Entrepreneurship Training Program of Yangtze University [Grant No. YZ2023069].</p>
</sec>
<ack>
<p>We thank Editage providing language assistance and proof reading for the article.</p>
</ack>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2024.1511456/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2024.1511456/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Akoglu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>User&#x27;s guide to correlation coefficients</article-title>. <source>Turk J. Emerg. Med.</source> <volume>18</volume> (<issue>3</issue>), <fpage>91</fpage>&#x2013;<lpage>93</lpage>. <comment>Epub 20180807</comment>. <pub-id pub-id-type="doi">10.1016/j.tjem.2018.08.001</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alvarez</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Giorgi</surname>
<given-names>F. M.</given-names>
</name>
<name>
<surname>Lachmann</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>B. B.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>B. H.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Functional characterization of somatic mutations in cancer using network-based inference of Protein activity</article-title>. <source>Nat. Genet.</source> <volume>48</volume> (<issue>8</issue>), <fpage>838</fpage>&#x2013;<lpage>847</lpage>. <comment>Epub 20160620</comment>. <pub-id pub-id-type="doi">10.1038/ng.3593</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arndt</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Turvey</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Andreasen</surname>
<given-names>N. C.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Correlating and predicting psychiatric symptom ratings: spearman&#x27;s R versus Kendall&#x27;s tau correlation</article-title>. <source>J. Psychiatr. Res.</source> <volume>33</volume> (<issue>2</issue>), <fpage>97</fpage>&#x2013;<lpage>104</lpage>. <pub-id pub-id-type="doi">10.1016/s0022-3956(98)90046-2</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bailey</surname>
<given-names>T. L.</given-names>
</name>
<name>
<surname>Boden</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Buske</surname>
<given-names>F. A.</given-names>
</name>
<name>
<surname>Frith</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Grant</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Clementi</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Meme suite: tools for motif discovery and searching</article-title>. <source>Nucleic Acids Res.</source> <volume>37</volume> (<issue>Web Server issue</issue>), <fpage>W202</fpage>&#x2013;<lpage>W208</lpage>. <comment>Epub 20090520</comment>. <pub-id pub-id-type="doi">10.1093/nar/gkp335</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Conesa</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Madrigal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Tarazona</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gomez-Cabrero</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Cervera</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>McPherson</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>A survey of best practices for rna-seq data analysis</article-title>. <source>Genome Biol.</source> <volume>17</volume>, <fpage>13</fpage>. <comment>Epub 20160126</comment>. <pub-id pub-id-type="doi">10.1186/s13059-016-0881-8</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Falcon</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gentleman</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Using gostats to test gene lists for go term association</article-title>. <source>Bioinformatics</source> <volume>23</volume> (<issue>2</issue>), <fpage>257</fpage>&#x2013;<lpage>258</lpage>. <comment>Epub 2006/11/14</comment>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btl567</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grant</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Bailey</surname>
<given-names>T. L.</given-names>
</name>
<name>
<surname>Noble</surname>
<given-names>W. S.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Fimo: scanning for occurrences of a given motif</article-title>. <source>Bioinformatics</source> <volume>27</volume> (<issue>7</issue>), <fpage>1017</fpage>&#x2013;<lpage>1018</lpage>. <comment>Epub 20110216</comment>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btr064</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heinrich</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Weiskopf</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Continuous parallel coordinates</article-title>. <source>IEEE Trans. Vis. Comput. Graph</source> <volume>15</volume> (<issue>6</issue>), <fpage>1531</fpage>&#x2013;<lpage>1538</lpage>. <pub-id pub-id-type="doi">10.1109/tvcg.2009.131</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heinz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Benner</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Spann</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Bertolino</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y. C.</given-names>
</name>
<name>
<surname>Laslo</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Simple combinations of lineage-determining transcription factors prime cis-regulatory elements required for macrophage and B cell identities</article-title>. <source>Mol. Cell</source> <volume>38</volume> (<issue>4</issue>), <fpage>576</fpage>&#x2013;<lpage>589</lpage>. <pub-id pub-id-type="doi">10.1016/j.molcel.2010.05.004</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Regulators of Salmonella-host interaction identified by peripheral blood transcriptome profiling: roles of Tgfb1 and Trp53 in intracellular Salmonella replication in pigs</article-title>. <source>Vet. Res.</source> <volume>49</volume> (<issue>1</issue>), <fpage>121</fpage>. <comment>Epub 2018/12/14</comment>. <pub-id pub-id-type="doi">10.1186/s13567-018-0616-9</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2020a</year>). <article-title>Salmonella enterica serovar typhimurium inhibits the innate immune response and promotes apoptosis in a ribosomal/trp53-dependent manner in swine neutrophils</article-title>. <source>Vet. Res.</source> <volume>51</volume> (<issue>1</issue>), <fpage>105</fpage>. <comment>Epub 2020/08/29</comment>. <pub-id pub-id-type="doi">10.1186/s13567-020-00828-3</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Identification of upstream transcription factor binding sites in orthologous genes using mixed student&#x27;s T-test statistics</article-title>. <source>PLoS Comput. Biol.</source> <volume>18</volume> (<issue>6</issue>), <fpage>e1009773</fpage>. <comment>Epub 20220607</comment>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1009773</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2020b</year>). <article-title>A transcriptional landscape of 28 porcine tissues obtained by super deepsage sequencing</article-title>. <source>BMC Genomics</source> <volume>21</volume> (<issue>1</issue>), <fpage>229</fpage>. <comment>Epub 2020/03/17</comment>. <pub-id pub-id-type="doi">10.1186/s12864-020-6628-7</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huminiecki</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lloyd</surname>
<given-names>A. T.</given-names>
</name>
<name>
<surname>Wolfe</surname>
<given-names>K. H.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Congruence of tissue expression profiles from gene expression Atlas, sagemap and tissueinfo databases</article-title>. <source>BMC Genomics</source> <volume>4</volume> (<issue>1</issue>), <fpage>31</fpage>. <comment>Epub 20030729</comment>. <pub-id pub-id-type="doi">10.1186/1471-2164-4-31</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kandogan</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2000</year>). &#x201c;<article-title>Star coordinates: a multi-dimensional visualization technique with uniform treatment of dimensions</article-title>,&#x201d; in <source>Proceedings of the IEEE information visualization symposium late breaking hot topics</source>, <fpage>9</fpage>&#x2013;<lpage>12</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Keenan</surname>
<given-names>A. B.</given-names>
</name>
<name>
<surname>Torre</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lachmann</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Leong</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Wojciechowicz</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Utti</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Chea3: transcription factor enrichment analysis by orthogonal omics integration</article-title>. <source>Nucleic Acids Res.</source> <volume>47</volume> (<issue>W1</issue>), <fpage>W212</fpage>&#x2013;<lpage>W224</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkz446</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kryuchkova-Mostacci</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Robinson-Rechavi</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A benchmark of gene expression tissue-specificity metrics</article-title>. <source>Brief. Bioinform</source> <volume>18</volume> (<issue>2</issue>), <fpage>205</fpage>&#x2013;<lpage>214</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbw008</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lachmann</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>H. L.</given-names>
</name>
<name>
<surname>Krishnan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Berger</surname>
<given-names>S. I.</given-names>
</name>
<name>
<surname>Mazloom</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Ma&#x27;ayan</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Chea: transcription factor regulation inferred from integrating genome-wide chip-X experiments</article-title>. <source>Bioinformatics</source> <volume>26</volume> (<issue>19</issue>), <fpage>2438</fpage>&#x2013;<lpage>2444</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btq466</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>MacQuarrie</surname>
<given-names>K. L.</given-names>
</name>
<name>
<surname>Fong</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Morse</surname>
<given-names>R. H.</given-names>
</name>
<name>
<surname>Tapscott</surname>
<given-names>S. J.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Genome-wide transcription factor binding: beyond direct target regulation</article-title>. <source>Trends Genet.</source> <volume>27</volume> (<issue>4</issue>), <fpage>141</fpage>&#x2013;<lpage>148</lpage>. <comment>Epub 20110204</comment>. <pub-id pub-id-type="doi">10.1016/j.tig.2011.01.001</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Magnusson</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Lubovac-Pilav</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Tftenricher: a Python toolbox for annotation enrichment analysis of transcription factor target genes</article-title>. <source>Bmc Bioinforma.</source> <volume>22</volume> (<issue>1</issue>), <fpage>440</fpage>&#x2013;<lpage>449</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-021-04357-4</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maleki</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ovens</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hogan</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Kusalik</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Gene set analysis: challenges, opportunities, and future research</article-title>. <source>Front. Genet.</source> <volume>11</volume>, <fpage>654</fpage>. <comment>Epub 2020/07/23</comment>. <pub-id pub-id-type="doi">10.3389/fgene.2020.00654</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Margolin</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Nemenman</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Basso</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wiggins</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Stolovitzky</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Dalla</surname>
<given-names>F. R.</given-names>
</name>
<etal/>
</person-group> (<year>2006</year>). <article-title>Aracne: an algorithm for the reconstruction of gene regulatory networks in a mammalian cellular context</article-title>. <source>BMC Bioinforma.</source> <volume>7</volume> (<issue>Suppl. 1</issue>), <fpage>S7</fpage>. <comment>Epub 20060320</comment>. <pub-id pub-id-type="doi">10.1186/1471-2105-7-s1-s7</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meyer</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X. S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Computational approaches to modeling transcription factor activity and gene regulation</article-title>. <source>Trends Biochem. Sci.</source> <volume>45</volume> (<issue>12</issue>), <fpage>1094</fpage>&#x2013;<lpage>1095</lpage>. <comment>Epub 20201010</comment>. <pub-id pub-id-type="doi">10.1016/j.tibs.2020.09.005</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pripp</surname>
<given-names>A. H.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Pearson&#x27;s or spearman&#x27;s correlation coefficients</article-title>. <source>Tidsskr. Nor. Laegeforen</source> <volume>138</volume> (<issue>8</issue>). <comment>Epub 20180508</comment>. <pub-id pub-id-type="doi">10.4045/tidsskr.18.0042</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rauluseviciute</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Riudavets-Puig</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Blanc-Mathieu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Castro-Mondragon</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Ferenc</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Jaspar 2024: 20th anniversary of the open-access database of transcription factor binding profiles</article-title>. <source>Nucleic Acids Res.</source> <volume>52</volume> (<issue>D1</issue>), <fpage>D174</fpage>&#x2013;<lpage>D182</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkad1059</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schug</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Schuller</surname>
<given-names>W. P.</given-names>
</name>
<name>
<surname>Kappen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Salbaum</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Bucan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Stoeckert</surname>
<given-names>C. J.</given-names>
<suffix>Jr</suffix>
</name>
</person-group> (<year>2005</year>). <article-title>Promoter features related to tissue specificity as measured by shannon entropy</article-title>. <source>Genome Biol.</source> <volume>6</volume> (<issue>4</issue>), <fpage>R33</fpage>. <comment>Epub 20050329</comment>. <pub-id pub-id-type="doi">10.1186/gb-2005-6-4-r33</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stark</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Grzelak</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hadfield</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Rna sequencing: the teenage years</article-title>. <source>Nat. Rev. Genet.</source> <volume>20</volume> (<issue>11</issue>), <fpage>631</fpage>&#x2013;<lpage>656</lpage>. <comment>Epub 20190724</comment>. <pub-id pub-id-type="doi">10.1038/s41576-019-0150-2</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Subramanian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tamayo</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mootha</surname>
<given-names>V. K.</given-names>
</name>
<name>
<surname>Mukherjee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ebert</surname>
<given-names>B. L.</given-names>
</name>
<name>
<surname>Gillette</surname>
<given-names>M. A.</given-names>
</name>
<etal/>
</person-group> (<year>2005</year>). <article-title>Gene set enrichment analysis: a knowledge-based approach for interpreting genome-wide expression profiles</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>102</volume> (<issue>43</issue>), <fpage>15545</fpage>&#x2013;<lpage>15550</lpage>. <comment>Epub 20050930</comment>. <pub-id pub-id-type="doi">10.1073/pnas.0506580102</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uhl&#xe9;n</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fagerberg</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hallstr&#xf6;m</surname>
<given-names>B. M.</given-names>
</name>
<name>
<surname>Lindskog</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Oksvold</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mardinoglu</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Proteomics. Tissue-based map of the human proteome</article-title>. <source>Science</source> <volume>347</volume> (<issue>6220</issue>), <fpage>1260419</fpage>. <comment>Epub 2015/01/24</comment>. <pub-id pub-id-type="doi">10.1126/science.1260419</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vorontsov</surname>
<given-names>I. E.</given-names>
</name>
<name>
<surname>Eliseeva</surname>
<given-names>I. A.</given-names>
</name>
<name>
<surname>Zinkevich</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nikonov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Abramov</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Boytsov</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Hocomoco in 2024: a rebuild of the curated collection of binding models for human and mouse transcription factors</article-title>. <source>Nucleic Acids Res.</source> <volume>52</volume> (<issue>D1</issue>), <fpage>D154</fpage>&#x2013;<lpage>D163</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkad1077</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Vasaikar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Greer</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Webgestalt 2017: a more comprehensive, powerful, flexible and interactive gene set enrichment analysis toolkit</article-title>. <source>Nucleic Acids Res.</source> <volume>45</volume> (<issue>W1</issue>), <fpage>W130</fpage>&#x2013;<lpage>W137</lpage>. <comment>Epub 2017/05/05</comment>. <pub-id pub-id-type="doi">10.1093/nar/gkx356</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiao</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>Z. L.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Tisged: a database for tissue-specific genes</article-title>. <source>Bioinformatics</source> <volume>26</volume> (<issue>9</issue>), <fpage>1273</fpage>&#x2013;<lpage>1275</lpage>. <comment>Epub 20100311</comment>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btq109</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yanai</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Benjamin</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shmoish</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chalifa-Caspi</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Shklar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ophir</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2005</year>). <article-title>Genome-wide midrange transcription profiles reveal expression level relationships in human tissue specification</article-title>. <source>Bioinformatics</source> <volume>21</volume> (<issue>5</issue>), <fpage>650</fpage>&#x2013;<lpage>659</lpage>. <comment>Epub 20040923</comment>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bti042</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Testing the significance of ranked gene sets in genome-wide transcriptome profiling data using weighted rank correlation statistics</article-title>. <source>Curr. Genomics</source> <volume>25</volume> (<issue>3</issue>), <fpage>202</fpage>&#x2013;<lpage>211</lpage>. <comment>Epub 2024/08/01</comment>. <pub-id pub-id-type="doi">10.2174/0113892029280470240306044159</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L. G.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Q. Y.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Clusterprofiler: an R package for comparing biological themes among gene clusters</article-title>. <source>OMICS</source> <volume>16</volume> (<issue>5</issue>), <fpage>284</fpage>&#x2013;<lpage>287</lpage>. <comment>Epub 2012/03/30</comment>. <pub-id pub-id-type="doi">10.1089/omi.2011.0118</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zack</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Qian</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Computational analysis of tissue-specific combinatorial gene regulation: predicting interaction between transcription factors in human tissues</article-title>. <source>Nucleic Acids Res.</source> <volume>34</volume> (<issue>17</issue>), <fpage>4925</fpage>&#x2013;<lpage>4936</lpage>. <comment>Epub 20060918</comment>. <pub-id pub-id-type="doi">10.1093/nar/gkl595</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ohta</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Oki</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Chip-atlas 3.0: a data-mining suite to explore chromosome architecture together with large-scale regulome data</article-title>. <source>Nucleic Acids Res.</source> <volume>52</volume> (<issue>W1</issue>), <fpage>W45</fpage>&#x2013;<lpage>w53</lpage>. <comment>Epub 2024/05/16</comment>. <pub-id pub-id-type="doi">10.1093/nar/gkae358</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>