<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Bioeng. Biotechnol.</journal-id>
<journal-title>Frontiers in Bioengineering and Biotechnology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Bioeng. Biotechnol.</abbrev-journal-title>
<issn pub-type="epub">2296-4185</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fbioe.2017.00048</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Bioengineering and Biotechnology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Navigating the Functional Landscape of Transcription Factors via Non-Negative Tensor Factorization Analysis of MEDLINE Abstracts</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Roy</surname> <given-names>Sujoy</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://frontiersin.org/people/u/438076"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Yun</surname> <given-names>Daqing</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://frontiersin.org/people/u/441179"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Madahian</surname> <given-names>Behrouz</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://frontiersin.org/people/u/467395"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Berry</surname> <given-names>Michael W.</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://frontiersin.org/people/u/465019"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Deng</surname> <given-names>Lih-Yuan</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://frontiersin.org/people/u/467307"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Goldowitz</surname> <given-names>Daniel</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<uri xlink:href="http://frontiersin.org/people/u/5944"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Homayouni</surname> <given-names>Ramin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<xref ref-type="corresp" rid="cor1">&#x0002A;</xref>
<uri xlink:href="http://frontiersin.org/people/u/467286"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Bioinformatics Program, University of Memphis</institution>, <addr-line>Memphis, TN</addr-line>, <country>United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>Center for Translational Informatics, University of Memphis</institution>, <addr-line>Memphis, TN</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>Computer and Information Sciences Program, Harrisburg University of Science and Technology</institution>, <addr-line>Harrisburg, PA</addr-line>, <country>United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Mathematical Sciences, University of Memphis</institution>, <addr-line>Memphis, TN</addr-line>, <country>United States</country></aff>
<aff id="aff5"><sup>5</sup><institution>Department of Electrical Engineering and Computer Science, University of Tennessee</institution>, <addr-line>Knoxville, TN</addr-line>, <country>United States</country></aff>
<aff id="aff6"><sup>6</sup><institution>Center for Molecular Medicine and Therapeutics, University of British Columbia</institution>, <addr-line>Vancouver, BC</addr-line>, <country>Canada</country></aff>
<aff id="aff7"><sup>7</sup><institution>Department of Biological Sciences, University of Memphis</institution>, <addr-line>Memphis, TN</addr-line>, <country>United States</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Mattia Pelizzola, Fondazione Istituto Italiano di Technologia, Italy</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Zhi-Ping Liu, Shandong University, China; Ka-Chun Wong, City University of Hong Kong, Hong Kong</p></fn>
<corresp content-type="corresp" id="cor1">&#x0002A;Correspondence: Ramin Homayouni, <email>rhomayon&#x00040;memphis.edu</email></corresp>
<fn fn-type="other" id="fn001"><p>Specialty section: This article was submitted to Bioinformatics and Computational Biology, a section of the journal Frontiers in Bioengineering and Biotechnology</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>28</day>
<month>08</month>
<year>2017</year>
</pub-date>
<pub-date pub-type="collection">
<year>2017</year>
</pub-date>
<volume>5</volume>
<elocation-id>48</elocation-id>
<history>
<date date-type="received">
<day>18</day>
<month>05</month>
<year>2017</year>
</date>
<date date-type="accepted">
<day>31</day>
<month>07</month>
<year>2017</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2017 Roy, Yun, Madahian, Berry, Deng, Goldowitz and Homayouni.</copyright-statement>
<copyright-year>2017</copyright-year>
<copyright-holder>Roy, Yun, Madahian, Berry, Deng, Goldowitz and Homayouni</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) or licensor are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>In this study, we developed and evaluated a novel text-mining approach, using non-negative tensor factorization (NTF), to simultaneously extract and functionally annotate transcriptional modules consisting of sets of genes, transcription factors (TFs), and terms from MEDLINE abstracts. A sparse 3-mode term&#x02009;&#x000D7;&#x02009;gene&#x02009;&#x000D7;&#x02009;TF tensor was constructed that contained weighted frequencies of 106,895 terms in 26,781 abstracts shared among 7,695 genes and 994 TFs. The tensor was decomposed into sub-tensors using non-negative tensor factorization (NTF) across 16 different approximation ranks. Dominant entries of each of 2,861 sub-tensors were extracted to form term&#x02013;gene&#x02013;TF annotated transcriptional modules (ATMs). More than 94% of the ATMs were found to be enriched in at least one KEGG pathway or GO category, suggesting that the ATMs are functionally relevant. One advantage of this method is that it can discover potentially new gene&#x02013;TF associations from the literature. Using a set of microarray and ChIP-Seq datasets as gold standard, we show that the precision of our method for predicting gene&#x02013;TF associations is significantly higher than chance. In addition, we demonstrate that the terms in each ATM can be used to suggest new GO classifications to genes and TFs. Taken together, our results indicate that NTF is useful for simultaneous extraction and functional annotation of transcriptional regulatory networks from unstructured text, as well as for literature based discovery. A web tool called Transcriptional Regulatory Modules Extracted from Literature (TREMEL), available at <uri xlink:href="http://binf1.memphis.edu/tremel">http://binf1.memphis.edu/tremel</uri>, was built to enable browsing and searching of ATMs.</p>
</abstract>
<kwd-group>
<kwd>biomedical text mining</kwd>
<kwd>tensor factorization</kwd>
<kwd>tensor decomposition</kwd>
<kwd>multiway analysis</kwd>
<kwd>applied multilinear algebra</kwd>
<kwd>transcription factors</kwd>
</kwd-group>
<counts>
<fig-count count="6"/>
<table-count count="4"/>
<equation-count count="16"/>
<ref-count count="75"/>
<page-count count="14"/>
<word-count count="10961"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="introduction">
<label>1</label> <title>Introduction</title>
<p>The complexity of organisms is correlated with the number of mechanisms by which gene expression is regulated in response to environmental and developmental signals (Levine and Tjian, <xref ref-type="bibr" rid="B43">2003</xref>; Chen and Rajewsky, <xref ref-type="bibr" rid="B19">2007</xref>; Davidson, <xref ref-type="bibr" rid="B22">2010</xref>). Transcriptional regulation involves complex gene regulatory networks (GRNs), consisting of structural proteins involved in chromatin remodeling and transcription factors that regulate the core transcriptional machinery (Djebali et al., <xref ref-type="bibr" rid="B24">2012</xref>). An active area of research is focused on integration of various high-throughput &#x0201C;omic&#x0201D; data in order to understand how genes are functionally regulated and involved in physiological and pathological processes (Gerstein et al., <xref ref-type="bibr" rid="B26">2012</xref>). However, aggregation and annotation of gene regulatory networks (GRNs) from various sources remain challenging. Some GRN annotation is available in repositories such as KEGG (Kanehisa et al., <xref ref-type="bibr" rid="B37">2004</xref>) and GO (Ashburner et al., <xref ref-type="bibr" rid="B6">2000</xref>). However, these knowledge bases are incomplete and too general to provide specific insights into GRNs. More recent efforts have focused on manually integrating GRN information from various data sources (Liu et al., <xref ref-type="bibr" rid="B46">2015</xref>). In addition, methods to automatically annotate GRNs based on semantic relationships in the biomedical literature are beginning to be developed (Chen et al., <xref ref-type="bibr" rid="B17">2014</xref>). Currently, there are more than 23 million citations in MEDLINE, many of which describe relationships between gene products and molecular and cellular processes. There is, therefore, a growing need to develop automated text-mining techniques to utilize knowledge in the biomedical literature to interpret genome-wide experimental data as well as to aid in the manual curation processes (Rebholz-Schuhmann et al., <xref ref-type="bibr" rid="B56">2012</xref>).</p>
<p>In addition to knowledge extraction, literature mining methods provide a valuable resource for knowledge discovery based on implicit associations in the literature. The concept of literature-based discovery (LBD) was introduced by Swanson several decades ago and is increasingly being discussed in the scientific community (Swanson, <xref ref-type="bibr" rid="B66">1986</xref>; Blagosklonny and Pardee, <xref ref-type="bibr" rid="B12">2002</xref>; Soldatova and Rzhetsky, <xref ref-type="bibr" rid="B64">2011</xref>). Several co-occurrence based-LBD approaches, such as CoPub Mapper (Alako et al., <xref ref-type="bibr" rid="B4">2005</xref>), PubGene (Jenssen et al., <xref ref-type="bibr" rid="B35">2001</xref>), Chilibot (Chen and Sharp, <xref ref-type="bibr" rid="B18">2004</xref>), and GeneWays (Rzhetsky et al., <xref ref-type="bibr" rid="B61">2004</xref>), have been developed. Other approaches have focused on capturing higher order implicit associations, i.e., associations between any pair of entities that do not directly share any abstracts but may share abstracts with other common entities (Burkart et al., <xref ref-type="bibr" rid="B14">2007</xref>). A few approaches have focused on mining TF specific regulatory associations from the literature. Dragon TF association miner (Pan et al., <xref ref-type="bibr" rid="B52">2004</xref>) is a web-based tool that accepts as input a set of abstracts, and identifies and extracts TF associations with Gene Ontology terms found within the text. Natural language processing (NLP) techniques have been used to identify sentences pertaining to transcriptional regulation and to extract relationships from PubMed abstracts for reconstructing regulatory networks (Chen and Sharp, <xref ref-type="bibr" rid="B18">2004</xref>; &#x00160;ari&#x00107; et al., <xref ref-type="bibr" rid="B62">2006</xref>; Rodr&#x000ED;guez-Penagos et al., <xref ref-type="bibr" rid="B57">2007</xref>; Chen et al., <xref ref-type="bibr" rid="B17">2014</xref>). Vector space models have been investigated in annotation of regulatory networks by prioritizing MEDLINE abstracts likely to have high cis-regulatory content (Aerts et al., <xref ref-type="bibr" rid="B3">2008</xref>). A bootstrapping method has been used to identify gene targets for input TFs (Wang et al., <xref ref-type="bibr" rid="B71">2011</xref>). Additional efforts have concentrated on novel TF discovery by analyzing protein mentions and related contextual information in literature to determine whether a given protein might be a TF (Yang et al., <xref ref-type="bibr" rid="B73">2009</xref>).</p>
<p>Matrix factorization based dimensionality reduction techniques such as singular value decomposition (SVD) and non-negative matrix factorization (NMF) have been used to extract latent functional relationships between genes and terms from the biomedical literature. We previously demonstrated that SVD can extract both explicit (direct) and implicit (indirect) relationships between genes, from the biomedical literature with better accuracy than term co-occurrence methods (Homayouni et al., <xref ref-type="bibr" rid="B33">2005</xref>). Subsequently, we applied this approach to prioritize putative TFs for microarray-derived differentially expressed gene sets (Roy et al., <xref ref-type="bibr" rid="B59">2011</xref>) and to prioritize, cluster, and functionally annotate microRNAs (Roy et al., <xref ref-type="bibr" rid="B58">2016</xref>). The main drawback of SVD is that while it is robust in identifying similarities between entities, it is difficult to determine exactly why they are related. This is due to the fact that the columns of factor matrices can contain negative values, needed to accomplish the best fit numerically in a lower dimensional subspace, which do not have a natural interpretation. As an alternative, non-negative matrix factorization (NMF) was developed to simplify the interpretation of factors by restricting the entries in factor matrices to have non-negative values (Lee and Seung, <xref ref-type="bibr" rid="B42">1999</xref>; Berry et al., <xref ref-type="bibr" rid="B11">2007</xref>). The columns can be interpreted as parts of the original data and the high magnitude entities in the like-numbered column pairs of the two factor matrices can be interpreted as a bicluster. NMF has been used successfully to simultaneously cluster genes along with their related terms (Chagoyen et al., <xref ref-type="bibr" rid="B16">2006</xref>; Heinrich et al., <xref ref-type="bibr" rid="B32">2008</xref>; Tjioe et al., <xref ref-type="bibr" rid="B69">2010</xref>).</p>
<p>Both NMF and SVD can only be applied to two mode data. However, biological networks may contain more than two types of entities whose interactions must be analyzed simultaneously. Tensor factorizations are multiway generalizations of matrix factorizations (De Lathauwer et al., <xref ref-type="bibr" rid="B23">2000</xref>; Kolda and Bader, <xref ref-type="bibr" rid="B40">2009</xref>; Qiao et al., <xref ref-type="bibr" rid="B55">2017</xref>). They have been used in the bioinformatics domain (Luo et al., <xref ref-type="bibr" rid="B47">2017</xref>) to integrate and analyze gene expression data from different sources simultaneously (Omberg et al., <xref ref-type="bibr" rid="B50">2007</xref>; Du et al., <xref ref-type="bibr" rid="B25">2009</xref>; Li and Ngom, <xref ref-type="bibr" rid="B45">2010</xref>; Li et al., <xref ref-type="bibr" rid="B44">2011</xref>; Acar et al., <xref ref-type="bibr" rid="B2">2012</xref>). In the text mining domain, multiway decompositions have been used for clustering chatroom data (Acar et al., <xref ref-type="bibr" rid="B1">2005</xref>), scenario discovery (Bader et al., <xref ref-type="bibr" rid="B8">2008b</xref>), discussion tracking (Bader et al., <xref ref-type="bibr" rid="B7">2008a</xref>), personalized web search (Sun et al., <xref ref-type="bibr" rid="B65">2005</xref>), and web link analysis (Kolda et al., <xref ref-type="bibr" rid="B41">2005</xref>).</p>
<p>Previously, we presented a proof of concept method to simultaneously extract and functionally annotate putative transcriptional modules for a small set of interferon modulated genes (Roy et al., <xref ref-type="bibr" rid="B60">2014</xref>). In this study, we aimed to simultaneously extract genes and their regulatory TFs along with the terms that functionally characterize their relationships on a genome-wide scale. We formulated a 3-mode term&#x02009;&#x000D7;&#x02009;gene&#x02009;&#x000D7;&#x02009;TF tensor containing log-scaled frequencies of terms in abstracts shared between genes and TFs. The tensor was decomposed using non-negative tensor factorization (NTF) into sub-tensors at different low rank approximations. The sub-tensors were interpreted as annotated transcriptional modules (ATMs) consisting of genes and TFs along with the terms that annotate the functional relationship between them. We assessed the validity of the ATMs using GO and KEGG annotations and the performance of the NTF method in literature-based discovery using a set of microarray and ChIP-Seq datasets. We demonstrate that the method can predict downstream target genes for TFs as well as GO classifications based on the knowledge in biomedical literature.</p>
</sec>
<sec id="S2" sec-type="materials|methods">
<label>2</label> <title>Materials and Methods</title>
<sec id="S2-1">
<label>2.1</label> <title>Gene&#x02013;TF Document Collection</title>
<p>For every mouse gene, PubMed citations were obtained from the gene2pubmed repository available at NCBI. These citations are assigned either by professional staff at the National Library of Medicine or by the scientific research community via Gene Reference into Function (Gene RIF) portal. Since these citations are manually curated, we expect to have a very high precision for tagging correct citations to genes. We further filtered the non-specific citations by removing PMIDs that referred to more than 10 genes as these citations usually described high-throughput experiments mentioning a large number of genes with little to no significant functional information. After filtering, 21,022 mouse genes with at least one assigned citation remained in the collection. Among these, 1,111 genes were identified as TFs in the AnimalTFDB transcription factor database (Zhang et al., <xref ref-type="bibr" rid="B75">2012</xref>). Out of the remaining 19,911 non TF genes, 7,695 genes were found to share at least one citation with at least one TF out of 994 TFs. A total of 45,229 gene&#x02013;TF pairs with at least one shared citation were identified. For each such gene&#x02013;TF pair, an abstract document was constructed by concatenation of titles and abstracts for each shared citation. A total of 26,781 unique citations were utilized in creating the abstract documents.</p>
</sec>
<sec id="S2-2">
<label>2.2</label> <title>Construction of Term&#x02009;&#x000D7;&#x02009;Gene&#x02009;&#x000D7;&#x02009;TF Tensor</title>
<p>Text to Matrix Generator parser (Zeimpekis and Gallopoulos, <xref ref-type="bibr" rid="B74">2006</xref>) was used to parse terms from the collection of gene&#x02013;TF documents. All punctuation (excluding hyphens and underscores) and capitalization were ignored. In addition, articles and other common, non-distinguishing words were discarded using a stop list. Terms less than three characters in length were filtered out. A total of 106,895 terms remained after all the filtering. A 3-mode term&#x02009;&#x000D7;&#x02009;gene&#x02009;&#x000D7;&#x02009;TF sparse tensor was created where the entries of the tensor were frequencies of terms in abstracts shared between genes and TFs. Tensor construction from gene&#x02013;TF documents is depicted in Figure S1 in Supplementary Material using a toy example with a small number of genes, TFs, and terms. The sparse tensor had 5,451,735 non-zero elements out of possible 817,621,682,850 (106,895&#x02009;&#x000D7;&#x02009;7,695&#x02009;&#x000D7;&#x02009;994) elements resulting in density of 6.66779&#x02009;&#x000D7;&#x02009;10<sup>&#x02212;06</sup>. In order to discount the effect of high frequency common terms in favor of more specific terms that might be better delineators between gene&#x02013;TF combinations, each tensor entry <italic>f<sub>ijk</sub></italic> was scaled and transformed into <italic>l<sub>ijk</sub></italic>:
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mrow><mml:msub><mml:mi>l</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mtext>log</mml:mtext><mml:mn>2</mml:mn></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
where <italic>f<sub>ijk</sub></italic> is the frequency of the <italic>i</italic>th term in the document corresponding to the <italic>j</italic>th gene and the <italic>k</italic>th TF.</p>
</sec>
<sec id="S2-3">
<label>2.3</label> <title>Calculation of Non-Negative Tensor Factorization</title>
<p>Given a 3-mode data tensor <italic>X</italic> of size <italic>m</italic>&#x02009;&#x000D7;&#x02009;<italic>n</italic>&#x02009;&#x000D7;&#x02009;<italic>p</italic>, with <italic>m</italic> mode-1 entities (terms), <italic>n</italic> mode-2 entities (genes), and <italic>p</italic> mode-3 entities (TFs), and a desired approximation rank <italic>k</italic>, the PARAFAC (Harshman, <xref ref-type="bibr" rid="B30">1970</xref>) or CANDECOMP (Carroll and Chang, <xref ref-type="bibr" rid="B15">1970</xref>) model approximates <italic>X</italic> as a sum of <italic>k</italic> rank-1 sub-tensors, each formed by the scaled outer product of a set of three vectors of lengths <italic>m, n</italic>, and <italic>p</italic>. The set of <italic>k</italic> vectors for each mode are usually grouped together in factor matrices <italic>A, B</italic>, and <italic>C</italic> of sizes <italic>m</italic>&#x02009;&#x000D7;&#x02009;<italic>k, n</italic>&#x02009;&#x000D7;&#x02009;<italic>k</italic>, and <italic>p</italic>&#x02009;&#x000D7;&#x02009;<italic>k</italic>, respectively. The columns of the factor matrices are normalized to unit length and the accumulated weight stored in a scaling vector <italic>&#x003BB;</italic>. In addition, a constraint is imposed on the solution such that <italic>&#x003BB;</italic><sub>1</sub>&#x02009;&#x02265;&#x02009;<italic>&#x003BB;</italic><sub>2</sub>&#x02009;&#x02265;&#x02009;&#x02026;&#x02009;&#x02265;&#x02009;<italic>&#x003BB;<sub>k</sub></italic>. The tensor is expressed as:
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mrow><mml:mi>X</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>k</mml:mi></mml:munderover></mml:mstyle><mml:mtext>&#x02009;</mml:mtext><mml:msub><mml:mi>&#x003BB;</mml:mi><mml:mrow><mml:mtext>&#x0200A;</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x02218;</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x02218;</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
where <italic>a<sub>i</sub>, b<sub>i</sub></italic>, and <italic>c<sub>i</sub></italic> represent the <italic>i</italic>th columns of the factor matrices <italic>A, B</italic>, and <italic>C</italic>, respectively; and &#x025E6; denotes the outer product.</p>
<p>A common approach to fitting the PARAFAC model to data is an alternating least squares (ALS) algorithm (Tomasi and Bro, <xref ref-type="bibr" rid="B70">2006</xref>), where one cycles over all factor matrices and performs a least-squares update for one factor matrix while holding all the others constant. We implemented a variant of the PARAFAC model called non-negative tensor factorization (NTF) that constrains the factor matrices to be non-negative. The goal of NTF is to find the best fitting non-negative matrices <inline-formula><mml:math id="M3"><mml:mrow><mml:mi>A</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mi>&#x0211D;</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M4"><mml:mrow><mml:mi>B</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mi>&#x0211D;</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>, and <inline-formula><mml:math id="M5"><mml:mrow><mml:mi>C</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mi>&#x0211D;</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> in the PARAFAC model that fit the data in <italic>X</italic>, corresponding to the following minimization problem:
<disp-formula id="E3"><label>(3)</label><mml:math id="M6"><mml:mrow><mml:munder><mml:mrow><mml:mtext>min</mml:mtext></mml:mrow><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>B</mml:mi><mml:mo>,</mml:mo><mml:mi>C</mml:mi></mml:mrow></mml:munder><mml:mo>&#x02225;</mml:mo><mml:mi>X</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>k</mml:mi></mml:munderover></mml:mstyle><mml:mtext>&#x02009;</mml:mtext><mml:msub><mml:mi>a</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x02218;</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x02218;</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:msub><mml:mn>&#x02225;</mml:mn><mml:mi>F</mml:mi></mml:msub></mml:mrow></mml:math></disp-formula>
where <italic>F</italic> is the Frobenius norm. The norm of a tensor is similar to that of a matrix:
<disp-formula id="E4"><label>(4)</label><mml:math id="M7"><mml:mrow><mml:mn>&#x02225;</mml:mn><mml:mi>X</mml:mi><mml:msup><mml:mo>&#x02225;</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>&#x02261;</mml:mo><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mtext>&#x02009;</mml:mtext><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>A tensor can be matricized or flattened by rearranging the elements in a matrix. <italic>X</italic><sup>(</sup><italic><sup>m</sup></italic><sup>&#x02009;&#x000D7;</sup>&#x02009;<italic><sup>np</sup></italic><sup>)</sup> represents 1-mode matricization of <italic>X</italic>, which is a matrix of size <italic>m</italic>&#x02009;&#x000D7;&#x02009;<italic>np</italic> where the index <italic>n</italic> runs the fastest over the columns and <italic>p</italic> the slowest. The matricized tensor can be expressed as:
<disp-formula id="E5"><label>(5)</label><mml:math id="M8"><mml:mrow><mml:msup><mml:mi>X</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>n</mml:mi><mml:mi>p</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02248;</mml:mo><mml:mi>A</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>C</mml:mi><mml:mo>&#x02299;</mml:mo><mml:mi>B</mml:mi><mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
where &#x02299; represents the Khatri&#x02013;Rao product (Smilde et al., <xref ref-type="bibr" rid="B63">2004</xref>) and &#x02032; denotes the matrix transpose operator. The Kronecker product of matrices <italic>A</italic> and <italic>B</italic> is given by:
<disp-formula id="E6"><label>(6)</label><mml:math id="M10"><mml:mrow><mml:mi>A</mml:mi><mml:mo>&#x02297;</mml:mo><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mn>11</mml:mn></mml:mrow></mml:msub><mml:mi>B</mml:mi></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mn>12</mml:mn></mml:mrow></mml:msub><mml:mi>B</mml:mi></mml:mrow></mml:mtd><mml:mtd><mml:mo>&#x02026;</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mi>B</mml:mi></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mn>21</mml:mn></mml:mrow></mml:msub><mml:mi>B</mml:mi></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mn>22</mml:mn></mml:mrow></mml:msub><mml:mi>B</mml:mi></mml:mrow></mml:mtd><mml:mtd><mml:mo>&#x02026;</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mn>2</mml:mn><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mi>B</mml:mi></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x022F1;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mi>B</mml:mi></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mi>B</mml:mi></mml:mrow></mml:mtd><mml:mtd><mml:mo>&#x02026;</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mi>B</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
and the Khatri&#x02013;Rao product is defined as column wise Kronecker product:
<disp-formula id="E7"><label>(7)</label><mml:math id="M11"><mml:mrow><mml:mi>A</mml:mi><mml:mo>&#x02299;</mml:mo><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mn>11</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mi>b</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mn>12</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mi>b</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mo>&#x02026;</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>b</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mn>21</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mi>b</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mn>22</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mi>b</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mo>&#x02026;</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mn>2</mml:mn><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>b</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x022F1;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mi>b</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mi>b</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mo>&#x02026;</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>b</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>The 2-mode and 3-mode matricizations can be similarly expressed as:
<disp-formula id="E8"><label>(8)</label><mml:math id="M12"><mml:mrow><mml:msup><mml:mi>X</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>m</mml:mi><mml:mi>p</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02248;</mml:mo><mml:mi>B</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>C</mml:mi><mml:mo>&#x02299;</mml:mo><mml:mi>A</mml:mi><mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02032;</mml:mo></mml:msup></mml:mrow></mml:math></disp-formula>
<disp-formula id="E9"><label>(9)</label><mml:math id="M13"><mml:mrow><mml:msup><mml:mi>X</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02248;</mml:mo><mml:mi>C</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>B</mml:mi><mml:mo>&#x02299;</mml:mo><mml:mi>A</mml:mi><mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>To compute a <italic>k</italic>-factorization with NTF, we first initialized <italic>A, B</italic>, and <italic>C</italic> using the absolute values of <italic>k</italic> leading left singular vectors of the 1-mode, 2-mode, and 3-mode matricizations, respectively. This initialization was performed to make the subsequent NTF computation deterministic as well as to provide a potentially suboptimal starting point that may require less iterations to converge (Boutsidis and Gallopoulos, <xref ref-type="bibr" rid="B13">2008</xref>). Each of the 3 matrices was treated as a non-negative matrix factorization (NMF) sub-problem:
<disp-formula id="E10"><label>(10)</label><mml:math id="M14"><mml:mrow><mml:munder><mml:mrow><mml:mtext>min</mml:mtext></mml:mrow><mml:mrow><mml:mi>A</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mi>&#x0211D;</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:munder><mml:mo>&#x02225;</mml:mo><mml:msup><mml:mi>X</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>m</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>n</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo>&#x02212;</mml:mo><mml:mi>A</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>C</mml:mi><mml:mo>&#x02299;</mml:mo><mml:mi>B</mml:mi><mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02032;</mml:mo></mml:msup><mml:msub><mml:mn>&#x02225;</mml:mn><mml:mi>F</mml:mi></mml:msub></mml:mrow></mml:math></disp-formula>
<disp-formula id="E11"><label>(11)</label><mml:math id="M15"><mml:mrow><mml:munder><mml:mrow><mml:mtext>min</mml:mtext></mml:mrow><mml:mrow><mml:mi>B</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mi>&#x0211D;</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:munder><mml:mo>&#x02225;</mml:mo><mml:msup><mml:mi>X</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>n</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>m</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo>&#x02212;</mml:mo><mml:mi>B</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>C</mml:mi><mml:mo>&#x02299;</mml:mo><mml:mi>A</mml:mi><mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02032;</mml:mo></mml:msup><mml:msub><mml:mn>&#x02225;</mml:mn><mml:mi>F</mml:mi></mml:msub></mml:mrow></mml:math></disp-formula>
<disp-formula id="E12"><label>(12)</label><mml:math id="M16"><mml:mrow><mml:munder><mml:mrow><mml:mtext>min</mml:mtext></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mi>&#x0211D;</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:munder><mml:mo>&#x02225;</mml:mo><mml:msup><mml:mi>X</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>p</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>m</mml:mi><mml:mi>n</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo>&#x02212;</mml:mo><mml:mi>C</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>B</mml:mi><mml:mo>&#x02299;</mml:mo><mml:mi>A</mml:mi><mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02032;</mml:mo></mml:msup><mml:msub><mml:mn>&#x02225;</mml:mn><mml:mi>F</mml:mi></mml:msub></mml:mrow></mml:math></disp-formula>
and solved in succession using the multiplicative update rule (Welling and Weber, <xref ref-type="bibr" rid="B72">2001</xref>) modified to incorporate <inline-formula><mml:math id="M17"><mml:mo>&#x02208;</mml:mo></mml:math></inline-formula> for stability. For example, to solve for <italic>A</italic> the update rule is:
<disp-formula id="E13"><label>(13)</label><mml:math id="M18"><mml:mrow><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>&#x003C1;</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>&#x003C1;</mml:mi></mml:mrow></mml:msub><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msup><mml:mi>X</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>m</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>n</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mi>Z</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>&#x003C1;</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>A</mml:mi><mml:msup><mml:mi>Z</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mi>Z</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>&#x003C1;</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mo>&#x02208;</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mi>Z</mml:mi><mml:mo>=</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mi>C</mml:mi><mml:mo>&#x02299;</mml:mo><mml:mi>B</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>The <inline-formula><mml:math id="M19"><mml:mo>&#x02208;</mml:mo></mml:math></inline-formula> is a small number 10<sup>&#x02212;9</sup> added to the denominator in order to add stability to the calculation and guard against introducing a negative number from numerical underflow. The approximation rank <italic>k</italic>&#x02009;&#x0003E;&#x02009;0 of NTF corresponds to the number of 3-way associations (sub-tensors) whose additive contributions approximate the information in the data tensor. Each triad {<italic>a<sub>i</sub>, b<sub>i</sub>, c<sub>i</sub></italic>}, for <italic>i</italic>&#x02009;&#x0003D;&#x02009;1, &#x02026;, <italic>k</italic>, defines scores for a set of terms, genes, and TFs for a particular 3-way association in the corpus. The scaling factor <italic>&#x003BB;<sub>i</sub></italic> (after normalization) indicates the weight of the association for triad <italic>i</italic>. NTF was applied to the term&#x02009;&#x000D7;&#x02009;gene&#x02009;&#x000D7;&#x02009;TF log weighted frequency tensor using functions from MATAB tensor toolbox (Bader and Kolda, <xref ref-type="bibr" rid="B9">2012</xref>). The procedure is described graphically in Figure <xref ref-type="fig" rid="F1">1</xref>.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Overview of the NTF-based procedure. Gene&#x02013;TF documents <bold>(A)</bold> are parsed to construct term&#x02009;&#x000D7;&#x02009;gene&#x02009;&#x000D7;&#x02009;TF tensor <bold>(B)</bold>. The tensor is factorized via NTF to generate non-negative factor matrices for terms, genes, and TFs <bold>(C)</bold>. For a given approximation rank <italic>k</italic>, the entities corresponding to the high magnitude entries in each triad of columns (one from each matrix) <bold>(D)</bold> can be interpreted as an annotated transcriptional module (ATM) comprising of genes and TFs functionally annotated by the corresponding terms <bold>(E)</bold>.</p></caption>
<graphic xlink:href="fbioe-05-00048-g001.tif"/>
</fig>
</sec>
<sec id="S2-4">
<label>2.4</label> <title>Interpretation of Sub-Tensors As Annotated Transcriptional Modules (ATMs)</title>
<p>A <italic>k</italic>-factorization delivers <italic>k</italic> sub-tensors. A sub-tensor <italic>i</italic> can be reconstructed via the outer products of the <italic>i</italic>th columns of the three non-negative factor matrices corresponding to terms, genes and TFs, respectively. For each such triad, the entities corresponding to the high magnitude scores in each column contribute more to the information content in the sub-tensor than the low magnitude ones and are, therefore, deemed more significant. We construed the triplet of sets of such significant genes, TFs, and terms as an annotated transcriptional module (ATM). Each ATM contained genes and TFs representing dominant elements of a putative transcriptional network, along with terms describing the functional interaction between these genes and TFs.</p>
<p>In order for the ATMs to be consistent with observed biological networks in terms of the number of entities, we established upper bounds for the number of genes, TFs, and terms in the ATMs. We observed the distribution of numbers of genes and TFs in 249 KEGG pathways and found a maximum of 184 genes and 21 TFs in a single pathway (after excluding the outliers). For the terms, we chose the upper bound of 300 as this is usually the maximum number of words allowed for publication abstracts and may be adequate to describe a transcriptional network.</p>
<p>For a given column and an upper bound <italic>n</italic>, we first obtained a truncated list <italic>D</italic> containing <italic>n</italic> highest scoring entities. We then calculated the number of significant (high magnitude) entities required to approximate the information content in the list as described in Alter et al. (<xref ref-type="bibr" rid="B5">2000</xref>). Briefly, contributions of each score <italic>g<sub>i</sub></italic> were computed as <inline-formula><mml:math id="M20"><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mstyle scriptlevel='+1'><mml:mfrac><mml:mrow><mml:msub><mml:mi>g</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mi>S</mml:mi></mml:mfrac></mml:mstyle></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="M21"><mml:mrow><mml:mi>S</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup></mml:mstyle><mml:mtext>&#x02009;</mml:mtext><mml:msub><mml:mi>g</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> (sum of all scores). Subsequently, we calculated the normalized entropy of the list as:
<disp-formula id="E14"><label>(14)</label><mml:math id="M22"><mml:mrow><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup></mml:mstyle><mml:mtext>&#x02009;</mml:mtext><mml:msub><mml:mi>p</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
which is the fraction of information content in the list relative to a completely random list of the same size. The number of significant entities <italic>s</italic> was calculated as:
<disp-formula id="E15"><label>(15)</label><mml:math id="M23"><mml:mrow><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x0003E;</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x022C5;</mml:mo><mml:mi>E</mml:mi></mml:mrow></mml:mfrac></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p>
</sec>
<sec id="S2-5">
<label>2.5</label> <title>Performance Evaluation of ATMs</title>
<p>Functional enrichment analysis was performed using two different human curated datasets. For the first set, we downloaded the list of 249 mouse related manually curated pathways and their associated genes present in the Kyoto Encyclopedia of Genes and Genomes (KEGG) (Kanehisa et al., <xref ref-type="bibr" rid="B37">2004</xref>). For the second set, we downloaded the list of 12,418 mouse related Gene Ontology (GO) (Ashburner et al., <xref ref-type="bibr" rid="B6">2000</xref>) categories and their associated genes. The set of genes and TFs belonging to each ATM was evaluated for functional enrichment in the aforementioned KEGG pathways and GO categories using a hypergeometric test as described in Tavazoie et al. (<xref ref-type="bibr" rid="B67">1999</xref>). As a control, we compared the enrichment frequencies of ATMs to a set of 1,000 randomly generated gene&#x02013;TF sets containing 8 genes and 2 TFs from the tensor. The numbers 8 and 2 depict the median number of genes and TFs across all 2,861 ATMs. To create each random gene&#x02013;TF set, 8 genes out of 7,695 genes constituting the tensor were randomly picked. Similarly 2 TFs were randomly picked from 994 TFs in the tensor. This process was repeated 1,000 times.</p>
<p>The area under the curve (AUC) was used as a measure of quality of the terms associated with ATMs. The AUC will have the value of 1 for perfect ranking (all relevant terms at the top), 0.5 for randomly generated ranking, and 0 for the worst possible ranking (all relevant terms at the bottom) (Hanley and McNeal, <xref ref-type="bibr" rid="B29">1982</xref>). The terms in the descriptions of KEGG and GO categories served as the gold standards.</p>
<p>Precision was used as a measure of accuracy of the gene&#x02013;TF associations in an ATM. It was calculated as the ratio of the number of common entities (genes and TFs) between an ATM and the gold standard set, to the number of entities in the ATM. Mathematically, the formula was defined as:
<disp-formula id="E16"><label>(16)</label><mml:math id="M24"><mml:mrow><mml:mfrac><mml:mrow><mml:mo>&#x0007C;</mml:mo><mml:mo>&#x0007B;</mml:mo><mml:mi mathvariant="italic">genes&#x000A0;and&#x000A0;TFs&#x000A0;in&#x000A0;ATM</mml:mi><mml:mo>&#x0007D;</mml:mo><mml:mo>&#x02229;</mml:mo><mml:mo>&#x0007B;</mml:mo><mml:mi mathvariant="italic">genes&#x000A0;and&#x000A0;TFs&#x000A0;in&#x000A0;gold&#x000A0;standard&#x000A0;set</mml:mi><mml:mo>&#x0007D;</mml:mo><mml:mo>&#x0007C;</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x0007C;</mml:mo><mml:mo>&#x0007B;</mml:mo><mml:mi mathvariant="italic">genes&#x000A0;and&#x000A0;TFs&#x000A0;in&#x000A0;ATM</mml:mi><mml:mo>&#x0007D;</mml:mo><mml:mo>&#x0007C;</mml:mo></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>In addition, a set of ATM precision values was tested for significance by computing a right-tailed two-sample Welch&#x02019;s <italic>t</italic>-test (Press, <xref ref-type="bibr" rid="B54">1992</xref>) between the precisions for the ATMs being evaluated, and the precisions for 200&#x02009;&#x000D7;&#x02009;<italic>n</italic> randomly generated gene&#x02013;TF sets from the tensor. For each of <italic>n</italic> ATMs being evaluated, 200 gene&#x02013;TF sets were randomly generated containing the same distribution of genes and TFs as the ATM.</p>
<p>Redundancy between any two sets of entities of the ATMs was computed as the Jaccard coefficient between the sets, which is defined as the ratio of the number of elements in the intersection and the number of elements in the union. The Jaccard coefficient will have a value of 0 for disjoint sets and a value of 1 for duplicate sets.</p>
</sec>
</sec>
<sec id="S3">
<label>3</label> <title>Results</title>
<p>Unlike other matrix factorization approaches, it is computationally difficult to estimate the true rank of a tensor (H&#x000E5;stad, <xref ref-type="bibr" rid="B31">1990</xref>). Therefore, we computed NTF at 16 approximation ranks <italic>k</italic>&#x02009;&#x0003D;&#x02009;1, 2, 3, 5, 10, 15, 20, 25, 30, 50, 100, 200, 300, 500, 700, and 900. For every <italic>k</italic>, NTF required less than 60 iterations to satisfy a tolerance of 10<sup>&#x02212;4</sup> in the relative change of fit. For a given approximation rank <italic>k</italic>, all genes, TFs, and terms in each of the <italic>k</italic> sub-tensors were ranked based on their scores as demonstrated in Figure <xref ref-type="fig" rid="F1">1</xref>. We then used an entropy based method to determine the score threshold and to define relevant genes, TFs, and terms within an annotated transcriptional module (ATM) as described in <xref ref-type="sec" rid="S2">Materials and Methods</xref>. Figure S2 in Supplementary Material shows the frequency distribution of genes, TFs, and terms computed for each of the 2,861 ATMs produced across all <italic>k</italic>s. The median number of TFs (&#x0007E;2) and terms (&#x0007E;60) was stable across all <italic>k</italic> factorizations. In contrast, the median number of genes in ATMs decreased with increasing <italic>k</italic>. A total of 3,608 unique genes, 772 unique TFs, and 11,697 unique terms occurred across all 2,861 ATMs.</p>
<sec id="S3-1">
<label>3.1</label> <title>Tensor Landscape As a Function of Approximation Rank <italic>k</italic></title>
<p>A <italic>k</italic>-factorization expresses the three-way gene regulatory interaction information in the tensor as additive superposition of latent information content in <italic>k</italic> sub-tensors. An approximation with <italic>k</italic>&#x02009;&#x0003D;&#x02009;1 is expected to summarize the information content as the most dominant (or representative) transcriptional network gleaned from the literature, while <italic>k</italic>&#x02009;&#x0003D;&#x02009;2 expresses the information content as superposition of two most dominant transcriptional networks. As <italic>k</italic> increases, the factorizations are expected to reveal more specific networks.</p>
<p>In order to quantitatively examine the information content across different <italic>k</italic>, we calculated a diversity coefficient <italic>d</italic> for each type of entity in the ATMs obtained at various <italic>k</italic> factorizations. For each entity type, we defined the diversity coefficient as the ratio of the number of unique entities in the union of all <italic>k</italic>&#x02009;&#x02265;&#x02009;2 ATMs, and the total number of entities in the tensor. Figure <xref ref-type="fig" rid="F2">2</xref> shows the diversity coefficients for genes, TFs, and terms. As expected, we found that the diversity of TFs, genes, and terms in the ATMs increased with increasing <italic>k</italic>. In addition, we computed the pairwise redundancy using the mean Jaccard coefficient between any two sets of entities in the ATMs for <italic>k</italic>&#x02009;&#x02265;&#x02009;2 (Figure S3 in Supplementary Material). We found that while the overall diversity increases with increasing <italic>k</italic>, the pairwise redundancy between the entities remained constant, and that for any given <italic>k</italic>-factorization, the ATMs were disjoint. Taken together, these results indicate that at higher <italic>k</italic>, we were able to extract unique regulatory modules with more specific functional annotations.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Diversity coefficients of genes, TFs, and terms in ATMs across various <italic>k</italic>-factorizations.</p></caption>
<graphic xlink:href="fbioe-05-00048-g002.tif"/>
</fig>
</sec>
<sec id="S3-2">
<label>3.2</label> <title>Functional Validation of ATMs</title>
<p>To comprehensively evaluate the functional relevance of ATMs generated by our method, we examined if the genes and TFs in the ATM were significantly (p-value&#x02009;&#x02264;&#x02009;0.05, hypergeometric test as described in <xref ref-type="sec" rid="S2">Materials and Methods</xref>) enriched in KEGG or GO categories. We found that more than 94% of the 2,861 ATMs were enriched in at least one KEGG or GO category. This result indicates that tensor factorization across all <italic>k</italic>s produced biologically relevant ATMs. The median number of enriched categories in all ATMs at each <italic>k</italic> value was greater than chance (Figure <xref ref-type="fig" rid="F3">3</xref>). Also, the number of enriched KEGG and GO categories per ATM was higher at lower <italic>k</italic>. This result indicates that at lower <italic>k</italic>, ATM genes, and TFs have a broad range of functions, consistent with our earlier observation that the diversity and specificity of ATMs increased with higher <italic>k</italic> (Figure <xref ref-type="fig" rid="F2">2</xref>; Figure S3 in Supplementary Material). Consistent with these observations, we found that the diversity of enriched KEGG and GO categories in ATMs increased at higher <italic>k</italic> (Figure S4 in Supplementary Material), while the pairwise redundancy between categories remained low as <italic>k</italic> increased (Figure S5 in Supplementary Material). Taken together, these results suggest that the gene&#x02013;TF ATMs are more functionally specific at higher <italic>k</italic>.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Frequency of significantly (<italic>p</italic>&#x02009;&#x02264;&#x02009;0.05) enriched KEGG <bold>(A)</bold> and GO <bold>(B)</bold> categories for ATMs across various <italic>k</italic>-factorizations. As a control, the enrichment frequency of 1,000 randomly (R) generated gene sets are shown at the far right of each panel.</p></caption>
<graphic xlink:href="fbioe-05-00048-g003.tif"/>
</fig>
<p>Interestingly, we found a higher diversity of KEGG pathways than GO categories at all <italic>k</italic> values (Figure S4 in Supplementary Material). This result suggests that KEGG pathways can be more specific than GO categories. Some KEGG and GO categories were very frequent among all ATMs. For instance, among the top ten overrepresented KEGG pathways, five were related to cancer, four related to infection, and one related to cytokine signaling (Table S1 in Supplementary Material). On the other hand, among the top ten overrepresented GO categories, four were related to development, three were related to regulation of gene expression, two related to cell proliferation, and one related to protein phosphorylation (Table S2 in Supplementary Material).</p>
<p>Finally, we evaluated the accuracy of the terms identified by NTF for each ATM by comparing them with KEGG and GO annotations. For each ATM enriched in at least one KEGG pathway or GO category, we created a gold standard using all of the terms in the titles and descriptions of the enriched pathways and categories. We compared the top ATM terms against these gold standards by examining the area under the curve (AUC) of receiver operating characteristics (ROC) curves. We found that more than 80% of the 2,861 ATMs produced AUCs above 0.6. There was no change in median AUC across <italic>k</italic>, albeit the range of AUCs increased with increasing <italic>k</italic> (Figure <xref ref-type="fig" rid="F4">4</xref>). The deviation of the median AUC from the lowest point increased faster with increasing <italic>k</italic> than the deviation from the highest point. This indicates that at higher <italic>k</italic>, some NTF-derived functional annotations are much less consistent with the descriptors of KEGG and GO categories.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Distribution of AUCs for the terms associated with each ATM across various <italic>k</italic>-factorizations against KEGG <bold>(A)</bold> and GO <bold>(B)</bold> category titles and descriptions.</p></caption>
<graphic xlink:href="fbioe-05-00048-g004.tif"/>
</fig>
</sec>
<sec id="S3-3">
<label>3.3</label> <title>Prediction of Gene&#x02013;TF Associations</title>
<p>A unique advantage of our approach is that it enables literature based discovery, i.e., new potential gene&#x02013;TF interactions may be deduced from implied associations in the literature in the absence of direct (experimental) evidence. To evaluate the performance of our method in predicting potentially new gene&#x02013;TF relationships, we used two different types of experimental data as gold standards. Chromatin immunoprecipitation sequencing (ChIP-Seq) is a genome wide technology that identifies direct binding sites for specific transcription factors on gene promoters. On the other hand, microarray expression analysis of tissues with a targeted deletion of specific transcription factor identifies downstream genes whose expression levels are either directly or indirectly altered by the transcription factor.</p>
<p>We used a set of previously published ChIP-Seq data for four transcription factors (<italic>Cebp&#x003B1;, E2f4, Foxa1</italic>, and <italic>Foxa2</italic>) in either liver or 3T3-L1 cells (MacIsaac et al., <xref ref-type="bibr" rid="B48">2010</xref>) as gold standard validation sets. To evaluate our method, we focused only on the ATMs which contained the gold-standard TF. The number of ATMs for the four gold-standard TFs ranged from 11 to 54 and the number of genes and TFs in all the ATMs for the gold-standard TFs ranged from 34 to 112 and 8 to 45, respectively (Table <xref ref-type="table" rid="T1">1</xref>). To calculate the average precision for each gold-standard TF, we compared the union of genes and TFs in all ATMs against the ChIP-seq validation set. The number of genes in the four validation sets ranged from 938 to 8,595 genes. As shown in Table <xref ref-type="table" rid="T1">1</xref>, the average precision of our method ranged between 11% (<italic>Foxa1</italic>, liver) and 91% (<italic>E2f4</italic>, 3T3-L1). Except for <italic>Foxa2</italic>, all gene&#x02013;TF associations extracted by our method were significantly (<italic>p</italic>&#x02009;&#x02264;&#x02009;0.05, Welch&#x02019;s <italic>t</italic>-test) higher than chance. The average precisions for a given TF in two different tissues varied, indicating that binding sites are tissue dependent. These results indicate that NTF is accurate for identifying TF target genes from the biomedical literature. Importantly, we found that 14&#x02013;39% of the target genes predicted by our method were based on implied associations in the literature. An explicit association refers to the cases when a gene and a TF share an abstract, whereas an implicit association is inferred based on shared terms among genes or TFs in the same ATM. This result suggests that our method is useful for making literature based discoveries.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Performance of NTF using ChIP-Seq data sets.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th valign="top" align="left">TF (GEO accession ID)</th>
<th valign="top" align="left">Tissue</th>
<th valign="top" align="center">ATM count</th>
<th valign="top" align="center">Genes in all ATMs</th>
<th valign="top" align="center">TFs in all ATMs</th>
<th valign="top" align="center">Genes in validation set</th>
<th valign="top" align="center">Average precision</th>
<th valign="top" align="center">p-Value</th>
<th valign="top" align="center">Explicit%</th>
<th valign="top" align="center">Implicit%</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Cebpa (GSM427088)</td>
<td align="left" valign="top">Liver</td>
<td align="center" valign="top">13</td>
<td align="center" valign="top">48</td>
<td align="center" valign="top">11</td>
<td align="center" valign="top">8,411</td>
<td align="center" valign="top">0.49</td>
<td align="center" valign="top">7.26E&#x02212;05</td>
<td align="center" valign="top">0.86</td>
<td align="center" valign="top">0.14</td>
</tr>
<tr>
<td align="left" valign="top">Cebpa (GSM427093)</td>
<td align="left" valign="top">3T3-L1 cells</td>
<td align="center" valign="top">13</td>
<td align="center" valign="top">48</td>
<td align="center" valign="top">11</td>
<td align="center" valign="top">2,836</td>
<td align="center" valign="top">0.34</td>
<td align="center" valign="top">9.39E&#x02212;06</td>
<td align="center" valign="top">0.82</td>
<td align="center" valign="top">0.18</td>
</tr>
<tr>
<td align="left" valign="top">E2f4 (GSM427091)</td>
<td align="left" valign="top">Liver</td>
<td align="center" valign="top">22</td>
<td align="center" valign="top">34</td>
<td align="center" valign="top">8</td>
<td align="center" valign="top">7,134</td>
<td align="center" valign="top">0.82</td>
<td align="center" valign="top">5.08E&#x02212;16</td>
<td align="center" valign="top">0.69</td>
<td align="center" valign="top">0.31</td>
</tr>
<tr>
<td align="left" valign="top">E2f4 (GSM427094)</td>
<td align="left" valign="top">3T3-L1 cells</td>
<td align="center" valign="top">22</td>
<td align="center" valign="top">34</td>
<td align="center" valign="top">8</td>
<td align="center" valign="top">8,595</td>
<td align="center" valign="top">0.91</td>
<td align="center" valign="top">4.71E&#x02212;17</td>
<td align="center" valign="top">0.61</td>
<td align="center" valign="top">0.39</td>
</tr>
<tr>
<td align="left" valign="top">Foxa1 (GSM427090)</td>
<td align="left" valign="top">Liver</td>
<td align="center" valign="top">11</td>
<td align="center" valign="top">47</td>
<td align="center" valign="top">9</td>
<td align="center" valign="top">938</td>
<td align="center" valign="top">0.11</td>
<td align="center" valign="top">3.46E&#x02212;02</td>
<td align="center" valign="top">0.71</td>
<td align="center" valign="top">0.29</td>
</tr>
<tr>
<td align="left" valign="top">Foxa2 (GSM427089)</td>
<td align="left" valign="top">Liver</td>
<td align="center" valign="top">54</td>
<td align="center" valign="top">112</td>
<td align="center" valign="top">45</td>
<td align="center" valign="top">6,366</td>
<td align="center" valign="top">0.21</td>
<td align="center" valign="top">9.14E&#x02212;01</td>
<td align="center" valign="top">0.74</td>
<td align="center" valign="top">0.26</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Our literature mining method identifies gene&#x02013;TF associations based on functional information in the biomedical abstracts. Thus, it is possible that some of the gene predictions by our method are not the primary gene targets of the TFs, but rather genes in similar functional pathways whose expression levels are affected indirectly via a downstream TF. To test this possibility, we evaluated our method using four microarray datasets as gold standards. These datasets include differentially expressed genes in different mouse tissues (cerebellum, retina, or choroid plexus) as a consequence of targeted deletions in specific TFs (<italic>Atoh1</italic> (Ha et al., <xref ref-type="bibr" rid="B28">2015</xref>), <italic>Pax6</italic> (Ha et al., <xref ref-type="bibr" rid="B28">2015</xref>), and <italic>Otx2</italic> (Omori et al., <xref ref-type="bibr" rid="B51">2011</xref>; Johansson et al., <xref ref-type="bibr" rid="B36">2013</xref>)). The number of genes and TFs in all the ATMs for the gold-standard TFs ranged from 57 to 152 and 34 to 62, respectively (Table <xref ref-type="table" rid="T2">2</xref>). The number of genes in the validation sets ranged from 2,137 (<italic>Atoh1</italic>, cerebellum) to 11,689 (<italic>Otx2</italic>, choroid plexus). The average precision for all four validation sets ranged from 0.33 to 0.42 and were all significantly (<italic>p</italic>&#x02009;&#x02264;&#x02009;0.05, Welch&#x02019;s <italic>t</italic>-test) higher than chance. Importantly, approximately 27&#x02013;63% of the predictions from our method were based on implied associations extracted from the literature. Since transcriptional regulation by TFs are tissue specific, we narrowed the ATMs to those which explicitly mentioned the experimental tissue. We calculated average precision for ATMs that contained either one or two keywords associated with the tissue. In all but one case (<italic>Otx2</italic>, choroid), the average precision values improved when we considered only the tissue-relevant ATMs.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Performance of NTF using microarray data sets.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th valign="top" align="left">TF knockout (GEO ID)</th>
<th valign="top" align="left">Tissue</th>
<th valign="top" align="center">&#x00023; of ATMs</th>
<th valign="top" align="center">Total &#x00023; genes</th>
<th valign="top" align="center">Total &#x00023; TFs</th>
<th valign="top" align="center">&#x00023; Genes in validation set</th>
<th valign="top" align="center" colspan="3">Average precision<hr/></th>
<th valign="top" align="center" colspan="2">Base ATM associations (%)<hr/></th>
</tr><tr>
<th valign="top" align="center" colspan="6"/>
<th valign="top" align="center">Base (p-value)</th>
<th valign="top" align="center">Base&#x02009;&#x0002B;&#x02009;(keyword)</th>
<th valign="top" align="center">Base&#x02009;&#x0002B;&#x02009;(keywords)</th>
<th valign="top" align="center">Explicit</th>
<th valign="top" align="center">Implicit</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Atoh1</td>
<td align="left" valign="top">Cerebellum</td>
<td align="center" valign="top">38</td>
<td align="center" valign="top">73</td>
<td align="center" valign="top">36</td>
<td align="center" valign="top">2,137</td>
<td align="center" valign="top">0.33 (<italic>p</italic>&#x02009;&#x0003C;&#x02009;9.07<italic>E</italic>&#x02212;07)</td>
<td align="center" valign="top">0.50 (cerebellum)</td>
<td align="center" valign="top">0.71 (cerebellum, rhombic)</td>
<td align="center" valign="top">0.37</td>
<td align="center" valign="top">0.63</td>
</tr>
<tr>
<td align="left" valign="top">Pax6</td>
<td align="left" valign="top">Cerebellum</td>
<td align="center" valign="top">74</td>
<td align="center" valign="top">152</td>
<td align="center" valign="top">62</td>
<td align="center" valign="top">3,036</td>
<td align="center" valign="top">0.38 (<italic>p</italic>&#x02009;&#x0003C;&#x02009;2.87<italic>E</italic>&#x02212;18)</td>
<td align="center" valign="top">0.54 (cerebellum)</td>
<td align="center" valign="top">0.57 (cerebellum, rhombic)</td>
<td align="center" valign="top">0.73</td>
<td align="center" valign="top">0.27</td>
</tr>
<tr>
<td align="left" valign="top">Otx2 (GSE21900)</td>
<td align="left" valign="top">Retina</td>
<td align="center" valign="top">41</td>
<td align="center" valign="top">57</td>
<td align="center" valign="top">34</td>
<td align="center" valign="top">6,181</td>
<td align="center" valign="top">0.36 (<italic>p</italic>&#x02009;&#x0003C;&#x02009;5.31<italic>E</italic>&#x02212;06)</td>
<td align="center" valign="top">0.523 (retina)</td>
<td align="center" valign="top">&#x02013;</td>
<td align="center" valign="top">0.51</td>
<td align="center" valign="top">0.49</td>
</tr>
<tr>
<td align="left" valign="top">Otx2 (GSE27630)</td>
<td align="left" valign="top">Choroid plexus</td>
<td align="center" valign="top">41</td>
<td align="center" valign="top">57</td>
<td align="center" valign="top">34</td>
<td align="center" valign="top">11,689</td>
<td align="center" valign="top">0.42 (<italic>p</italic>&#x02009;&#x0003C;&#x02009;3.1<italic>E</italic>&#x02212;05)</td>
<td align="center" valign="top">0.28 (choroid)</td>
<td align="center" valign="top">&#x02013;</td>
<td align="center" valign="top">0.68</td>
<td align="center" valign="top">0.32</td>
</tr>
</tbody>
</table>
<table-wrap-foot><p><italic>Average precision values were determined for all ATMs which included the target TF (base) or base modules that also included one or two keywords associated with the experimental tissue. An explicit association refers to the cases when a gene and a TF share an abstract, whereas an implicit association is inferred among genes or TFs in the same ATM</italic>.</p></table-wrap-foot></table-wrap>
<p>Next, we focused on one dataset (<italic>Atoh1</italic>) to carefully examine the characteristics of each ATM. As indicated above, the average precision across all 38 <italic>Atoh1</italic> ATMs was 0.33. However, the precision of individual <italic>Atoh1</italic> ATMs ranged between 0 and 0.85 (Figure <xref ref-type="fig" rid="F5">5</xref>). In general, higher <italic>k</italic> produced higher precision, indicating that higher classification specificity is correlated with precision. Interestingly, precision was not correlated with whether <italic>Atoh1</italic> was ranked first in the ATM. The average precision was 0.23 for the 17 ATMs in which <italic>Atoh1</italic> was top ranked. In contrast, <italic>Atoh1</italic> was first ranked in only one of the top three ATMs with the highest precision (Figure <xref ref-type="fig" rid="F5">5</xref>). Finally, six ATMs had a precision of 0. Upon examination, we found that these ATMs were significantly enirched (<italic>p</italic>&#x02009;&#x02264;&#x02009;0.05) for &#x0201C;auditory receptor cell differentiation&#x0201D; and other GO categories related to ear development. Consistent with these GO classifications, the terms in the ATMs were related to hair cell differentiation and development. Thus, it is likely that the precision of these ATMs is low because the gold standard was derived from microarray experiments using cerebellar tissue rather than auditory tissue.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Precision values for all 38 <italic>Atoh1</italic> ATMs. <sup>&#x02021;</sup> denotes ATMs in which <italic>Atoh1</italic> was ranked first; red bars denote ATMs that include the term &#x0201C;cerebellum&#x0201D;; green bars denote ATMs that include the term &#x0201C;rhombic&#x0201D;; brown bars denote ATMs which include both &#x0201C;cerebellum&#x0201D; and &#x0201C;rhombic&#x0201D; as terms; and blue bars denote ATMs that include neither &#x0201C;cerebellum&#x0201D; or &#x0201C;rhombic&#x0201D; as terms.</p></caption>
<graphic xlink:href="fbioe-05-00048-g005.tif"/>
</fig>
<p>Finally, as a benchmark, we compared the predictions of our method with those obtained by RegNetwork (Liu et al., <xref ref-type="bibr" rid="B46">2015</xref>) for the same experimental ChIP-Seq and TF knockout microarray datasets described above as gold standards. RegNetwork is a database containing target predictions for TFs, which have been sourced from more than 20 interaction databases. Tables S3 and S4 in Supplementary Material show the precisions obtained by NTF along with those obtained by RegNetwork predictions. With the ChIP-Seq gold standards, we found that the average precision of NTF method was 0.48 compared to 0.38 for RegNetwork. NTF outperformed in 4 out of 6 ChIP-Seq datasets. For the two other datasets, RegNetwork outperforms but only marginally. In contrast, with the TF knockout microarray gold standards, the average precision of NTF (0.37) was better than that of RegNetwork (0.16). NTF precision was higher for 3 out of the 4 microarray datasets. Importantly, RegNetwork predicted targets for <italic>Foxa1</italic> and <italic>Otx2</italic> did not match any targets in the gold-standard datasets, whereas NTF identified target genes with 0.11 and 0.42 precision, respectively (Tables S3 and S4 in Supplementary Material).</p>
</sec>
<sec id="S3-4">
<label>3.4</label> <title>Prediction of Functional Classifications</title>
<p>Current methods for functional interpretation of high-throughput genomic data rely on manually curated knowledge bases, such as GO, KEGG, and a variety of other resources. It is generally accepted that the rate of manual curation is not sufficient for the amount of genomic data that is being generated in various species (Baumgartner et al., <xref ref-type="bibr" rid="B10">2007</xref>). Moreover, there is a quantifiable bias in manually curated knowledge bases toward more popular genes and a substantial drift in the annotation of genes over time (Gillis and Pavlidis, <xref ref-type="bibr" rid="B27">2013</xref>). Previous work have focused on using text-mining approaches to enhance curation of GO and other knowledge bases (Chagoyen et al., <xref ref-type="bibr" rid="B16">2006</xref>; Couto et al., <xref ref-type="bibr" rid="B21">2006</xref>; Thomas et al., <xref ref-type="bibr" rid="B68">2015</xref>; Peng et al., <xref ref-type="bibr" rid="B53">2016</xref>).</p>
<p>Here, we examine the performance of our NTF approach in predicting (suggesting) GO classifications. GO predictions were based on two approaches: (1) Guilt-by-Association (GBA) or (2) term mapping. In the GBA approach, new genes or TFs were assigned to a GO category which is significantly enriched for one or more ATMs. These new genes or TFs are part of the ATMs but have not been explicitly assigned to the enriched GO category by the curators. In the term mapping approach, novel and more specific GO categories for a given ATM were predicted based on the number of terms from the ATM that overlapped with GO category descriptions. These new GO categories were not significantly enriched for the ATM under consideration. The results for both approaches were evaluated by manual examination of the biomedical literature.</p>
<p>In order to evaluate the GBA approach, for the 38 ATMs containing <italic>Atoh1</italic>, we randomly chose 3 GO categories that were found to be significantly enriched in some of those ATMs (Table <xref ref-type="table" rid="T3">3</xref>). For example, for category &#x0201C;Axon guidance&#x0201D; that was significantly enriched for 24 ATMs containing a total of 104 genes and TFs, only 16 genes and/or TFs were explicitly assigned to the category by the GO curators. Upon manual examination of sentences in Medline abstracts, we found that out of the 104 genes and TFs, 29 were functionally related to axon guidance. Among the 29, 11 were already assigned to the GO category but the rest (18) were not. The latter are candidates for assignment to the &#x0201C;axon guidance&#x0201D; category, potentially increasing the assignment of new genes and TFs to the category by 1.6-fold. Five of the existing GO assignments could not be validated through manual evaluation. Overall, for three randomly selected significantly enriched GO categories (Table <xref ref-type="table" rid="T3">3</xref>), the GBA method increased the assignment of new genes by an average of 1.2-fold. The average precision for the GO assignment by this method was 0.27 (0.14&#x02013;0.38 range). Notably, on average, 52% (26&#x02013;100% range) of the existing GO annotations were not validated by our manual analysis, indicating that there is substantial error in GO curation (Table <xref ref-type="table" rid="T3">3</xref>).</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Significantly enriched GO categories for <italic>Atoh1</italic> modules for evaluating guilt-by-association (module membership).</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th valign="top" align="left">GO category (&#x00023; of enriched ATMs)</th>
<th valign="top" align="center">ATM genes &#x00023;</th>
<th valign="top" align="center" colspan="2">GO curated<hr/></th>
<th valign="top" align="center" colspan="2">Manual validation<hr/></th>
</tr><tr>
<th valign="top" align="center" colspan="2"/>
<th valign="top" align="center">Assigned</th>
<th valign="top" align="center">Unassigned</th>
<th valign="top" align="center">Assigned</th>
<th valign="top" align="center">Unassigned</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Axon guidance (24)</td>
<td align="center" valign="top">104</td>
<td align="center" valign="top">16</td>
<td align="center" valign="top">88</td>
<td align="center" valign="top">11</td>
<td align="center" valign="top">18</td>
</tr>
<tr>
<td align="left" valign="top">Neuron migration (15)</td>
<td align="center" valign="top">78</td>
<td align="center" valign="top">15</td>
<td align="center" valign="top">63</td>
<td align="center" valign="top">11</td>
<td align="center" valign="top">19</td>
</tr>
<tr>
<td align="left" valign="top">Neural crest cell migration (1)</td>
<td align="center" valign="top">35</td>
<td align="center" valign="top">4</td>
<td align="center" valign="top">31</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">5</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In order to evaluate the term mapping approach, for the 38 ATMs containing <italic>Atoh1</italic>, we predicted 7 novel GO categories whose names and descriptions were found to have high overlap with the terms from some of those ATMs (Table <xref ref-type="table" rid="T4">4</xref>) but were not significantly enriched for any ATM. For example, for category &#x0201C;Spinal cord oligodendrocyte cell differentiation&#x0201D; whose description had high overlap with the terms of 3 ATMs containing a total of 44 genes and TFs, only 1 gene or TF was explicitly assigned to the category by the GO curators. This low number contributes to the reason the category was not significantly enriched for any of the 3 ATMs. Upon manual examination of sentences in Medline abstracts, we found that out of the 44 genes and TFs, 9 were functionally related to spinal cord oligodendrocyte cell differentiation. Among these, 1 was already assigned to the GO category but the rest (8) were not. The latter genes and TFs are candidates for assignment to the category, potentially increasing the assignment of new genes and TFs to the category by 8-fold, which make the category a plausible candidate for being significantly enriched for the ATMs whose terms have high overlap with the category description. On average, 3.9-fold (1- to 8-fold range) more genes were assigned to novel predicted GO categories, but with an average precision of 0.15 (0.06&#x02013;0.23 range) (Table <xref ref-type="table" rid="T4">4</xref>). Most of these categories were more specific than the currently enriched categories.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Predicted GO categories for <italic>Atoh1</italic> modules based on term mapping.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th valign="top" align="left">GO category (&#x00023; of ATMs with high term overlap)</th>
<th valign="top" align="left">ATM genes &#x00023;</th>
<th valign="top" align="center" colspan="2">GO curated<hr/></th>
<th valign="top" align="center" colspan="2">Manual validation<hr/></th>
</tr><tr>
<th valign="top" align="center" colspan="2"/>
<th valign="top" align="center">Assigned</th>
<th valign="top" align="center">Unassigned</th>
<th valign="top" align="center">Assigned</th>
<th valign="top" align="center">Unassigned</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Spinal cord oligodendrocyte cell differentiation (3)</td>
<td align="center" valign="top">44</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">43</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">8</td>
</tr>
<tr>
<td align="left" valign="top">Central nervous system vasculogenesis (1)</td>
<td align="center" valign="top">35</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">34</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">2</td>
</tr>
<tr>
<td align="left" valign="top">Cajal&#x02013;Retzius cell differentiation (2)</td>
<td align="center" valign="top">24</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">24</td>
<td align="center" valign="top">&#x02013;</td>
<td align="center" valign="top">2</td>
</tr>
<tr>
<td align="left" valign="top">Roof plate formation (3)</td>
<td align="center" valign="top">13</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">12</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">2</td>
</tr>
<tr>
<td align="left" valign="top">Schwann cell development (3)</td>
<td align="center" valign="top">15</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">15</td>
<td align="center" valign="top">&#x02013;</td>
<td align="center" valign="top">2</td>
</tr>
<tr>
<td align="left" valign="top">Cerebellar granular layer development (6)</td>
<td align="center" valign="top">34</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">34</td>
<td align="center" valign="top">&#x02013;</td>
<td align="center" valign="top">7</td>
</tr>
<tr>
<td align="left" valign="top">Cerebellar Purkinje cell layer development (7)</td>
<td align="center" valign="top">34</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">34</td>
<td align="center" valign="top">&#x02013;</td>
<td align="center" valign="top">5</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In general, the terms associated with ATMs provide a greater level of functional specificity than structured categories in ontologies. For example, Figure <xref ref-type="fig" rid="F6">6</xref>C shows the genes, TFs, and terms associated with one ATM in which <italic>Atoh1</italic> is top ranked. As a comparison, the enriched GO and KEGG categories are also displayed. Whereas the GO categories indicate that this group of genes and TFs are associated with inner ear development and auditory receptor cell differentiation, the terms (shown in <italic>italics</italic> below) in the ATM suggest that they are involved in <italic>differentiation</italic> of the <italic>sensory hair cells</italic> in the <italic>cochlea</italic> located in the <italic>organ</italic> of <italic>Corti</italic>. <italic>Atoh1</italic> is a basic helix loop helix (<italic>bhlh</italic>) <italic>domain transcription factor</italic> involved in <italic>early development</italic>. Targeted deletion of <italic>Atoh1</italic> gene in <italic>mice results</italic> in <italic>loss</italic> of <italic>hair cells</italic> in the <italic>organ</italic> of <italic>Corti</italic> (Chonko et al., <xref ref-type="bibr" rid="B20">2013</xref>). Several gene symbols also appear in the top ranked terms such as <italic>Sox2, Notch, Jag1, Prox1</italic>, and <italic>Math1</italic>. Interestingly, <italic>Math1</italic> is the alias for <italic>Atoh1</italic>. Direct evidence suggests that <italic>Notch1</italic> and <italic>Jag1 signaling pathway</italic> is required for activation of <italic>Sox2</italic> and <italic>Atoh1 expression</italic> (Neves et al., <xref ref-type="bibr" rid="B49">2011</xref>). Activation of <italic>Atoh1</italic> by <italic>Sox2 transcription factor</italic> is required for <italic>hair cell development</italic> in <italic>cochlea</italic> with respect to both expansion of the <italic>progenitor cells</italic> in the <italic>cochlear epithelium</italic> and initation of <italic>hair cell differentiation</italic> (Kiernan et al., <xref ref-type="bibr" rid="B39">2005</xref>; Kempfle et al., <xref ref-type="bibr" rid="B38">2016</xref>). Conversely, <italic>Prox1 transcription factor</italic> directly suppresses <italic>Atoh1 expression</italic>. <italic>Sox2</italic> is <italic>expressed</italic> in type 2 <italic>vestibular hair cells</italic> and in <italic>supporting cochlear</italic> and <italic>vestibular epithelium</italic> (Hume et al., <xref ref-type="bibr" rid="B34">2007</xref>).</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Screen shot of TREMEL (Transcription REgulatory Modules Extracted from Literature) tool. <bold>(A)</bold> Search feature allows the user to query genes, TFs, or terms across all 2,861 ATMs. In addition, complex queries involving a combination of any of the three entities can be performed by adding another query box. <bold>(B)</bold> Overview display shows all of the ATMs which relate to the query with respect to the rank of the query entity, <italic>k</italic>, and ATM number. <bold>(C)</bold>&#x02009;Displays the ranked genes, TFs, and terms for the selected ATM in panel <bold>(B)</bold>. In addition, the enriched GO and KEGG categories are displayed in order for the user to quickly compare the NTF terms against human curation.</p></caption>
<graphic xlink:href="fbioe-05-00048-g006.tif"/>
</fig>
</sec>
<sec id="S3-5">
<label>3.5</label> <title>TREMEL Web Tool</title>
<p>In order to facilitate manual examination of individual ATMs, e.g., the aforementioned <italic>Atoh1</italic> ATMs, the publicly available web tool called Transcriptional Regulatory Modules Extracted from Literature (TREMEL) was developed. It is accessible at <uri xlink:href="http://binf1.memphis.edu/tremel">http://binf1.memphis.edu/tremel</uri>. TREMEL provides a searchable interface for the genes, TFs, and terms for all 2,861 ATMs (Figure <xref ref-type="fig" rid="F6">6</xref>).</p>
<p>The user can query the tool with either genes, TFs or terms, or a combination of any of the 3 types of entities in additional query boxes. The entity type can be selected from the drop down list to the right of each search box. Clicking the &#x0201C;&#x0002B;&#x0201D; button to the left of the first search box opens an additional search box. A maximum of 3 search boxes are allowed. The gene and TF queries need to be the official symbols designated by the National Center for Biotechnology Information (NCBI). Only one symbol is allowed per search box. The term query can be any single keyword such as &#x0201C;cancer,&#x0201D; &#x0201C;neuron,&#x0201D; &#x0201C;transcription,&#x0201D; etc. All queries are case insensitive. The output of the tool consists of two panels.</p>
<p>The top panel is comprised of a 3-dimensional interactive plot that shows all ATMs containing the search box entities, as points. The axes of the plot correspond to the NTF approximation rank <italic>k</italic>, the ATM&#x00023;, and the rank of the queried entity (first search box only) in the ATMs. An ATM can be selected in the panel by clicking on its corresponding point in the 3-D plot. The color of the selected point changes to red and the colors of the remaining points corresponding to all other ATMs are depicted in terms of similarity to the selected ATM. The most similar ATMs are colored in shades of red while the least similar ones are colored in shades of blue. The similarity between any two ATMs is calculated as the Jaccard coefficient between the sets of genes and TFs in the respective ATMs. Upon initial search completion, one ATM is preselected and colored in red. This initial selection is performed in a manner such that the ATM with the lowest queried entity rank is picked. Ties are resolved in favor of the ATM with the lowest NTF approximation rank.</p>
<p>The bottom panel contains several sub-panels, each corresponding to an ATM point in the first panel. The top sub-panel corresponds to the selected ATM, and the remaining ATMs are ordered according to their similarity to the selected ATM. Each sub-panel displays the ranked genes, TFs and terms of the corresponding ATM, as well as the enriched GO categories and KEGG pathways. Clicking a sub-panel expands it to display its contents, and closes the previously open sub-panel. The contents of only one sub-panel are viewable at a time.</p>
</sec>
</sec>
<sec id="S4" sec-type="discussion">
<label>4</label> <title>Discussion</title>
<p>We have shown for the first time that NTF can be used effectively to simultaneously extract and functionally annotate transcriptional networks from the biomedical literature on a genomic scale. NTF is able to generalize and overcome data sparsity to produce interpretable low rank approximations. The ATMs comprised of genes, TFs, and terms were evaluated by several approaches. More than 94% of the gene&#x02013;TF ATMs were enriched in at least one KEGG or GO category. In addition, there were considerable overlap (AUC values above 0.6) between the NTF-derived terms and the descriptions of the enriched KEGG and GO categories. Importantly, NTF identified more specific terms related to genes and TFs than what is currently available in KEGG and GO databases (Table <xref ref-type="table" rid="T4">4</xref>; Figure <xref ref-type="fig" rid="F6">6</xref>). Thus, our approach provides more flexibility for scientists to search with a broader range of terms to identify relevant transcriptional networks directly from the literature. To assist researchers in exploring and discovering functional insights about transcriptional networks, we developed the web tool TREMEL, which provides a searchable interface for the genes, TFs, and terms for all 2,861 ATMs (Figure <xref ref-type="fig" rid="F6">6</xref>).</p>
<p>We validated our NTF derived ATMs using a number of different data sources, including GO (manually curated), KEGG (manually curated), ChIP-Seq (experimental), and TF knockout microarray (experimental). We also compared our gene&#x02013;TF association predictions with those obtained by RegNetwork and found that overall, the NTF method produced comparable or better precision (specificity) compared to the RegNetwork. However, calculating recall (sensitivity) using these data sources may not be appropriate because none of the data sources are 100% accurate and complete. Manual curation is known to be incomplete (Baumgartner et al., <xref ref-type="bibr" rid="B10">2007</xref>) and high-throughput experimental datasets are highly variable for technical reasons. Indeed, in Table <xref ref-type="table" rid="T4">4</xref> we show a few examples where NTF term annotations were more specific and relevant than the enriched GO categories. In addition, we showed that NTF performed comparably to or better than human curated RegNetwork, which is an aggregated database of TF&#x02013;gene interactions. However, unlike the NTF approach, predicted targets of RegNetwork for two out of ten transcription factors examined did not match any targets in the gold-standard datasets (Tables S3 and S4 in Supplementary Material). Finally, since the NTF literature mining approach extracts conceptual and functional associations between TFs and genes, we would not expect very high precision with any of the experimental benchmarks for several reasons. Genomic assays are highly tissue specific, whereas the functional associations in the literature may not necessarily be focused on a specific tissue. ChIP-seq assays detect only direct interaction between TFs with gene promoters, whereas the microarray experiments identify both direct and indirect TF&#x02013;gene associations. Finally, both ChIP-Seq and microarray experiments are prone to false-discovery due to multiple hypothesis testing.</p>
<p>As estimation of the true tensor rank is computationally difficult, we opted for a more exploratory approach where we evaluated factorization at several approximation ranks (<italic>k</italic> ranging from 1 to 900). KEGG/GO pathway divergence analysis (Figure S3 in Supplementary Material), and experimental data validation (Figure <xref ref-type="fig" rid="F5">5</xref>) and manual evaluation indicated that more specific pathways are delineated at higher <italic>k</italic>. Consistent with this notion, we found that the median number of enriched KEGG/GO categories were substantially lower at higher <italic>k</italic> (Figure <xref ref-type="fig" rid="F3">3</xref>). On the other hand, this seemingly low number of enrichment could be partially due to shortcomings of KEGG and GO annotations as we demonstrated after manual analysis (Tables <xref ref-type="table" rid="T3">3</xref> and <xref ref-type="table" rid="T4">4</xref>) and previously documented by Gillis and Pavlidis (<xref ref-type="bibr" rid="B27">2013</xref>).</p>
<p>The interpretation of sub-tensors in our method allows for redundancy in ATMs, where certain entities may occur in multiple ATMs from the same <italic>k</italic>-factorization. We believe that this makes biological sense as a TF might be involved in different functions or pathways depending on the set of genes with which it interacts. For instance, analysis of <italic>Atoh1</italic> ATMs revealed that some of its ATMs were related to early cerebellar development, whereas other ATMs were related to hair cell development in the auditory system. This approach is an improvement over previous work by our group and others using NMF (Heinrich et al., <xref ref-type="bibr" rid="B32">2008</xref>; Tjioe et al., <xref ref-type="bibr" rid="B69">2010</xref>), where the factors were interpreted to have only disjoint biclusters and a set of genes were associated with the best set of associated terms.</p>
<p>Taken together, our results demonstrate that NTF is a promising technique to simultaneously extract genes, TFs and related terms to identify and predict gene regulatory networks from the biomedical literature. The method and tool presented here are not intended to be comprehensive (high recall) nor a replacement of high-throughput experiments. However, we have shown that it is a valuable exploratory tool which can help researchers make new mechanistic discoveries and help human curators to quickly annotate gene function based on the vast amount of knowledge in the biomedical literature. In addition, our work sets the stage to apply this technique to other areas in systems biology such as simultaneous extraction of terms with miRNAs and their target genes, small molecules and protein targets, as well as drugs and diseases.</p>
</sec>
<sec id="S5" sec-type="author-contributor">
<title>Author Contributions</title>
<p>SR and RH designed the research and wrote the manuscript. SR and BM analyzed the results. DY developed the web tool. MB and L-YD contributed mathematical and statistical methods. DG contributed biological experimental data for validation. RH supervised the research and assisted with the interpretation of results. All authors reviewed the manuscript.</p>
</sec>
<sec id="S6">
<title>Conflict of Interest Statement</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
</body>
<back>
<fn-group>
<fn fn-type="financial-disclosure">
<p><bold>Funding.</bold> This work was supported in part by the Memphis Research Consortium and the University of Memphis Center for Translational Informatics.</p></fn>
</fn-group>
<sec id="S7" sec-type="supplementary-material">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at <uri xlink:href="http://journal.frontiersin.org/article/10.3389/fbioe.2017.00048/full&#x00023;supplementary-material">http://journal.frontiersin.org/article/10.3389/fbioe.2017.00048/full&#x00023;supplementary-material</uri>.</p>
<supplementary-material xlink:href="Data_Sheet_1.PDF" id="SM1" mimetype="applicationn/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Acar</surname> <given-names>E.</given-names></name> <name><surname>Camtepe</surname> <given-names>S. A.</given-names></name> <name><surname>Krishnamoorthy</surname> <given-names>M. S.</given-names></name> <name><surname>Yener</surname> <given-names>B.</given-names></name></person-group> (<year>2005</year>). <article-title>&#x0201C;Modeling and multiway analysis of chatroom tensors,&#x0201D;</article-title> in <source>Intelligence and Security Informatics</source>, eds <person-group person-group-type="editor"><name><surname>Kantor</surname> <given-names>P.</given-names></name> <name><surname>Muresan</surname> <given-names>G.</given-names></name> <name><surname>Roberts</surname> <given-names>F.</given-names></name> <name><surname>Zeng</surname> <given-names>D. D.</given-names></name> <name><surname>Wang</surname> <given-names>F.-Y.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name> <etal/></person-group> (<publisher-loc>Berlin-Heidelberg</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>256</fpage>&#x02013;<lpage>268</lpage>.</citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Acar</surname> <given-names>E.</given-names></name> <name><surname>Plopper</surname> <given-names>G. E.</given-names></name> <name><surname>Yener</surname> <given-names>B.</given-names></name></person-group> (<year>2012</year>). <article-title>Coupled analysis of in vitro and histology tissue samples to quantify structure-function relationship</article-title>. <source>PLoS ONE</source> <volume>7</volume>:<fpage>e32227</fpage>.<pub-id pub-id-type="doi">10.1371/journal.pone.0032227</pub-id><pub-id pub-id-type="pmid">22479315</pub-id></citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Aerts</surname> <given-names>S.</given-names></name> <name><surname>Haeussler</surname> <given-names>M.</given-names></name> <name><surname>Van Vooren</surname> <given-names>S.</given-names></name> <name><surname>Griffith</surname> <given-names>O. L.</given-names></name> <name><surname>Hulpiau</surname> <given-names>P.</given-names></name> <name><surname>Jones</surname> <given-names>S. J.</given-names></name> <etal/></person-group> (<year>2008</year>). <article-title>Text-mining assisted regulatory annotation</article-title>. <source>Genome Biol.</source> <volume>9</volume>, <fpage>R31</fpage>.<pub-id pub-id-type="doi">10.1186/gb-2008-9-2-r31</pub-id><pub-id pub-id-type="pmid">18271954</pub-id></citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alako</surname> <given-names>B.</given-names></name> <name><surname>Veldhoven</surname> <given-names>A.</given-names></name> <name><surname>Van Baal</surname> <given-names>S.</given-names></name> <name><surname>Jelier</surname> <given-names>R.</given-names></name> <name><surname>Verhoeven</surname> <given-names>S.</given-names></name> <name><surname>Rullmann</surname> <given-names>T.</given-names></name> <etal/></person-group> (<year>2005</year>). <article-title>CoPub mapper: mining MEDLINE based on search term co-publication</article-title>. <source>BMC Bioinformatics</source> <volume>6</volume>:<fpage>51</fpage>.<pub-id pub-id-type="doi">10.1186/1471-2105-6-51</pub-id><pub-id pub-id-type="pmid">15760478</pub-id></citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alter</surname> <given-names>O.</given-names></name> <name><surname>Brown</surname> <given-names>P. O.</given-names></name> <name><surname>Botstein</surname> <given-names>D.</given-names></name></person-group> (<year>2000</year>). <article-title>Singular value decomposition for genome-wide expression data processing and modeling</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A.</source> <volume>97</volume>, <fpage>10101</fpage>&#x02013;<lpage>10106</lpage>.<pub-id pub-id-type="doi">10.1073/pnas.97.18.10101</pub-id><pub-id pub-id-type="pmid">10963673</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ashburner</surname> <given-names>M.</given-names></name> <name><surname>Ball</surname> <given-names>C. A.</given-names></name> <name><surname>Blake</surname> <given-names>J. A.</given-names></name> <name><surname>Botstein</surname> <given-names>D.</given-names></name> <name><surname>Butler</surname> <given-names>H.</given-names></name> <name><surname>Cherry</surname> <given-names>J. M.</given-names></name> <etal/></person-group> (<year>2000</year>). <article-title>Gene ontology: tool for the unification of biology</article-title>. <source>Nat. Genet.</source> <volume>25</volume>, <fpage>25</fpage>.<pub-id pub-id-type="doi">10.1038/75556</pub-id></citation></ref>
<ref id="B7"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Bader</surname> <given-names>B. W.</given-names></name> <name><surname>Berry</surname> <given-names>M. W.</given-names></name> <name><surname>Browne</surname> <given-names>M.</given-names></name></person-group> (<year>2008a</year>). <article-title>&#x0201C;Discussion tracking in Enron email using PARAFAC,&#x0201D;</article-title> in <source>Survey of Text Mining II</source>, eds <person-group person-group-type="editor"><name><surname>Berry</surname> <given-names>M. W.</given-names></name> <name><surname>Castellanos</surname> <given-names>M.</given-names></name></person-group> (<publisher-loc>London, UK</publisher-loc>: <publisher-name>Springer-Verlag</publisher-name>), <fpage>147</fpage>&#x02013;<lpage>163</lpage>.</citation></ref>
<ref id="B8"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Bader</surname> <given-names>B.</given-names></name> <name><surname>Puretskiy</surname> <given-names>A.</given-names></name> <name><surname>Berry</surname> <given-names>M.</given-names></name></person-group> (<year>2008b</year>). <article-title>&#x0201C;Scenario discovery using nonnegative tensor factorization,&#x0201D;</article-title> in <source>Progress in Pattern Recognition, Image Analysis and Applications</source>, eds <person-group person-group-type="editor"><name><surname>Ruiz-Schulcloper</surname> <given-names>J.</given-names></name> <name><surname>Kropatsch</surname> <given-names>W.</given-names></name></person-group> (<publisher-loc>Berlin-Heidelberg</publisher-loc>: <publisher-name>Springer-Verlag</publisher-name>), <fpage>791</fpage>&#x02013;<lpage>805</lpage>.</citation></ref>
<ref id="B9"><citation citation-type="web"><person-group person-group-type="author"><name><surname>Bader</surname> <given-names>B. W.</given-names></name> <name><surname>Kolda</surname> <given-names>T. G.</given-names></name></person-group> (<year>2012</year>). <source>MATLAB Tensor Toolbox Version 2.5</source>. Available at: <uri xlink:href="http://www.sandia.gov/&#x0007E;tgkolda/TensorToolbox/index-2.5.html">http://www.sandia.gov/&#x0007E;tgkolda/TensorToolbox/index-2.5.html</uri></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Baumgartner</surname> <given-names>W. A.</given-names></name> <name><surname>Cohen</surname> <given-names>K. B.</given-names></name> <name><surname>Fox</surname> <given-names>L. M.</given-names></name> <name><surname>Acquaah-Mensah</surname> <given-names>G.</given-names></name> <name><surname>Hunter</surname> <given-names>L.</given-names></name></person-group> (<year>2007</year>). <article-title>Manual curation is not sufficient for annotation of genomic databases</article-title>. <source>Bioinformatics</source> <volume>23</volume>, <fpage>i41</fpage>&#x02013;<lpage>i48</lpage>.<pub-id pub-id-type="doi">10.1093/bioinformatics/btm229</pub-id><pub-id pub-id-type="pmid">17646325</pub-id></citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Berry</surname> <given-names>M. W.</given-names></name> <name><surname>Browne</surname> <given-names>M.</given-names></name> <name><surname>Langville</surname> <given-names>A. N.</given-names></name> <name><surname>Pauca</surname> <given-names>V. P.</given-names></name> <name><surname>Plemmons</surname> <given-names>R. J.</given-names></name></person-group> (<year>2007</year>). <article-title>Algorithms and applications for approximate nonnegative matrix factorization</article-title>. <source>Comput. Stat. Data Anal.</source> <volume>52</volume>, <fpage>155</fpage>&#x02013;<lpage>173</lpage>.<pub-id pub-id-type="doi">10.1016/j.csda.2006.11.006</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Blagosklonny</surname> <given-names>M. V.</given-names></name> <name><surname>Pardee</surname> <given-names>A. B.</given-names></name></person-group> (<year>2002</year>). <article-title>Conceptual biology: unearthing the gems</article-title>. <source>Nature</source> <volume>416</volume>, <fpage>373</fpage>&#x02013;<lpage>373</lpage>.<pub-id pub-id-type="doi">10.1038/416373a</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Boutsidis</surname> <given-names>C.</given-names></name> <name><surname>Gallopoulos</surname> <given-names>E.</given-names></name></person-group> (<year>2008</year>). <article-title>SVD based initialization: a head start for nonnegative matrix factorization</article-title>. <source>Pattern Recognit.</source> <volume>41</volume>, <fpage>1350</fpage>&#x02013;<lpage>1362</lpage>.<pub-id pub-id-type="doi">10.1016/j.patcog.2007.09.010</pub-id></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Burkart</surname> <given-names>M. F.</given-names></name> <name><surname>Wren</surname> <given-names>J. D.</given-names></name> <name><surname>Herschkowitz</surname> <given-names>J. I.</given-names></name> <name><surname>Perou</surname> <given-names>C. M.</given-names></name> <name><surname>Garner</surname> <given-names>H. R.</given-names></name></person-group> (<year>2007</year>). <article-title>Clustering microarray-derived gene lists through implicit literature relationships</article-title>. <source>Bioinformatics</source> <volume>23</volume>, <fpage>1995</fpage>.<pub-id pub-id-type="doi">10.1093/bioinformatics/btm261</pub-id><pub-id pub-id-type="pmid">17537751</pub-id></citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Carroll</surname> <given-names>J. D.</given-names></name> <name><surname>Chang</surname> <given-names>J. J.</given-names></name></person-group> (<year>1970</year>). <article-title>Analysis of individual differences in multidimensional scaling via an N-way generalization of Eckart-Young decomposition</article-title>. <source>Psychometrika</source> <volume>35</volume>, <fpage>283</fpage>&#x02013;<lpage>319</lpage>.<pub-id pub-id-type="doi">10.1007/BF02310791</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chagoyen</surname> <given-names>M.</given-names></name> <name><surname>Carmona-Saez</surname> <given-names>P.</given-names></name> <name><surname>Shatkay</surname> <given-names>H.</given-names></name> <name><surname>Carazo</surname> <given-names>J. M.</given-names></name> <name><surname>Pascual-Montano</surname> <given-names>A.</given-names></name></person-group> (<year>2006</year>). <article-title>Discovering semantic features in the literature: a foundation for building functional associations</article-title>. <source>BMC Bioinformatics</source> <volume>7</volume>:<fpage>41</fpage>.<pub-id pub-id-type="doi">10.1186/1471-2105-7-41</pub-id><pub-id pub-id-type="pmid">16438716</pub-id></citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>G.</given-names></name> <name><surname>Cairelli</surname> <given-names>M. J.</given-names></name> <name><surname>Kilicoglu</surname> <given-names>H.</given-names></name> <name><surname>Shin</surname> <given-names>D.</given-names></name> <name><surname>Rindflesch</surname> <given-names>T. C.</given-names></name></person-group> (<year>2014</year>). <article-title>Augmenting microarray data with literature-based knowledge to enhance gene regulatory network inference</article-title>. <source>PLoS Comput. Biol.</source> <volume>10</volume>:<fpage>e1003666</fpage>.<pub-id pub-id-type="doi">10.1371/journal.pcbi.1003666</pub-id><pub-id pub-id-type="pmid">24921649</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Sharp</surname> <given-names>B.</given-names></name></person-group> (<year>2004</year>). <article-title>Content-rich biological network constructed by mining PubMed abstracts</article-title>. <source>BMC Bioinformatics</source> <volume>5</volume>:<fpage>147</fpage>.<pub-id pub-id-type="doi">10.1186/1471-2105-5-147</pub-id><pub-id pub-id-type="pmid">15473905</pub-id></citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>K.</given-names></name> <name><surname>Rajewsky</surname> <given-names>N.</given-names></name></person-group> (<year>2007</year>). <article-title>The evolution of gene regulation by transcription factors and microRNAs</article-title>. <source>Nat. Rev. Genet.</source> <volume>8</volume>, <fpage>93</fpage>&#x02013;<lpage>103</lpage>.<pub-id pub-id-type="doi">10.1038/nrg1990</pub-id><pub-id pub-id-type="pmid">17230196</pub-id></citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chonko</surname> <given-names>K. T.</given-names></name> <name><surname>Jahan</surname> <given-names>I.</given-names></name> <name><surname>Stone</surname> <given-names>J.</given-names></name> <name><surname>Wright</surname> <given-names>M. C.</given-names></name> <name><surname>Fujiyama</surname> <given-names>T.</given-names></name> <name><surname>Hoshino</surname> <given-names>M.</given-names></name> <etal/></person-group> (<year>2013</year>). <article-title>Atoh1 directs hair cell differentiation and survival in the late embryonic mouse inner ear</article-title>. <source>Dev. Biol.</source> <volume>381</volume>, <fpage>401</fpage>&#x02013;<lpage>410</lpage>.<pub-id pub-id-type="doi">10.1016/j.ydbio.2013.06.022</pub-id><pub-id pub-id-type="pmid">23796904</pub-id></citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Couto</surname> <given-names>F. M.</given-names></name> <name><surname>Silva</surname> <given-names>M. J.</given-names></name> <name><surname>Lee</surname> <given-names>V.</given-names></name> <name><surname>Dimmer</surname> <given-names>E.</given-names></name> <name><surname>Camon</surname> <given-names>E.</given-names></name> <name><surname>Apweiler</surname> <given-names>R.</given-names></name> <etal/></person-group> (<year>2006</year>). <article-title>GOAnnotator: linking protein go annotations to evidence text</article-title>. <source>J. Biomed. Discov. Collab.</source> <volume>1</volume>, <fpage>19</fpage>.<pub-id pub-id-type="doi">10.1186/1747-5333-1-19</pub-id><pub-id pub-id-type="pmid">17181854</pub-id></citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Davidson</surname> <given-names>E. H.</given-names></name></person-group> (<year>2010</year>). <article-title>Emerging properties of animal gene regulatory networks</article-title>. <source>Nature</source> <volume>468</volume>, <fpage>911</fpage>&#x02013;<lpage>920</lpage>.<pub-id pub-id-type="doi">10.1038/nature09645</pub-id><pub-id pub-id-type="pmid">21164479</pub-id></citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>De Lathauwer</surname> <given-names>L.</given-names></name> <name><surname>De Moor</surname> <given-names>B.</given-names></name> <name><surname>Vandewalle</surname> <given-names>J.</given-names></name></person-group> (<year>2000</year>). <article-title>A multilinear singular value decomposition</article-title>. <source>SIAM J. Matrix Anal. Appl.</source> <volume>21</volume>, <fpage>1253</fpage>&#x02013;<lpage>1278</lpage>.<pub-id pub-id-type="doi">10.1137/S0895479896305696</pub-id></citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Djebali</surname> <given-names>S.</given-names></name> <name><surname>Davis</surname> <given-names>C.</given-names></name> <name><surname>Merkel</surname> <given-names>A.</given-names></name> <name><surname>Dobin</surname> <given-names>A.</given-names></name> <name><surname>Lassmann</surname> <given-names>T.</given-names></name> <name><surname>Mortazavi</surname> <given-names>A.</given-names></name> <etal/></person-group> (<year>2012</year>). <article-title>Landscape of transcription in human cells</article-title>. <source>Nature</source> <volume>489</volume>, <fpage>101</fpage>&#x02013;<lpage>108</lpage>.<pub-id pub-id-type="doi">10.1038/nature11233</pub-id><pub-id pub-id-type="pmid">22955620</pub-id></citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Du</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>S. W.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name></person-group> (<year>2009</year>). <article-title>Tumor classification using high-order gene expression profiles based on multilinear ICA</article-title>. <source>Adv. Bioinformatics</source> <volume>2009</volume>, <fpage>926450</fpage>.<pub-id pub-id-type="doi">10.1155/2009/926450</pub-id><pub-id pub-id-type="pmid">19956422</pub-id></citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gerstein</surname> <given-names>M. B.</given-names></name> <name><surname>Kundaje</surname> <given-names>A.</given-names></name> <name><surname>Hariharan</surname> <given-names>M.</given-names></name> <name><surname>Landt</surname> <given-names>S. G.</given-names></name> <name><surname>Yan</surname> <given-names>K.-K.</given-names></name> <name><surname>Cheng</surname> <given-names>C.</given-names></name> <etal/></person-group> (<year>2012</year>). <article-title>Architecture of the human regulatory network derived from encode data</article-title>. <source>Nature</source> <volume>489</volume>, <fpage>91</fpage>&#x02013;<lpage>100</lpage>.<pub-id pub-id-type="doi">10.1038/nature11245</pub-id><pub-id pub-id-type="pmid">22955619</pub-id></citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gillis</surname> <given-names>J.</given-names></name> <name><surname>Pavlidis</surname> <given-names>P.</given-names></name></person-group> (<year>2013</year>). <article-title>Assessing identity, redundancy and confounds in gene ontology annotations over time</article-title>. <source>Bioinformatics</source> <volume>29</volume>, <fpage>476</fpage>&#x02013;<lpage>482</lpage>.<pub-id pub-id-type="doi">10.1093/bioinformatics/bts727</pub-id><pub-id pub-id-type="pmid">23297035</pub-id></citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ha</surname> <given-names>T.</given-names></name> <name><surname>Swanson</surname> <given-names>D.</given-names></name> <name><surname>Larouche</surname> <given-names>M.</given-names></name> <name><surname>Glenn</surname> <given-names>R.</given-names></name> <name><surname>Weeden</surname> <given-names>D.</given-names></name> <name><surname>Zhang</surname> <given-names>P.</given-names></name> <etal/></person-group> (<year>2015</year>). <article-title>CbGRiTS: cerebellar gene regulation in time and space</article-title>. <source>Dev. Biol.</source> <volume>397</volume>, <fpage>18</fpage>&#x02013;<lpage>30</lpage>.<pub-id pub-id-type="doi">10.1016/j.ydbio.2014.09.032</pub-id><pub-id pub-id-type="pmid">25446528</pub-id></citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hanley</surname> <given-names>J. A.</given-names></name> <name><surname>McNeal</surname> <given-names>B. J.</given-names></name></person-group> (<year>1982</year>). <article-title>A simple generalization of the area under the ROC curve to multiple class classification problems</article-title>. <source>Radiology</source> <volume>143</volume>, <fpage>29</fpage>&#x02013;<lpage>36</lpage>.<pub-id pub-id-type="doi">10.1148/radiology.143.1.7063747</pub-id></citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Harshman</surname> <given-names>R. A.</given-names></name></person-group> (<year>1970</year>). <article-title>Foundations of the PARAFAC procedure: models and conditions for an &#x0201C;explanatory&#x0201D; multi-modal factor analysis</article-title>. <source>UCLA Work. Pap. Phon.</source> <volume>16</volume>, <fpage>1</fpage>&#x02013;<lpage>84</lpage>.</citation></ref>
<ref id="B31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>H&#x000E5;stad</surname> <given-names>J.</given-names></name></person-group> (<year>1990</year>). <article-title>Tensor rank is np-complete</article-title>. <source>J. Algorithms</source> <volume>11</volume>, <fpage>644</fpage>&#x02013;<lpage>654</lpage>.<pub-id pub-id-type="doi">10.1016/0196-6774(90)90014-6</pub-id></citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Heinrich</surname> <given-names>K. E.</given-names></name> <name><surname>Berry</surname> <given-names>M. W.</given-names></name> <name><surname>Homayouni</surname> <given-names>R.</given-names></name></person-group> (<year>2008</year>). <article-title>Gene tree labeling using nonnegative matrix factorization on biomedical literature</article-title>. <source>Comput. Intell. Neurosci.</source> <volume>2008</volume>, <fpage>2</fpage>.<pub-id pub-id-type="doi">10.1155/2008/276535</pub-id><pub-id pub-id-type="pmid">18431447</pub-id></citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Homayouni</surname> <given-names>R.</given-names></name> <name><surname>Heinrich</surname> <given-names>K.</given-names></name> <name><surname>Wei</surname> <given-names>L.</given-names></name> <name><surname>Berry</surname> <given-names>M. W.</given-names></name></person-group> (<year>2005</year>). <article-title>Gene clustering by latent semantic indexing of MEDLINE abstracts</article-title>. <source>Bioinformatics</source> <volume>21</volume>, <fpage>104</fpage>.<pub-id pub-id-type="doi">10.1093/bioinformatics/bth464</pub-id><pub-id pub-id-type="pmid">15308538</pub-id></citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hume</surname> <given-names>C. R.</given-names></name> <name><surname>Bratt</surname> <given-names>D. L.</given-names></name> <name><surname>Oesterle</surname> <given-names>E. C.</given-names></name></person-group> (<year>2007</year>). <article-title>Expression of LHX3 and SOX2 during mouse inner ear development</article-title>. <source>Gene Expr. Patterns</source> <volume>7</volume>, <fpage>798</fpage>&#x02013;<lpage>807</lpage>.<pub-id pub-id-type="doi">10.1016/j.modgep.2007.05.002</pub-id><pub-id pub-id-type="pmid">17604700</pub-id></citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jenssen</surname> <given-names>T. K.</given-names></name> <name><surname>Laegreid</surname> <given-names>A.</given-names></name> <name><surname>Komorowski</surname> <given-names>J.</given-names></name> <name><surname>Hovig</surname> <given-names>E.</given-names></name></person-group> (<year>2001</year>). <article-title>A literature network of human genes for high-throughput analysis of gene expression</article-title>. <source>Nat. Genet.</source> <volume>28</volume>, <fpage>21</fpage>&#x02013;<lpage>28</lpage>.<pub-id pub-id-type="doi">10.1038/88213</pub-id><pub-id pub-id-type="pmid">11326270</pub-id></citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Johansson</surname> <given-names>P. A.</given-names></name> <name><surname>Irmler</surname> <given-names>M.</given-names></name> <name><surname>Acampora</surname> <given-names>D.</given-names></name> <name><surname>Beckers</surname> <given-names>J.</given-names></name> <name><surname>Simeone</surname> <given-names>A.</given-names></name> <name><surname>G&#x000F6;tz</surname> <given-names>M.</given-names></name></person-group> (<year>2013</year>). <article-title>The transcription factor Otx2 regulates choroid plexus development and function</article-title>. <source>Development</source> <volume>140</volume>, <fpage>1055</fpage>&#x02013;<lpage>1066</lpage>.<pub-id pub-id-type="doi">10.1242/dev.090860</pub-id></citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kanehisa</surname> <given-names>M.</given-names></name> <name><surname>Goto</surname> <given-names>S.</given-names></name> <name><surname>Kawashima</surname> <given-names>S.</given-names></name> <name><surname>Okuno</surname> <given-names>Y.</given-names></name> <name><surname>Hattori</surname> <given-names>M.</given-names></name></person-group> (<year>2004</year>). <article-title>The KEGG resource for deciphering the genome</article-title>. <source>Nucleic Acids Res.</source> <volume>32</volume>, <fpage>D277</fpage>&#x02013;<lpage>D280</lpage>.<pub-id pub-id-type="doi">10.1093/nar/gkh063</pub-id><pub-id pub-id-type="pmid">14681412</pub-id></citation></ref>
<ref id="B38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kempfle</surname> <given-names>J. S.</given-names></name> <name><surname>Turban</surname> <given-names>J. L.</given-names></name> <name><surname>Edge</surname> <given-names>A. S.</given-names></name></person-group> (<year>2016</year>). <article-title>Sox2 in the differentiation of cochlear progenitor cells</article-title>. <source>Sci. Rep.</source> <volume>6</volume>, <fpage>23293</fpage>.<pub-id pub-id-type="doi">10.1038/srep23293</pub-id><pub-id pub-id-type="pmid">26988140</pub-id></citation></ref>
<ref id="B39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kiernan</surname> <given-names>A. E.</given-names></name> <name><surname>Pelling</surname> <given-names>A. L.</given-names></name> <name><surname>Leung</surname> <given-names>K. K.</given-names></name> <name><surname>Tang</surname> <given-names>A. S.</given-names></name> <name><surname>Bell</surname> <given-names>D. M.</given-names></name> <name><surname>Tease</surname> <given-names>C.</given-names></name> <etal/></person-group> (<year>2005</year>). <article-title>Sox2 is required for sensory organ development in the mammalian inner ear</article-title>. <source>Nature</source> <volume>434</volume>, <fpage>1031</fpage>&#x02013;<lpage>1035</lpage>.<pub-id pub-id-type="doi">10.1038/nature03487</pub-id><pub-id pub-id-type="pmid">15846349</pub-id></citation></ref>
<ref id="B40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kolda</surname> <given-names>T. G.</given-names></name> <name><surname>Bader</surname> <given-names>B. W.</given-names></name></person-group> (<year>2009</year>). <article-title>Tensor decompositions and applications</article-title>. <source>SIAM Rev.</source> <volume>51</volume>, <fpage>455</fpage>.<pub-id pub-id-type="doi">10.1137/07070111X</pub-id></citation></ref>
<ref id="B41"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Kolda</surname> <given-names>T. G.</given-names></name> <name><surname>Bader</surname> <given-names>B. W.</given-names></name> <name><surname>Kenny</surname> <given-names>J. P.</given-names></name></person-group> (<year>2005</year>). <article-title>&#x0201C;Higher-order web link analysis using multilinear algebra,&#x0201D;</article-title> in <conf-name>Fifth IEEE International Conference on Data Mining</conf-name> (<conf-loc>Houston, TX</conf-loc>: <conf-sponsor>IEEE Computer Society</conf-sponsor>), <fpage>8</fpage>.</citation></ref>
<ref id="B42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>D. D.</given-names></name> <name><surname>Seung</surname> <given-names>H. S.</given-names></name></person-group> (<year>1999</year>). <article-title>Learning the parts of objects by non-negative matrix factorization</article-title>. <source>Nature</source> <volume>401</volume>, <fpage>788</fpage>&#x02013;<lpage>791</lpage>.<pub-id pub-id-type="doi">10.1038/44565</pub-id><pub-id pub-id-type="pmid">10548103</pub-id></citation></ref>
<ref id="B43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Levine</surname> <given-names>M.</given-names></name> <name><surname>Tjian</surname> <given-names>R.</given-names></name></person-group> (<year>2003</year>). <article-title>Transcription regulation and animal diversity</article-title>. <source>Nature</source> <volume>424</volume>, <fpage>147</fpage>&#x02013;<lpage>151</lpage>.<pub-id pub-id-type="doi">10.1038/nature01763</pub-id><pub-id pub-id-type="pmid">12853946</pub-id></citation></ref>
<ref id="B44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>W.</given-names></name> <name><surname>Liu</surname> <given-names>C. C.</given-names></name> <name><surname>Zhang</surname> <given-names>T.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Waterman</surname> <given-names>M. S.</given-names></name> <name><surname>Zhou</surname> <given-names>X. J.</given-names></name></person-group> (<year>2011</year>). <article-title>Integrative analysis of many weighted co-expression networks using tensor computation</article-title>. <source>PLoS Comput. Biol.</source> <volume>7</volume>:<fpage>e1001106</fpage>.<pub-id pub-id-type="doi">10.1371/journal.pcbi.1001106</pub-id><pub-id pub-id-type="pmid">21698123</pub-id></citation></ref>
<ref id="B45"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Ngom</surname> <given-names>A.</given-names></name></person-group> (<year>2010</year>). <article-title>&#x0201C;Non-negative matrix and tensor factorization based classification of clinical microarray gene expression data,&#x0201D;</article-title> in <conf-name>2010 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</conf-name> (<conf-loc>Hong Kong</conf-loc>: <conf-sponsor>IEEE Computer Society</conf-sponsor>), <fpage>438</fpage>&#x02013;<lpage>443</lpage>.</citation></ref>
<ref id="B46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Z.-P.</given-names></name> <name><surname>Wu</surname> <given-names>C.</given-names></name> <name><surname>Miao</surname> <given-names>H.</given-names></name> <name><surname>Wu</surname> <given-names>H.</given-names></name></person-group> (<year>2015</year>). <article-title>RegNetwork: an integrated database of transcriptional and post-transcriptional regulatory networks in human and mouse</article-title>. <source>Database</source> <volume>2015</volume>, <fpage>bav095</fpage>.<pub-id pub-id-type="doi">10.1093/database/bav095</pub-id><pub-id pub-id-type="pmid">26424082</pub-id></citation></ref>
<ref id="B47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>F.</given-names></name> <name><surname>Szolovits</surname> <given-names>P.</given-names></name></person-group> (<year>2017</year>). <article-title>Tensor factorization toward precision medicine</article-title>. <source>Brief. Bioinform.</source> <volume>18</volume>, <fpage>511</fpage>&#x02013;<lpage>514</lpage>.<pub-id pub-id-type="doi">10.1093/bib/bbw026</pub-id><pub-id pub-id-type="pmid">26994614</pub-id></citation></ref>
<ref id="B48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>MacIsaac</surname> <given-names>K. D.</given-names></name> <name><surname>Lo</surname> <given-names>K. A.</given-names></name> <name><surname>Gordon</surname> <given-names>W.</given-names></name> <name><surname>Motola</surname> <given-names>S.</given-names></name> <name><surname>Mazor</surname> <given-names>T.</given-names></name> <name><surname>Fraenkel</surname> <given-names>E.</given-names></name></person-group> (<year>2010</year>). <article-title>A quantitative model of transcriptional regulation reveals the influence of binding location on expression</article-title>. <source>PLoS Comput. Biol.</source> <volume>6</volume>:<fpage>e1000773</fpage>.<pub-id pub-id-type="doi">10.1371/journal.pcbi.1000773</pub-id><pub-id pub-id-type="pmid">20442865</pub-id></citation></ref>
<ref id="B49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Neves</surname> <given-names>J.</given-names></name> <name><surname>Parada</surname> <given-names>C.</given-names></name> <name><surname>Chamizo</surname> <given-names>M.</given-names></name> <name><surname>Gir&#x000E1;ldez</surname> <given-names>F.</given-names></name></person-group> (<year>2011</year>). <article-title>Jagged 1 regulates the restriction of Sox2 expression in the developing chicken inner ear: a mechanism for sensory organ specification</article-title>. <source>Development</source> <volume>138</volume>, <fpage>735</fpage>&#x02013;<lpage>744</lpage>.<pub-id pub-id-type="doi">10.1242/dev.060657</pub-id><pub-id pub-id-type="pmid">21266409</pub-id></citation></ref>
<ref id="B50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Omberg</surname> <given-names>L.</given-names></name> <name><surname>Golub</surname> <given-names>G. H.</given-names></name> <name><surname>Alter</surname> <given-names>O.</given-names></name></person-group> (<year>2007</year>). <article-title>A tensor higher-order singular value decomposition for integrative analysis of DNA microarray data from different studies</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A.</source> <volume>104</volume>, <fpage>18371</fpage>.<pub-id pub-id-type="doi">10.1073/pnas.0709146104</pub-id></citation></ref>
<ref id="B51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Omori</surname> <given-names>Y.</given-names></name> <name><surname>Katoh</surname> <given-names>K.</given-names></name> <name><surname>Sato</surname> <given-names>S.</given-names></name> <name><surname>Muranishi</surname> <given-names>Y.</given-names></name> <name><surname>Chaya</surname> <given-names>T.</given-names></name> <name><surname>Onishi</surname> <given-names>A.</given-names></name> <etal/></person-group> (<year>2011</year>). <article-title>Analysis of transcriptional regulatory pathways of photoreceptor genes by expression profiling of the Otx2-deficient retina</article-title>. <source>PLoS ONE</source> <volume>6</volume>:<fpage>e19685</fpage>.<pub-id pub-id-type="doi">10.1371/journal.pone.0019685</pub-id><pub-id pub-id-type="pmid">21602925</pub-id></citation></ref>
<ref id="B52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pan</surname> <given-names>H.</given-names></name> <name><surname>Zuo</surname> <given-names>L.</given-names></name> <name><surname>Choudhary</surname> <given-names>V.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Leow</surname> <given-names>S. H.</given-names></name> <name><surname>Chong</surname> <given-names>F. T.</given-names></name> <etal/></person-group> (<year>2004</year>). <article-title>Dragon TF association miner: a system for exploring transcription factor associations through text-mining</article-title>. <source>Nucleic Acids Res.</source> <volume>32</volume>(<issue>Suppl. 2</issue>), <fpage>W230</fpage>&#x02013;<lpage>W234</lpage>.<pub-id pub-id-type="doi">10.1093/nar/gkh484</pub-id><pub-id pub-id-type="pmid">15215386</pub-id></citation></ref>
<ref id="B53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peng</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>Extending gene ontology with gene association networks</article-title>. <source>Bioinformatics</source> <volume>32</volume>, <fpage>1185</fpage>&#x02013;<lpage>1194</lpage>.<pub-id pub-id-type="doi">10.1093/bioinformatics/btv712</pub-id><pub-id pub-id-type="pmid">26644414</pub-id></citation></ref>
<ref id="B54"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Press</surname> <given-names>W. H.</given-names></name></person-group> (<year>1992</year>). <source>Numerical Recipes in C: The Art of Scientific Computing</source>. <publisher-loc>Cambridge</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>.</citation></ref>
<ref id="B55"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qiao</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Chen</surname> <given-names>Y.-w.</given-names></name> <name><surname>Liu</surname> <given-names>Z.-P.</given-names></name></person-group> (<year>2017</year>). <article-title>Multi-dimensional data representation using linear tensor coding</article-title>. <source>IET Image Process.</source> <volume>11</volume>, <fpage>492</fpage>&#x02013;<lpage>501</lpage>.<pub-id pub-id-type="doi">10.1049/iet-ipr.2016.0795</pub-id></citation></ref>
<ref id="B56"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rebholz-Schuhmann</surname> <given-names>D.</given-names></name> <name><surname>Oellrich</surname> <given-names>A.</given-names></name> <name><surname>Hoehndorf</surname> <given-names>R.</given-names></name></person-group> (<year>2012</year>). <article-title>Text-mining solutions for biomedical research: enabling integrative biology</article-title>. <source>Nat. Rev. Genet.</source> <volume>13</volume>, <fpage>829</fpage>&#x02013;<lpage>839</lpage>.<pub-id pub-id-type="doi">10.1038/nrg3337</pub-id><pub-id pub-id-type="pmid">23150036</pub-id></citation></ref>
<ref id="B57"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rodr&#x000ED;guez-Penagos</surname> <given-names>C.</given-names></name> <name><surname>Salgado</surname> <given-names>H.</given-names></name> <name><surname>Mart&#x000ED;nez-Flores</surname> <given-names>I.</given-names></name> <name><surname>Collado-Vides</surname> <given-names>J.</given-names></name></person-group> (<year>2007</year>). <article-title>Automatic reconstruction of a bacterial regulatory network using natural language processing</article-title>. <source>BMC Bioinformatics</source> <volume>8</volume>:<fpage>293</fpage>.<pub-id pub-id-type="doi">10.1186/1471-2105-8-293</pub-id><pub-id pub-id-type="pmid">17683642</pub-id></citation></ref>
<ref id="B58"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Roy</surname> <given-names>S.</given-names></name> <name><surname>Curry</surname> <given-names>B. C.</given-names></name> <name><surname>Madahian</surname> <given-names>B.</given-names></name> <name><surname>Homayouni</surname> <given-names>R.</given-names></name></person-group> (<year>2016</year>). <article-title>Prioritization, clustering and functional annotation of micrornas using latent semantic indexing of medline abstracts</article-title>. <source>BMC Bioinformatics</source> <volume>17</volume>:<fpage>350</fpage>.<pub-id pub-id-type="doi">10.1186/s12859-016-1223-2</pub-id><pub-id pub-id-type="pmid">27766940</pub-id></citation></ref>
<ref id="B59"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Roy</surname> <given-names>S.</given-names></name> <name><surname>Heinrich</surname> <given-names>K.</given-names></name> <name><surname>Phan</surname> <given-names>V.</given-names></name> <name><surname>Berry</surname> <given-names>M. W.</given-names></name> <name><surname>Homayouni</surname> <given-names>R.</given-names></name></person-group> (<year>2011</year>). <article-title>Latent semantic indexing of PubMed abstracts for identification of transcription factor candidates from microarray derived gene sets</article-title>. <source>BMC Bioinformatics</source> <volume>12</volume>(<issue>Suppl. 10</issue>):<fpage>S19</fpage>.<pub-id pub-id-type="doi">10.1186/1471-2105-12-S10-S19</pub-id><pub-id pub-id-type="pmid">22165960</pub-id></citation></ref>
<ref id="B60"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Roy</surname> <given-names>S.</given-names></name> <name><surname>Homayouni</surname> <given-names>R.</given-names></name> <name><surname>Berry</surname> <given-names>M. W.</given-names></name> <name><surname>Puretskiy</surname> <given-names>A. A.</given-names></name></person-group> (<year>2014</year>). <article-title>&#x0201C;Nonnegative tensor factorization of biomedical literature for analysis of genomic data,&#x0201D;</article-title> in <source>In Data Mining for Service</source>, ed. <person-group person-group-type="editor"><name><surname>Yada</surname> <given-names>K.</given-names></name></person-group> (<publisher-loc>Berlin-Heidelberg</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>97</fpage>&#x02013;<lpage>110</lpage>.</citation></ref>
<ref id="B61"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rzhetsky</surname> <given-names>A.</given-names></name> <name><surname>Iossifov</surname> <given-names>I.</given-names></name> <name><surname>Koike</surname> <given-names>T.</given-names></name> <name><surname>Krauthammer</surname> <given-names>M.</given-names></name> <name><surname>Kra</surname> <given-names>P.</given-names></name> <name><surname>Morris</surname> <given-names>M.</given-names></name> <etal/></person-group> (<year>2004</year>). <article-title>Geneways: a system for extracting, analyzing, visualizing, and integrating molecular pathway data</article-title>. <source>J. Biomed. Inform.</source> <volume>37</volume>, <fpage>43</fpage>&#x02013;<lpage>53</lpage>.<pub-id pub-id-type="doi">10.1016/j.jbi.2003.10.001</pub-id><pub-id pub-id-type="pmid">15016385</pub-id></citation></ref>
<ref id="B62"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>&#x00160;ari&#x00107;</surname> <given-names>J.</given-names></name> <name><surname>Jensen</surname> <given-names>L. J.</given-names></name> <name><surname>Ouzounova</surname> <given-names>R.</given-names></name> <name><surname>Rojas</surname> <given-names>I.</given-names></name> <name><surname>Bork</surname> <given-names>P.</given-names></name></person-group> (<year>2006</year>). <article-title>Extraction of regulatory gene protein networks from Medline</article-title>. <source>Bioinformatics</source> <volume>22</volume>, <fpage>645</fpage>.<pub-id pub-id-type="doi">10.1093/bioinformatics/bti597</pub-id></citation></ref>
<ref id="B63"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Smilde</surname> <given-names>A. K.</given-names></name> <name><surname>Bro</surname> <given-names>R.</given-names></name> <name><surname>Geladi</surname> <given-names>P.</given-names></name> <name><surname>Wiley</surname> <given-names>J.</given-names></name></person-group> (<year>2004</year>). <source>Multi-Way Analysis with Applications in the Chemical Sciences</source>. <publisher-loc>Chichester, UK</publisher-loc>: <publisher-name>Wiley</publisher-name>.</citation></ref>
<ref id="B64"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Soldatova</surname> <given-names>L. N.</given-names></name> <name><surname>Rzhetsky</surname> <given-names>A.</given-names></name></person-group> (<year>2011</year>). <article-title>Representation of research hypotheses</article-title>. <source>J. Biomed. Semantics</source> <volume>2</volume>, <fpage>1</fpage>.<pub-id pub-id-type="doi">10.1186/2041-1480-2-S2-I1</pub-id></citation></ref>
<ref id="B65"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>J. T.</given-names></name> <name><surname>Zeng</surname> <given-names>H. J.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Lu</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>Z.</given-names></name></person-group> (<year>2005</year>). <article-title>&#x0201C;CubeSVD: a novel approach to personalized Web search,&#x0201D;</article-title> in <conf-name>Proceedings of the 14th International Conference on World Wide Web</conf-name> (<conf-loc>Chiba</conf-loc>: <conf-sponsor>ACM</conf-sponsor>), <fpage>382</fpage>&#x02013;<lpage>390</lpage>.</citation></ref>
<ref id="B66"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Swanson</surname> <given-names>D. R.</given-names></name></person-group> (<year>1986</year>). <article-title>Fish oil, Raynaud&#x02019;s syndrome, and undiscovered public knowledge</article-title>. <source>Perspect. Biol. Med.</source> <volume>30</volume>, <fpage>7</fpage>.<pub-id pub-id-type="doi">10.1353/pbm.1986.0087</pub-id></citation></ref>
<ref id="B67"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tavazoie</surname> <given-names>S.</given-names></name> <name><surname>Hughes</surname> <given-names>J. D.</given-names></name> <name><surname>Campbell</surname> <given-names>M. J.</given-names></name> <name><surname>Cho</surname> <given-names>R. J.</given-names></name> <name><surname>Church</surname> <given-names>G. M.</given-names></name></person-group> (<year>1999</year>). <article-title>Systematic determination of genetic network architecture</article-title>. <source>Nat. Genet.</source> <volume>22</volume>, <fpage>281</fpage>&#x02013;<lpage>285</lpage>.<pub-id pub-id-type="doi">10.1038/10343</pub-id><pub-id pub-id-type="pmid">10391217</pub-id></citation></ref>
<ref id="B68"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thomas</surname> <given-names>P.</given-names></name> <name><surname>Durek</surname> <given-names>P.</given-names></name> <name><surname>Solt</surname> <given-names>I.</given-names></name> <name><surname>Klinger</surname> <given-names>B.</given-names></name> <name><surname>Witzel</surname> <given-names>F.</given-names></name> <name><surname>Schulthess</surname> <given-names>P.</given-names></name> <etal/></person-group> (<year>2015</year>). <article-title>Computer-assisted curation of a human regulatory core network from the biological literature</article-title>. <source>Bioinformatics</source> <volume>31</volume>, <fpage>1258</fpage>&#x02013;<lpage>1266</lpage>.<pub-id pub-id-type="doi">10.1093/bioinformatics/btu795</pub-id><pub-id pub-id-type="pmid">25433699</pub-id></citation></ref>
<ref id="B69"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tjioe</surname> <given-names>E.</given-names></name> <name><surname>Berry</surname> <given-names>M.</given-names></name> <name><surname>Homayouni</surname> <given-names>R.</given-names></name></person-group> (<year>2010</year>). <article-title>Discovering gene functional relationships using FAUN (Feature Annotation Using Nonnegative matrix factorization)</article-title>. <source>BMC Bioinformatics</source> <volume>11</volume>(<issue>Suppl. 6</issue>):<fpage>S14</fpage>.<pub-id pub-id-type="doi">10.1186/1471-2105-11-S6-S14</pub-id></citation></ref>
<ref id="B70"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tomasi</surname> <given-names>G.</given-names></name> <name><surname>Bro</surname> <given-names>R.</given-names></name></person-group> (<year>2006</year>). <article-title>A comparison of algorithms for fitting the PARAFAC model</article-title>. <source>Comput. Stat. Data Anal.</source> <volume>50</volume>, <fpage>1700</fpage>&#x02013;<lpage>1734</lpage>.<pub-id pub-id-type="doi">10.1016/j.csda.2004.11.013</pub-id></citation></ref>
<ref id="B71"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H. C.</given-names></name> <name><surname>Chen</surname> <given-names>Y. H. S.</given-names></name> <name><surname>Kao</surname> <given-names>H. Y.</given-names></name> <name><surname>Tsai</surname> <given-names>S. J.</given-names></name></person-group> (<year>2011</year>). <article-title>Inference of transcriptional regulatory network by bootstrapping patterns</article-title>. <source>Bioinformatics</source> <volume>27</volume>, <fpage>1422</fpage>&#x02013;<lpage>1428</lpage>.<pub-id pub-id-type="doi">10.1093/bioinformatics/btr155</pub-id><pub-id pub-id-type="pmid">21450714</pub-id></citation></ref>
<ref id="B72"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Welling</surname> <given-names>M.</given-names></name> <name><surname>Weber</surname> <given-names>M.</given-names></name></person-group> (<year>2001</year>). <article-title>Positive tensor factorization</article-title>. <source>Pattern Recognit. Lett.</source> <volume>22</volume>, <fpage>1255</fpage>&#x02013;<lpage>1261</lpage>.<pub-id pub-id-type="doi">10.1016/S0167-8655(01)00070-8</pub-id></citation></ref>
<ref id="B73"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>H.</given-names></name> <name><surname>Keane</surname> <given-names>J.</given-names></name> <name><surname>Bergman</surname> <given-names>C. M.</given-names></name> <name><surname>Nenadic</surname> <given-names>G.</given-names></name></person-group> (<year>2009</year>). <article-title>Assigning roles to protein mentions: the case of transcription factors</article-title>. <source>J. Biomed. Inform.</source> <volume>42</volume>, <fpage>887</fpage>&#x02013;<lpage>894</lpage>.<pub-id pub-id-type="doi">10.1016/j.jbi.2009.04.001</pub-id><pub-id pub-id-type="pmid">19364541</pub-id></citation></ref>
<ref id="B74"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Zeimpekis</surname> <given-names>D.</given-names></name> <name><surname>Gallopoulos</surname> <given-names>E.</given-names></name></person-group> (<year>2006</year>). <article-title>&#x0201C;TMG: a MATLAB toolbox for generating term-document matrices from text collections,&#x0201D;</article-title> in <source>Grouping Multidimensional Data</source>, eds <person-group person-group-type="editor"><name><surname>Kogan</surname> <given-names>J.</given-names></name> <name><surname>Nicholas</surname> <given-names>C.</given-names></name> <name><surname>Teboulle</surname> <given-names>M.</given-names></name></person-group> (<publisher-loc>Berlin-Heidelberg</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>187</fpage>&#x02013;<lpage>210</lpage>.</citation></ref>
<ref id="B75"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>H. M.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Gong</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <etal/></person-group> (<year>2012</year>). <article-title>AnimalTFDB: a comprehensive animal transcription factor database</article-title>. <source>Nucleic Acids Res.</source> <volume>40</volume>, <fpage>D144</fpage>&#x02013;<lpage>D149</lpage>.<pub-id pub-id-type="doi">10.1093/nar/gkr965</pub-id><pub-id pub-id-type="pmid">22080564</pub-id></citation></ref>
</ref-list>
</back>
</article>