<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="brief-report" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">778416</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2021.778416</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Perspective</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Open Problems in Extracellular RNA Data Analysis: Insights From an ERCC Online Workshop</article-title>
<alt-title alt-title-type="left-running-head">Alexander et&#x20;al.</alt-title>
<alt-title alt-title-type="right-running-head">Issues in exRNA Data Analysis</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Alexander</surname>
<given-names>Roger P.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1495580/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Kitchen</surname>
<given-names>Robert R</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1513580/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Tosar</surname>
<given-names>Juan Pablo</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/965556/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Roth</surname>
<given-names>Matthew</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1594200/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mestdagh</surname>
<given-names>Pieter</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/213967/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Max</surname>
<given-names>Klaas E. A.</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1595535/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Rozowsky</surname>
<given-names>Joel</given-names>
</name>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Kaczor-Urbanowicz</surname>
<given-names>Karolina El&#x17c;bieta</given-names>
</name>
<xref ref-type="aff" rid="aff8">
<sup>8</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chang</surname>
<given-names>Justin</given-names>
</name>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1482836/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Balaj</surname>
<given-names>Leonora</given-names>
</name>
<xref ref-type="aff" rid="aff9">
<sup>9</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Losic</surname>
<given-names>Bojan</given-names>
</name>
<xref ref-type="aff" rid="aff10">
<sup>10</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Van Nostrand</surname>
<given-names>Eric L.</given-names>
</name>
<xref ref-type="aff" rid="aff11">
<sup>11</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1553347/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>LaPlante</surname>
<given-names>Emily</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1516556/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mateescu</surname>
<given-names>Bogdan</given-names>
</name>
<xref ref-type="aff" rid="aff12">
<sup>12</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1505902/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>White</surname>
<given-names>Brian S.</given-names>
</name>
<xref ref-type="aff" rid="aff13">
<sup>13</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1506253/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yu</surname>
<given-names>Rongshan</given-names>
</name>
<xref ref-type="aff" rid="aff14">
<sup>14</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Milosavljevic</surname>
<given-names>Aleksander</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Stolovitzky</surname>
<given-names>Gustavo</given-names>
</name>
<xref ref-type="aff" rid="aff15">
<sup>15</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/206873/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Spengler</surname>
<given-names>Ryan M.</given-names>
</name>
<xref ref-type="aff" rid="aff16">
<sup>16</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1190707/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Extracellular RNA Communication Consortium</institution>, <addr-line>Phoenix</addr-line>, <addr-line>AZ</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Cardiovascular Research Center, Massachusetts General Hospital and Harvard Medical School</institution>, <addr-line>Charlestown</addr-line>, <addr-line>MA</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Pasteur Institute of Montevideo and University of the Republic of Uruguay</institution>, <addr-line>Montevideo</addr-line>, <country>Uruguay</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Molecular and Human Genetics, Baylor College of Medicine</institution>, <addr-line>Houston</addr-line>, <addr-line>TX</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Center for Medical Genetics, Department of Biomolecular Medicine, Cancer Research Institute Ghent (CRIG), Ghent University</institution>, <addr-line>Ghent</addr-line>, <country>Belgium</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Laboratory of RNA Molecular Biology, Rockefeller University</institution>, <addr-line>New York</addr-line>, <addr-line>NY</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>Department of Molecular Biophysics and Biochemistry, Yale University</institution>, <addr-line>New Haven</addr-line>, <addr-line>CT</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff8">
<sup>8</sup>
<institution>School of Dentistry, University of California, Los Angeles</institution>, <addr-line>Los Angeles</addr-line>, <addr-line>CA</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff9">
<sup>9</sup>
<institution>Department of Neurosurgery, Massachusetts General Hospital and Harvard Medical School</institution>, <addr-line>Boston</addr-line>, <addr-line>MA</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff10">
<sup>10</sup>
<institution>Icahn School of Medicine at Mount Sinai</institution>, <addr-line>New York</addr-line>, <addr-line>NY</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff11">
<sup>11</sup>
<institution>Verna and Marrs McLean Department of Biochemistry and Molecular Biology, Baylor College of Medicine</institution>, <addr-line>Houston</addr-line>, <addr-line>TX</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff12">
<sup>12</sup>
<institution>Brain Research Institute, University of Zurich</institution>, <addr-line>Zurich</addr-line>, <country>Switzerland</country>
</aff>
<aff id="aff13">
<sup>13</sup>
<institution>Jackson Laboratory</institution>, <addr-line>Bar Harbor</addr-line>, <addr-line>ME</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff14">
<sup>14</sup>
<institution>Department of Computer Science, Xiamen University, Aginome Scientific, Ltd.</institution>, <addr-line>Xiamen</addr-line>, <country>China</country>
</aff>
<aff id="aff15">
<sup>15</sup>
<institution>IBM T.J. Watson Research Center</institution>, <addr-line>Yorktown Heights</addr-line>, <addr-line>NY</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff16">
<sup>16</sup>
<institution>School of Medicine and Public Health, University of Wisconsin</institution>, <addr-line>Madison</addr-line>, <addr-line>WI</addr-line>, <country>United&#x20;States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1384666/overview">Stefan Muljo</ext-link>, National Institute of Allergy and Infectious Diseases (NIH), United&#x20;States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/148957/overview">Silvia Monticelli</ext-link>, Institute for Research in Biomedicine (IRB), Switzerland</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1204577/overview">Emmanouil Maragkakis</ext-link>, Laboratory of Genetics and Genomics (NIA), United&#x20;States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Roger P. Alexander, <email>rogerpalexander@gmail.com</email>; Ryan M. Spengler, <email>rspengler@medicine.wisc.edu</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to RNA, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>01</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>12</volume>
<elocation-id>778416</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>09</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>30</day>
<month>11</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Alexander, Kitchen, Tosar, Roth, Mestdagh, Max, Rozowsky, Kaczor-Urbanowicz, Chang, Balaj, Losic, Van Nostrand, LaPlante, Mateescu, White, Yu, Milosavljevic, Stolovitzky and Spengler.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Alexander, Kitchen, Tosar, Roth, Mestdagh, Max, Rozowsky, Kaczor-Urbanowicz, Chang, Balaj, Losic, Van Nostrand, LaPlante, Mateescu, White, Yu, Milosavljevic, Stolovitzky and Spengler</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these&#x20;terms.</p>
</license>
</permissions>
<abstract>
<p>We now know RNA can survive the harsh environment of biofluids when encapsulated in vesicles or by associating with lipoproteins or RNA binding proteins. These extracellular RNA (exRNA) play a role in intercellular signaling, serve as biomarkers of disease, and form the basis of new strategies for disease treatment. The Extracellular RNA Communication Consortium (ERCC) hosted a two-day online workshop (April 19&#x2013;20, 2021) on the unique challenges of exRNA data analysis. The goal was to foster an open dialog about best practices and discuss open problems in the field, focusing initially on small exRNA sequencing data. Video recordings of workshop presentations and discussions are available (<ext-link ext-link-type="uri" xlink:href="https://exrna.org/exRNAdata2021-videos/">https://exRNA.org/exRNAdata2021-videos/</ext-link>). There were three target audiences: experimentalists who generate exRNA sequencing data, computational and data scientists who work with those groups to analyze their data, and experimental and data scientists new to the field. Here we summarize issues explored during the workshop, including progress on an effort to develop an exRNA data analysis challenge to engage the community in solving some of these open problems.</p>
</abstract>
<kwd-group>
<kwd>extracellular RNA</kwd>
<kwd>RNA-seq</kwd>
<kwd>RNA sequencing</kwd>
<kwd>deconvolulion</kwd>
<kwd>batch variation</kwd>
<kwd>DREAM challenge</kwd>
<kwd>tissue of origin</kwd>
<kwd>biomarker discovery</kwd>
</kwd-group>
<contract-sponsor id="cn001">National Institute on Drug Abuse<named-content content-type="fundref-id">10.13039/100000026</named-content>
</contract-sponsor>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>In 2013, the NIH Common Fund launched the Extracellular RNA Communication Consortium to stimulate research into the fundamental biology of exRNA and its clinical applications in disease diagnosis and treatment. One product of the first stage, ERCC1 (2013&#x2013;2018), is the exRNA Atlas, a database of small RNA-sequencing and RT-qPCR data. To date, the Atlas holds over 7,700 samples from 14 biofluids and 16 disease conditions (<xref ref-type="bibr" rid="B23">Subramanian et&#x20;al., 2015</xref>; <xref ref-type="bibr" rid="B16">Murillo et&#x20;al., 2019</xref>). The current stage, ERCC2, focuses on development of technologies to characterize exRNA carriers and to isolate and characterize the contents of individual extracellular vesicles (EVs). A key strength of the exRNA Atlas is that the small RNA-seq datasets are uniformly processed by the extracellular RNA processing toolkit (exceRpt) (<xref ref-type="bibr" rid="B19">Rozowsky et&#x20;al., 2019</xref>). A hard-learned lesson from ERCC1 was the difficulty of eradicating systematic biases from the data, which makes it difficult to compare exRNA profiles across conditions. The April 2021 online workshop was held to address these and other problems in exRNA data analysis (<xref ref-type="fig" rid="F1">Figure&#x20;1</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Key topics discussed during the 2021 ERCC exRNA data analysis workshop.</p>
</caption>
<graphic xlink:href="fgene-12-778416-g001.tif"/>
</fig>
</sec>
<sec id="s2">
<title>Open Problems in exRNA Data Analysis</title>
<p>Rob Kitchen opened the workshop by outlining open problems in exRNA data analysis. One key challenge is that exRNA data quality varies widely and systematically between different experimental methods. Isolating and purifying exRNA and extracellular vesicles (EVs) from experimental samples is itself difficult, and RNA isolation kits used for these tasks are known to be a major source of variability in the resulting exRNA data. Compensating for this variation is a central challenge, as each kit and RNA sequencing method has different sequence biases that must accounted for when doing larger analyses (<xref ref-type="bibr" rid="B16">Murillo et&#x20;al., 2019</xref>; <xref ref-type="bibr" rid="B22">Srinivasan et&#x20;al., 2019</xref>). Dr. Kitchen stressed that comparison of the relative amounts of exRNAs across samples should be attempted in samples prepared using identical methods wherever possible. Even then, large sample-to-sample variation in exRNA carrier abundance persists, obscuring biological signal in case-control studies (<xref ref-type="bibr" rid="B16">Murillo et&#x20;al., 2019</xref>).</p>
<p>Another central question of interest in exRNA and EV biology is determining tissue and cell type of origin of different clusters of exRNAs in a biofluid. Dr. Kitchen argues that it might be possible in peripheral biofluids like urine and saliva, but more challenging in circulating blood serum and plasma, where the exRNA complement may be too diverse to parse. For vesicular exRNAs, the problem should be made simpler by improvements in experimental techniques to isolate EV sub-fractions, such as selecting for EVs with cell-type-specific surface proteins. As fractionation techniques improve, however, it will be necessary to compensate for variable enrichment efficiency.</p>
<p>Juan Pablo Tosar focused specifically on the quality of non-coding RNA annotations. He outlined how the mechanisms of biogenesis of miRNA and piRNA serve as the basis of existing annotations like miRbase. The problem is that such databases often lack strict curation, resulting in many sequences that are not miRNAs or piRNAs under any reasonable definition. Tosar described two examples from miRbase annotations, showing that miR-1202 is, in fact the small nucleolar RNA, SNORD126, and miR-1246, a microRNA enriched in EVs, is likely a contaminant from fetal bovine serum (FBS) in the cell culture media and likely fragment of the small nuclear RNA RNU2-1 (<xref ref-type="bibr" rid="B20">Sakha et&#x20;al., 2016</xref>; <xref ref-type="bibr" rid="B25">Tosar et&#x20;al., 2017</xref>). This is not a problem of miRbase itself, which is designed as a community-driven repository of putative miRNA sequences, providing minimal quality control at the point of submission (<xref ref-type="bibr" rid="B12">Kozomara et&#x20;al., 2019</xref>). His proposed solution to the problem of mis-annotation is to use a curated miRNA database like MirGeneDB (<xref ref-type="bibr" rid="B5">Fromm et&#x20;al., 2015</xref>; <xref ref-type="bibr" rid="B6">Fromm et&#x20;al., 2020</xref>), resulting in a smaller number of higher quality miRNA&#x20;calls.</p>
<p>Tosar emphasized that piRNAs have a complex biogenesis that imposes a strong bias to start with U or to have A in the 10th position, and they are expressed from genomic clusters with a high density of piRNA sequences (<xref ref-type="bibr" rid="B1">Czech et&#x20;al., 2018</xref>). PiRNA are mostly expressed in gonads and early embryos where their main role is to dampen the expression of transposable elements. However, existing piRNA databases include a very small number (&#x3c;1%) of contaminating sequences that do not match these criteria and have 100% overlap with other ncRNA families (<xref ref-type="bibr" rid="B26">Tosar et&#x20;al., 2018a</xref>). Findings of piRNA expression in cancers and biofluids are very often highly enriched in sequences from that set of false positive contaminants (<xref ref-type="bibr" rid="B26">Tosar et&#x20;al., 2018a</xref>). For example, the level of piR-54265 in the serum of colorectal cancer patients has been found to be predictive of tumor relapse after surgery. Its value as a biomarker notwithstanding, piR-54265 is, in fact, a mis-annotated full-length snoRNA, SNORD57 (<xref ref-type="bibr" rid="B28">Tosar et&#x20;al., 2021</xref>). Going back to the cell type of origin problem described above, while piRNA expression might be cancer-specific, snoRNAs are ubiquitously expressed. Consequently, annotation accuracy affects biological interpretation of the data. The final message is that it is important to realize our bioinformatics is only as strong as our RNA annotations.</p>
</sec>
<sec id="s3">
<title>exRNA Data Sources</title>
<p>In a session on exRNA data sources, Matt Roth gave an overview of the ERCC&#x2019;s exRNA Atlas (<xref ref-type="bibr" rid="B23">Subramanian et&#x20;al., 2015</xref>; <xref ref-type="bibr" rid="B16">Murillo et&#x20;al., 2019</xref>), a curated catalog of exRNA sequencing and qPCR data generated from a wide array of biofluids and disease states. Roth outlined features of the exRNA Atlas that facilitate accessing, querying, interpreting, and reusing experimental data and sample metadata. He also described ongoing efforts to expand Atlas content to include data and metadata from additional exRNA technologies being developed as part of ERCC2, and to integrate exRNA Atlas data into the NIH Common Fund Data Ecosystem (<ext-link ext-link-type="uri" xlink:href="https://app.nih-cfde.org/">https://app.nih-cfde.org/</ext-link>). Justin Chang gave a preview of the exRNA Explorer tool, a data exploration and visualization tool that will soon be integrated into the public exRNA Atlas. Joel Rozowsky later outlined the exceRpt pipeline used to process short exRNA sequencing data in the exRNA Atlas (<xref ref-type="bibr" rid="B19">Rozowsky et&#x20;al., 2019</xref>).</p>
<p>Pieter Mestdagh presented the Human Biofluid RNA Atlas (<xref ref-type="bibr" rid="B10">Hulstaert et&#x20;al., 2020</xref>), which characterizes and compares exRNA transcriptome profiles in a wide variety of biological fluids (<italic>n</italic>&#x20;&#x3d; 20) using both small RNA-sequencing and mRNA-capture sequencing. Introducing short and long synthetic spike-in RNAs before and after RNA extraction, enabled comparison of the absolute miRNA and mRNA content between samples, revealing large differences between fluids. Mestdagh also summarized efforts to identify the relative contribution of different tissues to exRNA transcriptomes and how this varied across biofluids. Deconstruction of exRNA profiles into contributing tissues can be achieved through computational deconvolution, a topic explored in-depth later in the workshop. He showed that the accuracy of deconvolution depends on several factors, including 1) proper transformation and normalization of the exRNA-seq read count, 2) choice of deconvolution algorithm and 3) quality and completeness of the reference data. Mestdagh also presented evidence suggesting that circular RNAs (circRNAs) are present in biofluids, potentially at a higher proportional abundance relative to linear transcripts than they are in cells and tissues.</p>
<p>Klaas Max discussed healthy reference profiles of extracellular miRNA in serum and plasma (<xref ref-type="bibr" rid="B15">Max et&#x20;al., 2018</xref>). Based on an initial cohort of 13 individuals and a second larger cohort of over 200 individuals, they found very little variation between males and females, but more noticeable differences between serum and plasma samples. Interestingly, they found that the exRNA profiles of pregnant women were distinct from non-pregnant women, and that the variant miRNAs could predict not only whether a woman were pregnant, but in which stage of pregnancy they were in. Among the most variable miRNAs was cluster-miR-498(46), which was upregulated in pregnant versus non-pregnant women, increasing between 50- and 250-fold from first to third trimester. Max noted that many of the most variable miRNAs in healthy subjects were cell-lineage-specific miRNAs of the liver, neuroendocrine organs, adrenal glands, epithelial cells and muscle. Abundance of several such miRNAs sharing a common origin were moderately correlated. Although they identified additional sets of variably expressed miRNAs, Max noted that few miRNAs are known to be cell-lineage-specific, which complicates methods to deconvolute and identify tissues of origin. Plasma subfractionation by ultracentrifugation to enrich for non-hematopoietic miRNAs did not result in a strong enrichment of organ- or cell-type-specific miRNAs.</p>
</sec>
<sec id="s4">
<title>exRNA-Seq Processing</title>
<sec id="s4-1">
<title>Exogenous exRNA</title>
<p>Karolina El&#x17c;bieta Kaczor-Urbanowicz discussed bioinformatic analysis of salivary RNA sequencing data (<xref ref-type="bibr" rid="B11">Kaczor-Urbanowicz et&#x20;al., 2018</xref>) in the context of a search for exRNA biomarkers of gastric cancer (GC). PI David Wong, co-PI Yong Kim and collaborator Sung Kim in South Korea collected 2000 saliva samples from GC patients and non-GC controls and noticed that saliva has a much higher proportion of microbial RNA than other biofluids. In fact, quality control (QC) criteria for the ERCC&#x2019;s exceRpt pipeline had to be modified to account for the disproportionately high microbial RNA content. The research team evaluated whether to map RNA-seq reads to microbial RNA before or after mapping to the human genome. They found that the best approach for salivary long RNA-seq data is to map to the microbiome first and remove the aligned bacterial reads before mapping to human. The research team also found that Asian-specific strains of the <italic>Helicobacter pylori</italic> bacteria and Epstein-Barr virus associated with GC for determining disease state when analyzing saliva samples of Asian origin. This highlights the need for ensuring ethnic diversity in human genome and microbiome sequences in the age of personalized medicine. More recently, the researchers have found that performing deconvolution and variance partition analyses to isolate extraneous sources of variation improves their ability to identify exRNA biomarkers.</p>
</sec>
<sec id="s4-2">
<title>Considerations for exRNA Library Preparation</title>
<p>Ryan Spengler emphasized that standard small RNA-seq library preparation methods require RNAs to have 5&#x2032; phosphate and 3&#x2032; hydroxyl groups, but a substantial fraction of exRNAs lack these end chemistries (<xref ref-type="bibr" rid="B8">Giraldez et&#x20;al., 2019</xref>). By incubating the RNA pool with polynucleotide kinase (PNK) before adapter ligation, Spengler showed that exRNA transcriptome profiles markedly changed, which, for example, significantly increased the number of mRNA and lncRNA fragments found in plasma. These fragments likely originate from specific regions of mRNA transcripts that are protected from RNase degradation, and similar regions are protected across individuals. Careful filtering of reads mapping to repetitive and non-human sequences is essential for identification of <italic>bona fide</italic> mRNA fragments. In a longitudinal study of hematopoietic stem cell transplant recipients, mRNA fragments segregated into several distinct temporal co-expression signatures associated with the transcripts&#x2019; likely tissue of origin (namely, liver and bone marrow).</p>
</sec>
<sec id="s4-3">
<title>Biomarker Discovery</title>
<p>Leonora Balaj discussed efforts to identify extracellular mRNA signatures associated with glioma. They examined long RNA-seq reads from EVs isolated from glioma patients compared to healthy individuals matched by age and sex. They also demonstrated a two-hybrid capture method which uses exome capture panels to enrich for exRNA reads from protein-coding mRNAs. Including a ribosomal RNA depletion step, they were able to substantially enrich for mRNA sequences and largely eliminate the non-mRNA reads that dominate non-captured libraries. They also performed long mRNA-seq on exosomal RNAs isolated from patients before and after undergoing dacomitinib treatment for recurrent glioblastoma and found exosomal mRNA signatures that distinguished responders from non-responders to the treatment. Importantly, in follow-up validation cohorts, these signatures showed promise in predicting which individuals would respond to treatment.</p>
</sec>
<sec id="s4-4">
<title>
<italic>De Novo</italic> Discovery of Small RNA Clusters</title>
<p>Bojan Losic presented a method for the discovery of small RNA clusters (smRCs, pronounced smirks) expressed in circulating liver cancer EVs (<xref ref-type="bibr" rid="B30">von Felden et&#x20;al., 2021</xref>). The method is <italic>de novo</italic>, not relying on mapping reads to annotated RNAs. In fact, they found that most expressed smRCs emanate from unannotated genomic regions in a cell-type- and biofluid-specific manner and have EV-specific properties which can be exploited for biomarker discovery. Three such smRCs were found to be strong biomarkers for early-stage liver cancer (hepatocellular carcinoma, HCC), significantly outperforming the clinical surveillance standard in an independent Phase 2 clinical study. These findings raise the possibility of a blood-only, operator independent, minimally invasive liquid biopsy test for&#x20;HCC.</p>
</sec>
</sec>
<sec id="s5">
<title>exRNA and RNA Binding Proteins</title>
<p>Extracellular RNA that circulates in biofluids must be shielded from the harsh environment, particularly from enzymes that digest RNA (RNases). Some exRNAs are resistant to RNase digestion, e.g., Gly/Glu tRNA fragments that can form stable homo- and hetero-dimers (<xref ref-type="bibr" rid="B27">Tosar et&#x20;al., 2018b</xref>). Other exRNAs are protected inside vesicles or by association with RNA-binding proteins (RBPs). Recent work shows that some cell-surface exRNAs are protected by glycosylation (<xref ref-type="bibr" rid="B4">Flynn et&#x20;al., 2021</xref>). Vesicular exRNAs are the best studied class of exRNAs. RBP-associated exRNAs have been difficult to study because of the delicate protein biochemistry required to isolate and characterize RNA binding sites for each of the hundreds of RBPs in the human genome (<xref ref-type="bibr" rid="B7">Gerstberger et&#x20;al., 2014</xref>). Eric Van Nostrand outlined resources from the Encyclopedia of RNA Elements (ENCORE) to aid in this effort, including validated antibodies and shRNA reagents (<xref ref-type="bibr" rid="B24">Sundararaman et&#x20;al., 2016</xref>). ENCORE experiments systematically characterized aspects of RBP regulation including RBP <italic>in&#x20;vitro</italic> motifs and RNA interactions in K562 and HepG2 cell lines for over 350 RNA binding proteins (<xref ref-type="bibr" rid="B29">Van Nostrand et&#x20;al., 2020</xref>).</p>
<p>Emily LaPlante described initial work scanning the exRNA Atlas for ENCORE RNA binding sites and establishing the infrastructure to allow Atlas users to study their own regions of interest. Bogdan Mateescu outlined current knowledge about exRNA-associated RBPs (exRBPs) (<xref ref-type="bibr" rid="B3">Fabbiano et&#x20;al., 2020</xref>) as well as experimental challenges to discovering new ones. Then he laid out the ERCC2 PRISM (Purification of exRNA by Immuno-capture and Sorting using Microfluidics) project designed to meet those challenges. The exRBP-HIT bioinformatic pipeline identifies candidate exRBPs by overlapping peak calls from exRNA-seq and RBP eCLIP (enhanced CrossLinking and ImmunoPrecipitation) experiments. Permutation analysis of the data yields a ranked list of the most likely candidates. Initial analysis of data from plasma, saliva, and urine samples from the Van Keuren Jensen lab reassuringly identifies exRBPs known to be associated with particular classes of exRNA, for example RNA silencing factors with miRNA and Ro60 with YRNA. More interesting were new candidate exRBPs targeting extracellular mRNA fragments.</p>
<p>The next step in the PRISM group&#x2019;s strategy is experimental validation of candidate exRBPs in a model system. After using CRISPR to create knockout strains in the 293T&#x20;cell line for each candidate exRBP gene, exRNA profiles from wild-type and knockout conditioned media are compared. A case study including knockout lines for 10 genes in the RNA silencing pathway led to changes in expression in classes of exRNA beyond miRNA: tRNA, snRNA, snoRNA, and YRNA. Whereas the bioinformatic pipeline identifies individual sites of exRNA interaction with RBPs, the model system provides a global perspective on the impact of each RBP on expression of all exRNAs and perhaps gives insight into exRBP function. The ERCC2 PRISM group plans to identify over 100 candidate exRBPs and create an atlas of exRNA profiles from the model system exRBP knockout strains.</p>
</sec>
<sec id="s6">
<title>Deconvolution</title>
<p>Two major open problems in the field are identifying tissue of origin of the exRNAs in a biofluid and associating them with their molecular carrier, whether that be an RNA binding protein, a lipid like HDL or LDL, or a variety of classes of extracellular vesicle. Computational deconvolution, a method for partitioning a heterogeneous dataset into contributions from different independent constituents, can complement experimental approaches to addressing such challenges. There are two broad classes of deconvolution algorithms&#x2014;reference-based and reference-free. Reference-based algorithms require a known signature matrix of gene expression profiles from all the cell types or molecular carriers to be separated. Reference-free methods estimate simultaneously both the signature matrix and the relative ratios of each cell type or carrier in the mixture.</p>
<p>In a session on deconvolution, Brian White described a Tumor Deconvolution DREAM Challenge (<xref ref-type="bibr" rid="B31">White et&#x20;al., 2019</xref>) to assess existing and inspire novel methods for deconvolving bulk RNA expression data. DREAM (Dialogue on Reverse Engineering Assessment and Methods) Challenges use crowd-sourcing to address fundamental questions in biomedical research. Challenges are posed by domain experts in concert with DREAM organizers, who are responsible for curating data, defining objective evaluation criteria, and engineering a computational framework for method submission and execution. Recently, Challenges have leveraged a &#x201c;model-to-data&#x201d; paradigm (<xref ref-type="bibr" rid="B9">Guinney and Saez-Rodriguez 2018</xref>) in which models are executed in the cloud, facilitating reproducible deployment of methods and ensuring that those methods are not overfit to data. The DREAM community includes over 30,000&#x20;cross-disciplinary participants who have contributed to more than 60 Challenges resulting in over 100 publications (<ext-link ext-link-type="uri" xlink:href="http://dreamchallenges.org/">http://DREAMchallenges.org/</ext-link>).</p>
<p>In the Tumor Deconvolution DREAM Challenge, the organizers provided teams RNA expression profiles from <italic>in&#x20;vitro</italic> and <italic>in silico</italic> admixtures spanning 14 different immune, stromal, and cancer cell types and asked teams to predict the relative ratios of each cell type in the admixtures. To help assess its potential relevance for deconvolving exRNA data, Dr. White described CIBERSORTx, one of the baseline reference methods used in the Challenge. CIBERSORTx is based on the earlier Cell-type Identification By Estimating Relative Subsets Of RNA Transcripts (CIBERSORT) algorithm (<xref ref-type="bibr" rid="B17">Newman et&#x20;al., 2015</xref>). Both CIBERSORT and CIBERSORTx are reference-based. Given an input matrix of reference gene expression signatures from all cell types expected to be present in a mixture, CIBERSORT uses support-vector regression (SVR) to estimate from bulk RNA sequencing data the ratios of each cell type in the mixture. CIBERSORTx (<xref ref-type="bibr" rid="B18">Newman et&#x20;al., 2019</xref>) is a two-stage deconvolution algorithm, including a batch correction step to reduce technical variation across the single-cell RNA-seq datasets used to develop the signature matrix. Results, including those from CIBERSORTx, are summarized on the DREAM Challenge website (<ext-link ext-link-type="uri" xlink:href="https://www.synapse.org/tumorDeconvolutionChallenge">https://www.synapse.org/tumorDeconvolutionChallenge</ext-link>).</p>
<p>Rongshan Yu described the DAISM-DNN algorithm (Data Augmentation through <italic>In Silico</italic> Mixing and Deep Neural Network) used to win the Tumor Deconvolution DREAM Challenge. Dr. Yu explained that the choice of a neural network method was guided by the non-linearity of the input data. He showed that expression levels of different genes vary across cell types in a non-linear way. Neural networks perform better than linear regression in that case. The problem with using DNN is that it requires a very large number of training datasets, on the order of ten thousand. The team&#x2019;s solution was to augment the existing training datasets by shuffling them together <italic>in silico</italic> with expression data from target cells, either bulk RNAseq from purified cell samples or single-cell RNAseq. Finally, although DNN models are generally considered to be difficult to interpret black boxes, it is possible to apply methods such as the SHapley Additive explanation (SHAP) model from game theory (<xref ref-type="bibr" rid="B14">Lundberg and Lee 2017</xref>) to output a ranked list of the genes that contribute most to the deconvolution results from DAISM-DNN (<xref ref-type="bibr" rid="B13">Lin et&#x20;al., 2021</xref>).</p>
<p>Finally, Aleks Milosavljevic described the XDec algorithm for analyzing small RNA-seq datasets in the exRNA Atlas. XDec is a two-stage reference-free deconvolution algorithm that separates exRNAs in a biofluid sample into sets associated with different molecular carriers&#x2014;extracellular vesicles, lipoproteins HDL and LDL, and three categories of RNA binding proteins (<xref ref-type="bibr" rid="B16">Murillo et&#x20;al., 2019</xref>).</p>
</sec>
<sec id="s7">
<title>Formulating an exRNA Data Analysis Challenge</title>
<p>A major goal of the workshop was to lay the foundation for creating an exRNA-themed data analysis challenge. The DREAM challenge framework is ideal for presenting problems in data analysis to the wider scientific community. The last session of the workshop was chaired by Gustavo Stolovitzky, co-founder of the DREAM challenges. He gave a talk asking the question: &#x201c;Can we use a crowdsourcing challenge to benchmark exRNA transcriptomics analyses?&#x201d; Building on that foundation, Roger Alexander discussed specific classes of challenges that the community might put forth. The ensuing discussion and a post-workshop survey narrowed the field to two classes of challenge: first, a biomarker discovery challenge, and second, a deconvolution challenge to address two problems: identifying cell type of origin and molecular carrier of groups of exRNAs within a biofluid.</p>
<sec id="s7-1">
<title>Biomarker Discovery Challenge</title>
<p>Given a set of case-control experiments for a specific disease, what are the most informative sets of exRNA biomarkers for diagnosing and tracking treatment progress for that disease? One of the most difficult aspects of the challenge is acquiring the number of case-control experiments necessary to have sufficient statistical power for biomarker discovery. Improved batch correction algorithms would ease the problem by making it easier to stitch together datasets from multiple related studies. Given large testing and training datasets, teams would compete to develop algorithms that identify exRNA biomarkers that best discriminate cases of disease from controls. The main challenge would reward algorithms that can solve the batch-to-batch variation problem to combine public datasets into a larger testing dataset. A sub-challenge would select for algorithms that perform well even as the original training dataset is repeatedly sub-sampled to be smaller and less powerful. The authors welcome community participation in this effort.</p>
</sec>
<sec id="s7-2">
<title>Deconvolution Challenge</title>
<p>Is it possible to determine the cell type of origin or molecular carrier of different clusters of exRNAs in a biofluid? These questions can begin to be addressed in a data analysis challenge using synthetic datasets. Beyond improved algorithms, a beneficial outcome of such a challenge would be the creation of better gold standards of cell-, tissue-, and molecular carrier-associated exRNA profiles for use with reference-based deconvolution methods.</p>
<p>Moving beyond a synthetic deconvolution challenge requires the generation of an experimental gold standard, which is more challenging for exRNA than for tumor deconvolution. For tumor deconvolution, 14 cell lines were chosen to mimic a tumor microenvironment. For exRNA, the community must come together to agree on a similar proxy environment, perhaps by collecting culture media from cell lines representing several different tissues, with or without a vesicle purification step before RNA isolation and sequencing. Finding that proxy environment was beyond the scope of the workshop but will be necessary to make possible a non-synthetic exRNA deconvolution challenge.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s8">
<title>Discussion</title>
<p>The workshop introduced experimental and data scientists to the field of exRNA data analysis. The standard for small RNA-seq data is uniform processing by the exceRpt pipeline and storage in the exRNA Atlas. Long extracellular RNA is less thoroughly studied, and methods for its analysis are still under development. It is important to be mindful that standard RNA-seq methods are blind to many RNA base modifications, and mis-annotated RNAs can lead to misinterpretation. Biomarker and other studies requiring comparison across datasets would benefit greatly from improved algorithms to compensate for batch-to-batch variation and other systematic errors.</p>
</sec>
<sec sec-type="conclusion" id="s9">
<title>Conclusion</title>
<p>With the discovery of new classes of exRNA (<xref ref-type="bibr" rid="B4">Flynn et&#x20;al., 2021</xref>) and development of new types of exRNA-based disease therapies (<xref ref-type="bibr" rid="B21">Segel et&#x20;al., 2021</xref>), now is an exciting time in the field of extracellular RNA research. To better our understanding of exRNA biology and improve our ability to use exRNA in the clinic, we wish to issue a call to action to the scientific community. We are seeking a validation dataset that will enable us to launch an exRNA biomarker discovery challenge. Specifically, we seek small exRNA sequencing data for several hundred cases and controls that can be shared prior to publication. The DREAM challenge &#x201c;model to data&#x201d; paradigm will ensure that the data remains private in a secure computing environment and is not shared with challenge participants (<xref ref-type="bibr" rid="B2">Ellrott et&#x20;al., 2019</xref>).</p>
</sec>
</body>
<back>
<sec id="s10">
<title>Data Availability Statement</title>
<p>Publicly available datasets were analyzed in this study. Data sources include the exRNA Atlas (<ext-link ext-link-type="uri" xlink:href="https://exRNA-Atlas.org">https://exRNA-Atlas.org</ext-link>) and the Human Biofluid RNA Atlas (<ext-link ext-link-type="uri" xlink:href="https://r2.amc.nl">https://r2.amc.nl</ext-link>). Data from the Tumour Deconvolution DREAM Challenge is available at <ext-link ext-link-type="uri" xlink:href="https://www.synapse.org/#!Synapse:syn15589870/wiki/582446">https://www.synapse.org/#!Synapse:syn15589870/wiki/582446</ext-link>.</p>
</sec>
<sec id="s11">
<title>Author Contributions</title>
<p>RA, AM, MR, and RS conceived this meeting report. RA and RS drafted the manuscript. RA organized and led the workshop. RK, JT, MR, PM, KM, JR, KK-U, JC, LB, BL, EV, EL, BM, BW, RS, AM, and GS participated in the workshop and helped edit the text of this manuscript.</p>
</sec>
<sec id="s12">
<title>Funding</title>
<p>The Extracellular RNA Communication Consortium is an NIH Common Fund program. This work was supported by NIH U54-DA049098 (RA, AM, MR, JR, and JC). Publication costs were provided by the University of Wisconsin-Madison, Department of Internal Medicine, Division of Gastroenterology and Hepatology.</p>
</sec>
<sec sec-type="COI-statement" id="s13">
<title>Conflict of Interest</title>
<p>EV is co-founder, member of the Board of Directors, on the SAB, equity holder, and paid consultant for Eclipse BioInnovations. EV&#x2019;s interests have been reviewed and approved by the Baylor College of Medicine in accordance with its conflict of interest policies. RS is a paid consultant for Lacerta Therapeutics.</p>
<p>Author RY was employed by the company Aginome Scientific, Ltd.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s14">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>The authors wish to recognize the many contributions of their colleagues and lab members that made their presentations at the workshop possible.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Czech</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Munaf&#xf2;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ciabrelli</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Eastwood</surname>
<given-names>E. L.</given-names>
</name>
<name>
<surname>Fabry</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Kneuss</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>piRNA-Guided Genome Defense: From Biogenesis to Silencing</article-title>. <source>Annu. Rev. Genet.</source> <volume>52</volume>, <fpage>131</fpage>&#x2013;<lpage>157</lpage>. <pub-id pub-id-type="doi">10.1146/annurev-genet-120417-031441</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ellrott</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Buchanan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Creason</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mason</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schaffter</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hoff</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Reproducible Biomedical Benchmarking in the Cloud: Lessons from Crowd-Sourced Data Challenges</article-title>. <source>Genome Biol.</source> <volume>20</volume>, <fpage>195</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-019-1794-0</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fabbiano</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Corsi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gurrieri</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Trevisan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Notarangelo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>D&#x27;Agostino</surname>
<given-names>V. G.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>RNA Packaging into Extracellular Vesicles: An Orchestra of RNA-Binding Proteins?</article-title> <source>J.&#x20;Extracell Vesicles</source> <volume>10</volume>, <fpage>e12043</fpage>. <pub-id pub-id-type="doi">10.1002/jev2.12043</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Flynn</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Pedram</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Malaker</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Batista</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>B. A. H.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>A. G.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Small RNAs Are Modified with N-Glycans and Displayed on the Surface of Living Cells</article-title>. <source>Cell</source> <volume>184</volume>, <fpage>3109</fpage>&#x2013;<lpage>3124</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2021.04.023</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fromm</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Billipp</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Peck</surname>
<given-names>L. E.</given-names>
</name>
<name>
<surname>Johansen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tarver</surname>
<given-names>J.&#x20;E.</given-names>
</name>
<name>
<surname>King</surname>
<given-names>B. L.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>A Uniform System for the Annotation of Vertebrate microRNA Genes and the Evolution of the Human microRNAome</article-title>. <source>Annu. Rev. Genet.</source> <volume>49</volume>, <fpage>213</fpage>&#x2013;<lpage>242</lpage>. <pub-id pub-id-type="doi">10.1146/annurev-genet-120213-092023</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fromm</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Domanska</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>H&#xf8;ye</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ovchinnikov</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Aparicio-Puerta</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>MirGeneDB 2.0: the Metazoan microRNA Complement</article-title>. <source>Nucleic Acids Res.</source> <volume>48</volume>, <fpage>D132</fpage>&#x2013;<lpage>D141</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkz885</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gerstberger</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hafner</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tuschl</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>A Census of Human RNA-Binding Proteins</article-title>. <source>Nat. Rev. Genet.</source> <volume>15</volume>, <fpage>829</fpage>&#x2013;<lpage>845</lpage>. <pub-id pub-id-type="doi">10.1038/nrg3813</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Giraldez</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Spengler</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Etheridge</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Goicochea</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Tuck</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>S. W.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Phospho-RNA-seq: a Modified Small RNA-Seq Method that Reveals Circulating mRNA and lncRNA Fragments as Potential Biomarkers in Human Plasma</article-title>. <source>EMBO J.</source> <volume>38</volume>, <fpage>e101695</fpage>. <pub-id pub-id-type="doi">10.15252/embj.2019101695</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guinney</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Saez-Rodriguez</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Alternative Models for Sharing Confidential Biomedical Data</article-title>. <source>Nat. Biotechnol.</source> <volume>36</volume>, <fpage>391</fpage>&#x2013;<lpage>392</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.4128</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hulstaert</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Morlion</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Avila Cobos</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Verniers</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Nuytens</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Vanden Eynde</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Charting Extracellular Transcriptomes in the Human Biofluid RNA Atlas</article-title>. <source>Cel Rep.</source> <volume>33</volume>, <fpage>108552</fpage>. <pub-id pub-id-type="doi">10.1016/j.celrep.2020.108552</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaczor-Urbanowicz</surname>
<given-names>K. E.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Galeev</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kitchen</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Gerstein</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Novel Approaches for Bioinformatic Analysis of Salivary RNA Sequencing Data for Development</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx504</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kozomara</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Birgaoanu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Griffiths-Jones</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>miRBase: from microRNA Sequences to Function</article-title>. <source>Nucleic Acids Res.</source> <volume>47</volume>, <fpage>D155</fpage>&#x2013;<lpage>D162</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky1141</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>DAISM-DNN<sup>XMBD</sup>: Highly Accurate Cell Type Proportion Estimation with In Silico Data Augmentation and Deep Neural Networks</article-title>. <source>bioRxiv</source>. <pub-id pub-id-type="doi">10.1101/2020.03.26.009308v3</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lundberg</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S. I.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>A Unified Approach to Interpreting Model Predictions</article-title>,&#x201d; in <conf-name>NIPS&#x27;17: Proceedings of the 31st International Conference on Neural Information Processing Systems, Long Beach, CA, December 4&#x2013;9, 2017</conf-name>, <fpage>4768</fpage>&#x2013;<lpage>4777</lpage>. </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Max</surname>
<given-names>K. E. A.</given-names>
</name>
<name>
<surname>Bertram</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Akat</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Bogardus</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Morozov</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Human Plasma and Serum Extracellular Small RNA Reference Profiles and Their Clinical Utility</article-title>. <source>Proc. Natl. Acad. Sci. USA</source> <volume>115</volume>, <fpage>E5334</fpage>&#x2013;<lpage>E5343</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1714397115</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Murillo</surname>
<given-names>O. D.</given-names>
</name>
<name>
<surname>Thistlethwaite</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Rozowsky</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Subramanian</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Lucero</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Shah</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>exRNA Atlas Analysis Reveals Distinct Extracellular RNA Cargo Types and Their Carriers Present across Human Biofluids</article-title>. <source>Cell</source> <volume>177</volume>, <fpage>463</fpage>&#x2013;<lpage>477</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2019.02.018</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Newman</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Gentles</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Robust Enumeration of Cell Subsets from Tissue Expression Profiles</article-title>. <source>Nat. Methods</source> <volume>12</volume>, <fpage>453</fpage>&#x2013;<lpage>457</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.3337</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Newman</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Steen</surname>
<given-names>C. B.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Gentles</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Chaudhuri</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Scherer</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Determining Cell Type Abundance and Expression from Bulk Tissues with Digital Cytometry</article-title>. <source>Nat. Biotechnol.</source> <volume>37</volume>, <fpage>773</fpage>&#x2013;<lpage>782</lpage>. <pub-id pub-id-type="doi">10.1038/s41587-019-0114-2</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rozowsky</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kitchen</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>J.&#x20;J.</given-names>
</name>
<name>
<surname>Galeev</surname>
<given-names>T. R.</given-names>
</name>
<name>
<surname>Diao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Warrell</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>exceRpt: A Comprehensive Analytic Platform for Extracellular RNA Profiling</article-title>. <source>Cel Syst.</source> <volume>8</volume>, <fpage>352</fpage>&#x2013;<lpage>357</lpage>. <pub-id pub-id-type="doi">10.1016/j.cels.2019.03.004</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sakha</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Muramatsu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ueda</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Inazawa</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Exosomal microRNA miR-1246 Induces Cell Motility and Invasion through the Regulation of DENND2D in Oral Squamous Cell Carcinoma</article-title>. <source>Sci. Rep.</source> <volume>6</volume>, <fpage>38750</fpage>. <pub-id pub-id-type="doi">10.1038/srep38750</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Segel</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lash</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ladha</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Mammalian Retrovirus-like Protein PEG10 Packages its Own mRNA and Can Be Pseudotyped for mRNA Delivery</article-title>. <source>Science</source> <volume>373</volume>, <fpage>882</fpage>&#x2013;<lpage>889</lpage>. <pub-id pub-id-type="doi">10.1126/science.abg6155</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Srinivasan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yeri</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cheah</surname>
<given-names>P. S.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Danielson</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>De Hoff</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Small RNA Sequencing across Diverse Biofluids Identifies Optimal Methods for exRNA Isolation</article-title>. <source>Cell</source> <volume>177</volume>, <fpage>446</fpage>&#x2013;<lpage>462</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2019.03.024</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Subramanian</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Kitchen</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Alexander</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Carter</surname>
<given-names>B. S.</given-names>
</name>
<name>
<surname>Cheung</surname>
<given-names>K.-H.</given-names>
</name>
<name>
<surname>Laurent</surname>
<given-names>L. C.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Integration of Extracellular RNA Profiling Data Using Metadata, Biomedical Ontologies and Linked Data Technologies</article-title>. <source>J.&#x20;Extracellular Vesicles</source> <volume>4</volume>, <fpage>27497</fpage>. <pub-id pub-id-type="doi">10.3402/jev.v4.27497</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sundararaman</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Blue</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Stanton</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Elkins</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Olson</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Resources for the Comprehensive Discovery of Functional RNA Elements</article-title>. <source>Mol. Cel</source> <volume>61</volume>, <fpage>903</fpage>&#x2013;<lpage>913</lpage>. <pub-id pub-id-type="doi">10.1016/j.molcel.2016.02.012</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tosar</surname>
<given-names>J.&#x20;P.</given-names>
</name>
<name>
<surname>Cayota</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Eitan</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Halushka</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Witwer</surname>
<given-names>K. W.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Ribonucleic Artefacts: Are Some Extracellular RNA Discoveries Driven by Cell Culture Medium Components?</article-title> <source>J.&#x20;Extracell. Vesicles</source> <volume>6</volume>, <fpage>1272832</fpage>. <pub-id pub-id-type="doi">10.1080/20013078.2016.1272832</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tosar</surname>
<given-names>J.&#x20;P.</given-names>
</name>
<name>
<surname>Rovira</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Cayota</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018a</year>). <article-title>Non-coding RNA Fragments Account for the Majority of Annotated piRNAs Expressed in Somatic Non-gonadal Tissues</article-title>. <source>Commun. Biol.</source> <volume>1</volume>, <fpage>2</fpage>. <pub-id pub-id-type="doi">10.1038/s42003-017-0001-7</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tosar</surname>
<given-names>J.&#x20;P.</given-names>
</name>
<name>
<surname>G&#xe1;mbaro</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Darr&#xe9;</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Pantano</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Westhof</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Cayota</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018b</year>). <article-title>Dimerization Confers Increased Stability to Nucleases in 5&#x2032; Halves from glycine and Glutamic Acid tRNAs</article-title>. <source>Nucleic Acids Res.</source> <volume>46</volume>, <fpage>9081</fpage>&#x2013;<lpage>9093</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky495</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tosar</surname>
<given-names>J.&#x20;P.</given-names>
</name>
<name>
<surname>Garc&#xed;a-Silva</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Cayota</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Circulating SNORD57 rather Than piR-54265 Is a Promising Biomarker for Colorectal Cancer: Common Pitfalls in the Study of Somatic piRNAs in Cancer</article-title>. <source>RNA</source> <volume>27</volume>, <fpage>403</fpage>&#x2013;<lpage>410</lpage>. <pub-id pub-id-type="doi">10.1261/rna.078444.120</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Nostrand</surname>
<given-names>E. L.</given-names>
</name>
<name>
<surname>Freese</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Pratt</surname>
<given-names>G. A.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>A Large-Scale Binding and Functional Map of Human RNA-Binding Proteins</article-title>. <source>Nature</source> <volume>583</volume>, <fpage>711</fpage>&#x2013;<lpage>719</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-020-2077-3</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>von Felden</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Garcia-Lezana</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Dogra</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Gonzalez-Kozlova</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ahsen</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Craig</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Unannotated Small RNA Clusters Associated with Circulating Extracellular Vesicles Detect Early Stage Liver Cancer</article-title>. <source>Gut</source>. <pub-id pub-id-type="doi">10.1136/gutjnl-2021-325036</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>White</surname>
<given-names>B. S.</given-names>
</name>
<name>
<surname>Gentles</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>de Reyni&#xe8;s</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Newman</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Lamb</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Heiser</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). &#x201c;<article-title>A Tumor Deconvolution DREAM Challenge: Inferring Immune Infiltration from Bulk Gene Expression Data [abstract]</article-title>,&#x201d; in <conf-name>Proceedings of the American Association for Cancer Research Annual Meeting 2019</conf-name>, <conf-loc>Atlanta, GA. Philadelphia (PA)</conf-loc>, <conf-date>Mar 29-Apr 3,&#x20;2019</conf-date>. </citation>
</ref>
</ref-list>
</back>
</article>