<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Bioeng. Biotechnol.</journal-id>
<journal-title>Frontiers in Bioengineering and Biotechnology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Bioeng. Biotechnol.</abbrev-journal-title>
<issn pub-type="epub">2296-4185</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fbioe.2020.01021</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Bioengineering and Biotechnology</subject>
<subj-group>
<subject>Technology and Code</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>GeTallele: A Method for Analysis of DNA and RNA Allele Frequency Distributions</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>S&#x00142;owi&#x00144;ski</surname> <given-names>Piotr</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/955974/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Muzi</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Restrepo</surname> <given-names>Paula</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Alomran</surname> <given-names>Nawaf</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Spurr</surname> <given-names>Liam F.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1074836/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Miller</surname> <given-names>Christian</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Tsaneva-Atanasova</surname> <given-names>Krasimira</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/224888/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Horvath</surname> <given-names>Anelia</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff8"><sup>8</sup></xref>
<xref ref-type="aff" rid="aff9"><sup>9</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/341381/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Mathematics, College of Engineering, Mathematics and Physical Sciences, Living Systems Institute, Translational Research Exchange &#x00040; Exeter and The Engineering and Physical Sciences Research Council Centre for Predictive Modelling in Healthcare, University of Exeter</institution>, <addr-line>Exeter</addr-line>, <country>United Kingdom</country></aff>
<aff id="aff2"><sup>2</sup><institution>McCormick Genomics and Proteomics Center, School of Medicine and Health Sciences, The George Washington University</institution>, <addr-line>Washington, DC</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Genetics and Genomics Sciences, Icahn School of Medicine at Mount Sinai</institution>, <addr-line>New York, NY</addr-line>, <country>United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>Cancer Program, Broad Institute of MIT and Harvard</institution>, <addr-line>Cambridge, MA</addr-line>, <country>United States</country></aff>
<aff id="aff5"><sup>5</sup><institution>Medical Oncology, Dana-Farber Cancer Institute</institution>, <addr-line>Boston, MA</addr-line>, <country>United States</country></aff>
<aff id="aff6"><sup>6</sup><institution>Biological Sciences Division, Pritzker School of Medicine, The University of Chicago</institution>, <addr-line>Chicago, IL</addr-line>, <country>United States</country></aff>
<aff id="aff7"><sup>7</sup><institution>Department of Bioinformatics and Mathematical Modelling, Institute of Biophysics and Biomedical Engineering, Bulgarian Academy of Sciences</institution>, <addr-line>Sofia</addr-line>, <country>Bulgaria</country></aff>
<aff id="aff8"><sup>8</sup><institution>Department of Pharmacology and Physiology, School of Medicine and Health Sciences, The George Washington University</institution>, <addr-line>Washington, DC</addr-line>, <country>United States</country></aff>
<aff id="aff9"><sup>9</sup><institution>Department of Biochemistry and Molecular Medicine, School of Medicine and Health Sciences, The George Washington University</institution>, <addr-line>Washington, DC</addr-line>, <country>United States</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Quan Zou, University of Electronic Science and Technology of China, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Stephen J. Bush, University of Oxford, United Kingdom; Xiaojian Shao, National Research Council Canada - Conseil national de recherches Canada, Canada</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Piotr S&#x00142;owi&#x00144;ski <email>p.m.slowinski&#x00040;exeter.ac.uk</email></corresp>
<fn fn-type="other" id="fn001"><p>This article was submitted to Computational Genomics, a section of the journal Frontiers in Bioengineering and Biotechnology</p></fn></author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>09</month>
<year>2020</year>
</pub-date>
<pub-date pub-type="collection">
<year>2020</year>
</pub-date>
<volume>8</volume>
<elocation-id>1021</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>04</month>
<year>2020</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>08</month>
<year>2020</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2020 S&#x00142;owi&#x00144;ski, Li, Restrepo, Alomran, Spurr, Miller, Tsaneva-Atanasova and Horvath.</copyright-statement>
<copyright-year>2020</copyright-year>
<copyright-holder>S&#x00142;owi&#x00144;ski, Li, Restrepo, Alomran, Spurr, Miller, Tsaneva-Atanasova and Horvath</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract><p>Variant allele frequencies (VAF) are an important measure of genetic variation that can be estimated at single-nucleotide variant (SNV) sites. RNA and DNA VAFs are used as indicators of a wide-range of biological traits, including tumor purity and ploidy changes, allele-specific expression and gene-dosage transcriptional response. Here we present a novel methodology to assess gene and chromosomal allele asymmetries and to aid in identifying genomic alterations in RNA and DNA datasets. Our approach is based on analysis of the VAF distributions in chromosomal segments (continuous multi-SNV genomic regions). In each segment we estimate variant probability, a parameter of a random process that can generate synthetic VAF samples that closely resemble the observed data. We show that variant probability is a biologically interpretable quantitative descriptor of the VAF distribution in chromosomal segments which is consistent with other approaches. To this end, we apply the proposed methodology on data from 72 samples obtained from patients with breast invasive carcinoma (BRCA) from The Cancer Genome Atlas (TCGA). We compare DNA and RNA VAF distributions from matched RNA and whole exome sequencing (WES) datasets and find that both genomic signals give very similar segmentation and estimated variant probability profiles. We also find a correlation between variant probability with copy number alterations (CNA). Finally, to demonstrate a practical application of variant probabilities, we use them to estimate tumor purity. Tumor purity estimates based on variant probabilities demonstrate good concordance with other approaches (Pearson&#x00027;s correlation between 0.44 and 0.76). Our evaluation suggests that variant probabilities can serve as a dependable descriptor of VAF distribution, further enabling the statistical comparison of matched DNA and RNA datasets. Finally, they provide conceptual and mechanistic insights into relations between structure of VAF distributions and genetic events. The methodology is implemented in a Matlab toolbox that provides a suite of functions for analysis, statistical assessment and visualization of Genome and Transcriptome allele frequencies distributions. GeTallele is available at: <ext-link ext-link-type="uri" xlink:href="https://github.com/SlowinskiPiotr/GeTallele">https://github.com/SlowinskiPiotr/GeTallele</ext-link>.</p></abstract>
<kwd-group>
<kwd>variant allele fraction (VAF)</kwd>
<kwd>RNA&#x02014;DNA</kwd>
<kwd>earth mover&#x00027;s distance (EMD)</kwd>
<kwd>circos plot</kwd>
<kwd>farey sequence</kwd>
</kwd-group>
<contract-sponsor id="cn001">Wellcome Trust<named-content content-type="fundref-id">10.13039/100004440</named-content></contract-sponsor>
<contract-sponsor id="cn002">Engineering and Physical Sciences Research Council<named-content content-type="fundref-id">10.13039/501100000266</named-content></contract-sponsor>
<counts>
<fig-count count="9"/>
<table-count count="0"/>
<equation-count count="3"/>
<ref-count count="41"/>
<page-count count="16"/>
<word-count count="8766"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>Introduction</title>
<p>RNA and DNA carry and present genetic variation in related yet distinct manners; the differences encoding information about functional and structural traits. In diploid organisms, an important measure of genetic variation is the variant allele frequency (VAF), which can be measured from both genomic (DNA) and transcriptomic (RNA) sequencing data as the encoded and expressed allele frequencies, respectively. Differential DNA-RNA allele frequencies are associated with a variety of biological processes, such as genome admixture, and allele-specific transcriptional regulation (Ha et al., <xref ref-type="bibr" rid="B11">2012</xref>; Shah et al., <xref ref-type="bibr" rid="B32">2012</xref>; Han et al., <xref ref-type="bibr" rid="B12">2015</xref>; Ferreira et al., <xref ref-type="bibr" rid="B10">2016</xref>; Movassagh et al., <xref ref-type="bibr" rid="B26">2016</xref>).</p>
<p>RNA-DNA allele comparisons from sequencing have mostly been approached at the nucleotide level, where they have proven to be highly informative for determining the allelic functional consequences (ENCODE Project Consortium, <xref ref-type="bibr" rid="B9">2012</xref>; Ha et al., <xref ref-type="bibr" rid="B11">2012</xref>; Shah et al., <xref ref-type="bibr" rid="B32">2012</xref>; Morin et al., <xref ref-type="bibr" rid="B25">2013</xref>; Han et al., <xref ref-type="bibr" rid="B12">2015</xref>; Ferreira et al., <xref ref-type="bibr" rid="B10">2016</xref>; Macaulay et al., <xref ref-type="bibr" rid="B23">2016</xref>; Movassagh et al., <xref ref-type="bibr" rid="B26">2016</xref>; Reuter et al., <xref ref-type="bibr" rid="B30">2016</xref>; Shi et al., <xref ref-type="bibr" rid="B33">2016</xref>; Shlien et al., <xref ref-type="bibr" rid="B34">2016</xref>; Yang et al., <xref ref-type="bibr" rid="B39">2016</xref>). Comparatively, integration of allele signals at the molecular level, as derived from linear DNA and RNA, is less comprehensively explored due to the challenges presented by limited compatibility of the outputs from the two sequencing assays.</p>
<p>Herein, we introduce a novel methodology for the analysis of DNA and RNA VAF distributions. This methodology is motivated by the following observations that, to our knowledge, have not been integrated into existing VAF analysis methodologies:
<list list-type="order">
<list-item><p>VAF distributions can change along a chromosome and differ between chromosomal segments (continuous multi-SNV genomic regions);</p></list-item>
<list-item><p>VAF distribution in a chromosomal segment is approximately symmetric;</p></list-item>
<list-item><p>VAF distribution in a chromosomal segment is a reflection of contributions from all the genetic events in all of the cells constituting the sequenced sample;</p></list-item>
<list-item><p>the variant and reference read counts can be modeled as random numbers from a binomial distribution; and</p></list-item>
<list-item><p>the support of the VAF distributions is a Farey sequence.</p></list-item>
</list></p>
<p>Guided by the first three observations, our methodology is designed to provide an aggregate description of VAF distribution in chromosomal segments.</p>
<p>The fourth observation motivates the development of a stochastic model for generating synthetic VAF samples. The model is a binomial mixture model, meaning that each of the mixture components is a binomial distribution parametrised by probability of success given a number of trials. The probability of success is equal across all binomial distributions in the mixture, while the number of trials varies between the mixture components. Each individual component has number of trials that is sampled from the total read counts in the dataset. We interpret the random numbers from this binomial mixture model as the number of variant reads at individual SNV loci. Namely, the common probability of success becomes variant probability, or v<sub>PR</sub>, defined as the probability of observing a variant allele at any site in a given chromosomal segment. We sample the total read counts from the data to account for technical variance arising from the sequencing process. The binomial mixture model implies that a variant or reference read at a given site is a result of a Bernoulli process.</p>
<p>Finally, the fifth observation allows for the rigorous comparison of observed and synthetic VAF distributions, resulting in the estimation of v<sub>PR</sub> of observed VAF distributions.</p>
<p>The potential benefits of the proposed approach are 2-fold: first, by exploiting the statistical relations between SNVs in chromosomal segment, v<sub>PR</sub> is less dependent on read depth and hence can help to utilize sequencing signals more efficiently; second, since v<sub>PR</sub> is a high-level descriptor of VAF distributions, it allows for the direct comparison of DNA and RNA VAF distributions without the effects of limited comparability of DNA and RNA sequencing data.</p></sec>
<sec sec-type="materials and methods" id="s2">
<title>Materials and Methods</title>
<sec>
<title>Data</title>
<p>We evaluate and demonstrate GeTallele&#x00027;s functionality using matched whole exome and RNA sequencing datasets from paired normal and tumor tissue obtained from 72 female patients with breast invasive carcinoma (BRCA) from TCGA. Each dataset contains four matched sequencing sets: normal exome (Nex), normal transcriptome (Ntr), tumor exome (Tex), and tumor transcriptome (Ttr) (see <xref ref-type="supplementary-material" rid="SM1">Supplementary Table 1</xref>). The raw sequencing data were processed as previously described (Movassagh et al., <xref ref-type="bibr" rid="B26">2016</xref>) to generate the inputs for GeTallele.</p>
<p>In short, all datasets were generated through paired-end sequencing on an Illumina HiSeq platform. The human genome reference (hg38)-aligned sequencing reads (Binary Alignment Maps, bams) were downloaded from the Genomic Data Commons Data Portal (<ext-link ext-link-type="uri" xlink:href="https://portal.gdc.cancer.gov/">https://portal.gdc.cancer.gov/</ext-link>) and processed downstream through an in-house pipeline. After variant calling (Li, <xref ref-type="bibr" rid="B20">2011</xref>), the RNA-seq and whole exome sequencing (WES) alignments, together with their respective variant calls, were processed through the read count module of the package RNA2DNAlign (Movassagh et al., <xref ref-type="bibr" rid="B26">2016</xref>), to produce variant and reference sequencing read counts for all the variant positions in all four sequencing signals (normal exome, normal transcriptome, tumor exome and tumor transcriptome). Selected read count assessments were visually examined using the Integrative Genomics Viewer (Thorvaldsd&#x000F3;ttir et al., <xref ref-type="bibr" rid="B37">2013</xref>).</p>
<p>For each sample, to select SNV positions for analysis, we start with heterozygous SNV calls in the normal exome (Li et al., <xref ref-type="bibr" rid="B21">2009</xref>). In each of these positions, we estimate the counts of the variant and reference reads (n<sub>VAR</sub> and n<sub>REF</sub>, respectively) across the 4 matching datasets, and retain positions covered by a minimum total (variant &#x0002B; reference) read depth for further analyses. This threshold is flexible and is required to ensure that only sufficiently covered positions will be analyzed; it is set to 3 in the herein presented results. For further analysis (without loss of generality), we transform all the original VAF values to VAF = |VAF&#x02212;0.5|&#x0002B;0.5. We introduce this transformation due to the symmetric nature of the VAF distributions.</p>
<p>In addition, we required each tumor sample to have at least three of the following five purity estimates&#x02014;Estimate, Absolute, LUMP, IHC, and the consensus purity estimate (CPE) (Katkovnik et al., <xref ref-type="bibr" rid="B16">2002</xref>; Pag&#x000E8;s et al., <xref ref-type="bibr" rid="B29">2010</xref>; Carter et al., <xref ref-type="bibr" rid="B4">2012</xref>; Yoshihara et al., <xref ref-type="bibr" rid="B40">2013</xref>; Zheng et al., <xref ref-type="bibr" rid="B41">2014</xref>; Aran et al., <xref ref-type="bibr" rid="B1">2015</xref>). On the same datasets, we applied THetA (Oesper et al., <xref ref-type="bibr" rid="B27">2013</xref>, <xref ref-type="bibr" rid="B28">2014</xref>)&#x02014;a popular tool for assessing CNA and admixture from sequencing data&#x02014;was also applied to the datasets.</p></sec>
<sec>
<title>Statistics</title>
<p>To test statistical significance, GeTallele uses parametric and non-parametric methods and statistical tests (Hollander et al., <xref ref-type="bibr" rid="B14">2013</xref>; Corder and Foreman, <xref ref-type="bibr" rid="B5">2014</xref>). Namely, to compare distributions of the variant allele frequencies (VAF) we use the Kolmogorov&#x02013;Smirnov test (examples of VAF distributions are depicted in <bold>Figures 2</bold>, <bold>3</bold>). To study concurrence of windows, we use permutation/bootstrap tests. To test relations between v<sub>PR</sub> and copy number alterations (CNA), we use Pearson&#x00027;s correlation coefficient.</p>
<p>To account for multiple comparisons, we set the probability for rejecting the null hypothesis at <italic>p</italic> &#x0003C; 1e&#x02212;5, which corresponds to Bonferroni (Dunn, <xref ref-type="bibr" rid="B8">1961</xref>) family-wise error rate (FWER) correction against 100,000 comparisons. We use a fixed value, rather than other approaches, to ensure better consistency and reproducibility of the results. Alternatively, we apply Benjamini and Hochberg (Benjamini and Hochberg, <xref ref-type="bibr" rid="B2">1995</xref>) false discovery rate (FDR) correction with a probability of accepting false positive results p<sub>FDR</sub> &#x0003C;0.05. We specify the method used in the text when reporting the results.</p>
</sec></sec>
<sec id="s3">
<title>Description of the Novel Methodology</title>
<p>The overall workflow of the proposed methodology as implemented in the GeTallele is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. As input, GeTallele requires the absolute number of sequencing reads bearing the variant and reference nucleotide in each single-nucleotide variant (SNV) position. For each available dataset (4 in the presented analysis) GeTallele estimates VAF based on the variant and reference reads (n<sub>VAR</sub> and n<sub>REF</sub>, respectively) covering the positions of interest: VAF = n<sub>VAR</sub>/(n<sub>VAR</sub> &#x0002B; n<sub>REF</sub>). An example of genome-wide VAF values estimated from tumor exome Tex dataset, and their corresponding histogram is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>GeTallele and visualization of VAF data. <bold>(A)</bold> Toolbox description. <bold>(B)</bold> Visualization of the whole dataset on the level of genome using Circos plot (blue, normal exome; cyan, normal transcriptome; orange, tumor exome; yellow, tumor transcriptome). <bold>(C)</bold> CNA values for chromosome 1. <bold>(D&#x02013;F)</bold> Visualization of the VAF values with fitted variant probability (v<sub>PR</sub>&#x02013;see section Estimation of Variant Probability v<sub>PR</sub> and <xref ref-type="fig" rid="F3">Figure 3</xref>). VAF<sub>TEX</sub> and VAF<sub>TTR</sub> values at the level of: chromosome (chromosome 1) <bold>(D)</bold>, custom genome region <bold>(E)</bold>, and gene <bold>(F)</bold>. <bold>(D)</bold> Shows that there are two chromosomal segments with different VAF distributions, likely representing a region of copy-neutral loss of heterozygosity. <bold>(C)</bold> Shows that large scale change in the CNA is concurrent with the change in the VAF distributions. In panel titles: Tex, v<sub>PR</sub> estimate for VAF distributions of tumor exome (orange); Ttr, v<sub>PR</sub> estimate for VAF distributions of tumor transcriptome (yellow).</p></caption>
<graphic xlink:href="fbioe-08-01021-g0001.tif"/>
</fig>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Sample and distribution of variant allele frequencies (VAF) values. <bold>(A)</bold> All the VAF values of a tumor exome sequencing signal (chromosomes 1&#x02013;22) from one of the datasets. <bold>(B)</bold> Histogram of the VAF values from <bold>(A)</bold>. Centers of bins of the histogram are located at elements of a Farey sequence.</p></caption>
<graphic xlink:href="fbioe-08-01021-g0002.tif"/>
</fig>
<sec>
<title>Data Segmentation</title>
<p>To analyse variant allele frequencies (VAF) at genome-wide level, GeTallele first divides the VAF sequence into a set of non-overlapping segments along the chromosomes. To partition the data into segments, GeTallele uses a parametric global method, which detects the breakpoints in a signal using its mean, as implemented in the Matlab function findchangepts (Lavielle, <xref ref-type="bibr" rid="B18">2005</xref>; Killick et al., <xref ref-type="bibr" rid="B17">2012</xref>) In each segment, the VAFs of the chosen signal must have a different mean than that of the adjacent segment. In the Matlab implementation, sensitivity of breakpoint detection can be controlled using parameter MinThreshold; with a default setting of 0.2. Segments containing fewer than 10 data points were merged with the preceding segment. For analysis of matched signals, segmentation is based on one signal, and then applied to the others. In the presented analysis, segmentation is based mainly on Tex dataset, for comparison we also use Ttr dataset (dataset used for segmentation is specified in the description of the results presented in Section Results).</p></sec>
<sec>
<title>Estimation of Variant Probability v<sub>PR</sub></title>
<p>Variant probability is a biologically interpretable quantitative descriptor of the VAF distribution. It is the common probability of observing a variant allele at any site in a given chromosomal segment. The v<sub>PR</sub> is a measure describing the genomic event that, through the sequencing process, was transformed into an observed distribution of VAFs. For example, in VAF<sub>DNA</sub> from a diploid genome, we assume variant probability v<sub>PR</sub> = 0.5 (meaning that both alleles are equally probable) corresponds to a true allelic ratio of 1:1 for heterozygous sites. The value might differ from 0.5 due to reference mapping biases (Degner et al., <xref ref-type="bibr" rid="B6">2009</xref>). For heterozygous sites in the DNA from a diploid monoclonal samples, the corresponding tumor VAF<sub>DNA</sub> is expected to have the following interpretations: v<sub>PR</sub> = 1 or v<sub>PR</sub> = 0 corresponding to a monoallelic status resulting from a deletion, and v<sub>PR</sub> = 0.8 (or 0.2), 0.75 (or 0.25), 0.67 (or 0.33) corresponding to allele-specific tetra-, tri-, and duplication of the variant-bearing allele, respectively.</p>
<p>The v<sub>PR</sub> of the VAF<sub>RNA</sub> is interpreted as follows. In positions corresponding to heterozygote sites in DNA, alleles not preferentially targeted by regulatory traits are expected to have expression rates with variant probability v<sub>PR</sub> = 0.5, which (by default) scale with the DNA allele distribution. Differences between VAF<sub>DNA</sub> and VAF<sub>RNA</sub> values are observed in special cases of transcriptional regulation where one of the alleles is preferentially transcribed over the other. In the absence of allele-preferential transcription, VAF<sub>DNA</sub>, and VAF<sub>RNA</sub> are anticipated to have similar v<sub>PR</sub> across both diploid (normal) and copy number altered genomic regions. Consequently, VAF<sub>DNA</sub>, and VAF<sub>RNA</sub> are expected to synchronously switch between allelic patterns along the chromosomes, with the switches indicating breakpoints of DNA deletions or amplifications.</p>
<p>Since we observed that DNA and RNA signals have different distributions of total reads and also that the distributions of total reads vary between participants, the synthetic VAF distributions are generated individually for each sequencing signal and each participant.</p>
<p>To estimate v<sub>PR</sub> in the signals, GeTallele first generates synthetic VAF distributions and then uses the earth mover&#x00027;s distance (EMD) (Kantorovich and Rubinstein, <xref ref-type="bibr" rid="B15">1958</xref>; Levina and Bickel, <xref ref-type="bibr" rid="B19">2001</xref>) to fit them to the data. To generate a synthetic VAF distribution with a given variant probability, v<sub>PR</sub>, GeTallele, bootstraps 10,000 values of the total reads (sum of the variant and reference reads; n<sub>VAR</sub> &#x0002B; n<sub>REF</sub>) from the analyzed signal in the dataset. It then uses binomial pseudorandom number generator to get number of successes for given number of total reads and a given value of v<sub>PR</sub> (implemented in the Matlab function binornd). The v<sub>PR</sub> is the common value of the probability of success and generated number of successes is interpreted as an n<sub>VAR</sub>. Since the v<sub>PR</sub> of the synthetic sample can take any value, it can correspond to a single genomic event as well as any combination of genomic events in any mixture of normal and tumor populations (See section v<sub>PR</sub> Values in Mixtures of Normal and Tumour Populations).</p>
<p>The analysis presented in the paper uses 51 synthetic VAF distributions with v<sub>PR</sub> values that vary from 0.5 to 1 with step (increment of) 0.01. The synthetic VAF distributions are parametrized using only v<sub>PR</sub> &#x02265; 0.5, however, to generate them we use v<sub>PR</sub> and its symmetric counterpart 1-v<sub>PR</sub>. The process of generating synthetic VAF distributions along with examples of synthetic and real VAF distributions with different values of v<sub>PR</sub> are illustrated in <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Synthetic and observed VAF distributions. <bold>(A)</bold> Description of the model used to generate synthetic VAF distributions, Bin(n,p) stands for binomial distribution with parameters n (number of trials) and p (probability of success). <bold>(B&#x02013;F)</bold> Synthetic VAF distributions for different values of v<sub>PR</sub>. <bold>(B,F)</bold> Show additionally distributions of VAF<sub>TEX</sub> for the two windows shown in <xref ref-type="fig" rid="F1">Figure 1D</xref>.</p></caption>
<graphic xlink:href="fbioe-08-01021-g0003.tif"/>
</fig>
<p>To estimate v<sub>PR</sub>, we compute the Earth mover&#x00027;s distance between the distribution of VAF values in the considered window and the 51 synthetic VAF distributions (i.e., observed vs. synthetic VAF). The estimate is given by the v<sub>PR</sub> of the synthetic VAF distribution that is closest to the VAF distribution in the segment.</p>
<p>Earth mover&#x00027;s distance (EMD) is a metric for quantifying differences between probability distributions (Kantorovich and Rubinstein, <xref ref-type="bibr" rid="B15">1958</xref>; Levina and Bickel, <xref ref-type="bibr" rid="B19">2001</xref>) and in the case of univariate distributions it can be computed as:
<disp-formula id="E1"><mml:math id="M1"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:mtext>EMD</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>PD</mml:mtext><mml:msub><mml:mrow><mml:mtext>F</mml:mtext></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>PD</mml:mtext><mml:msub><mml:mrow><mml:mtext>F</mml:mtext></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:mi>Z</mml:mi></mml:mrow></mml:msub></mml:mstyle><mml:mo>|</mml:mo><mml:mi>C</mml:mi><mml:mi>D</mml:mi><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>C</mml:mi><mml:mi>D</mml:mi><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>|</mml:mo><mml:mi>d</mml:mi><mml:mi>z</mml:mi><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
Here, PDF<sub>1</sub> and PDF<sub>2</sub> are two probability density functions, and CDF<sub>1</sub> and CDF<sub>2</sub> are their respective cumulative distribution functions. Z is the support of the PDFs (i.e., set of all the possible values of the random variables described by them). Because VAFs are defined as simple fractions with values between 0 and 1, their support is given by a Farey sequence (Hardy and Wright, <xref ref-type="bibr" rid="B13">2008</xref>) of order n; n is the highest denominator in the sequence. For example, Farey sequence of order 2 is 0, 1/2, 1, and Farey sequence of order 3 is 0, 1/3, 1/2, 2/3, 1. We use a Farey sequence of order 1,000 as the support Z for estimating the v<sub>PR</sub>.</p>
<p>Examples of VAF distributions with fitted synthetic VAF distributions are shown in <xref ref-type="fig" rid="F3">Figures 3A,D</xref>. The dependence of the confidence intervals of the estimation on the number of VAF values in a segment is illustrated in <xref ref-type="fig" rid="F4">Figure 4</xref>, which clearly demonstrates that the accuracy of the estimate is positively correlated with the number of VAFs in the chosen segment.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Confidence intervals for artificial samples with different numbers of VAFs. Each confidence interval is based on estimation of v<sub>PR</sub> in 1,000 randomly generated samples with a fixed v<sub>PR</sub> (True value). Light gray bar is 95% confidence interval (950 samples lay within this interval), dark gray bar is 50% confidence interval (500 samples lay within this interval), red dot is median value.</p></caption>
<graphic xlink:href="fbioe-08-01021-g0004.tif"/>
</fig></sec>
<sec>
<title>v<sub>PR</sub> Values in Mixtures of Normal and Tumor Populations</title>
<p>Since the v<sub>PR</sub> can take any value between 0.5 and 1 it can correspond to a single genomic event as well as any combination of genomic events in any mixture of normal and tumor populations. A mixture v<sub>PR</sub> value that corresponds to a combination of genomic events can be computed using the following expression:</p>
<disp-formula id="E2"><mml:math id="M2"><mml:mrow><mml:msub><mml:mtext>v</mml:mtext><mml:mrow><mml:mtext>PR</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mtext>pl</mml:mtext><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mtext>pl</mml:mtext><mml:mo>=</mml:mo><mml:mtext>N</mml:mtext></mml:mrow></mml:msubsup><mml:mrow><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:msub><mml:mtext>e</mml:mtext><mml:mrow><mml:mtext>VAR</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>&#x0007B;</mml:mo><mml:mtext>events</mml:mtext><mml:mo>&#x0007D;</mml:mo></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mtext>e</mml:mtext><mml:mrow><mml:mtext>VAR</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mtext>p</mml:mtext><mml:mrow><mml:mtext>PL</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mtext>pl</mml:mtext><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mtext>pl</mml:mtext><mml:mo>=</mml:mo><mml:mtext>N</mml:mtext></mml:mrow></mml:msubsup><mml:mrow><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:msub><mml:mtext>e</mml:mtext><mml:mrow><mml:mtext>VAR</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>&#x0007B;</mml:mo><mml:mtext>events</mml:mtext><mml:mo>&#x0007D;</mml:mo></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mtext>e</mml:mtext><mml:mrow><mml:mtext>VAR</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mtext>p</mml:mtext><mml:mrow><mml:mtext>PL</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle><mml:mo>+</mml:mo><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mtext>pl</mml:mtext><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mtext>pl=N</mml:mtext></mml:mrow></mml:msubsup><mml:mrow><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:msub><mml:mtext>e</mml:mtext><mml:mrow><mml:mtext>REF</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>&#x0007B;</mml:mo><mml:mtext>events</mml:mtext><mml:mo>&#x0007D;</mml:mo></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mtext>e</mml:mtext><mml:mrow><mml:mtext>REF</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mtext>p</mml:mtext><mml:mrow><mml:mtext>PL</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula>
<p>Where e<sub>VAR</sub> and e<sub>REF</sub> are the multiplicities of variant and reference alleles and p<sub>PL</sub> is a proportion of one of the populations. For heterozygote sites e<sub>VAR</sub> = 1 and e<sub>REF</sub> = 1, for deletions e<sub>VAR</sub> = 0 or e<sub>REF</sub> = 0, for du-, tri- and tetraplications e<sub>VAR</sub> or e<sub>REF</sub> can be equal to 2, 3 or 4, respectively. The sum of proportions p<sub>PL</sub> over the populations is equal 1. For example, for a mixture of 1 normal (N, p<sub>N</sub> = 0.44) and 2 tumor populations (T1, p<sub>T1</sub> = 0.39 and T2, p<sub>T2</sub> = 0.17), T1 with deletion and T2 with deletion the mixture v<sub>PR</sub> value can be computed as follows:</p>
<disp-formula id="E3"><mml:math id="M3"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:msub><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;v</mml:mtext><mml:mrow><mml:mtext>PR</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mtext>p</mml:mtext><mml:mtext>N</mml:mtext></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mtext>B</mml:mtext><mml:mo>+</mml:mo><mml:msub><mml:mtext>p</mml:mtext><mml:mrow><mml:mtext>T</mml:mtext><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mtext>B</mml:mtext><mml:mo>+</mml:mo><mml:msub><mml:mtext>p</mml:mtext><mml:mrow><mml:mtext>T</mml:mtext><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mtext>B</mml:mtext></mml:mrow><mml:mrow><mml:msub><mml:mtext>p</mml:mtext><mml:mtext>N</mml:mtext></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtext>A</mml:mtext><mml:mo>+</mml:mo><mml:mtext>B</mml:mtext></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:msub><mml:mtext>p</mml:mtext><mml:mrow><mml:mtext>T</mml:mtext><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>+</mml:mo><mml:mtext>B</mml:mtext></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:msub><mml:mtext>p</mml:mtext><mml:mrow><mml:mtext>T</mml:mtext><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>+</mml:mo><mml:mtext>B</mml:mtext></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>0.44</mml:mn><mml:mo>&#x000B7;</mml:mo><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mn>0.39</mml:mn><mml:mo>&#x000B7;</mml:mo><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mn>0.17</mml:mn><mml:mo>&#x000B7;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>0.44</mml:mn><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mn>0.39</mml:mn><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mn>0.17</mml:mn><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mn>0.694.</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>By comparing the v<sub>PR</sub> values estimated from data with possible mixture v<sub>PR</sub> values we propose to estimate sample purity and its clonal composition. To this end, we first generate a full set of proportions of all the population in the mixture with step (increment of) 0.01 and compute all the possible v<sub>PR</sub> values that each of the mixtures could produce. For step 0.01: two populations (1 tumor) give 99 proportions, three populations (2 tumors) give 4,851 proportions, four populations (3 tumors) give 156,849 proportions. The matrices with mixture v<sub>PR</sub> values for each proportion, vary from 2 &#x000D7; 2, for two populations with deletions, to 35 &#x000D7; 35 for four populations with all events up to tetra-plications. Then, we run an exhaustive approximate search over all the matrices with mixture v<sub>PR</sub> values over all the proportions. The search is approximate because the estimated v<sub>PR</sub> values have limited accuracy and because we consider only discrete values of proportions. In the analysis we define a match between estimated and mixture v<sub>PR</sub> values if they differ by &#x0003C;0.009 (we chose a value that is smaller than the smallest difference between possible v<sub>PR</sub> estimates). The search returns a large number of admissible mixtures that could produce the estimated v<sub>PR</sub> values. This process is illustrated in <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Mixtures admissible by the v<sub>PR</sub> values estimated from data. To uncover mixtures that could produce the three estimated v<sub>PR</sub> values we perform an exhaustive approximate search of all the possible v<sub>PR</sub> values produced by any mixture of the populations with a given set of genetic events. In each case we generate a full set of proportions with a given step (e.g., 0.01) and compute all the possible v<sub>PR</sub> values that such a mixture could produce. In the illustrated cases: 2 populations (1 tumor) could produce the estimated v<sub>PR</sub> values through a deletion (estimated v<sub>PR</sub> = 0.62 and v<sub>PR</sub> = 0.63) and via deletion of one allele and duplication of another (estimated v<sub>PR</sub> = 0.69); 3 populations (2 tumors) could produce the estimated v<sub>PR</sub> values through a deletion in one of the tumor populations (estimated v<sub>PR</sub> = 0.62 and v<sub>PR</sub> = 0.63) and via deletion in both of the tumor populations (estimated v<sub>PR</sub> = 0.69). The 2 populations case admits a single mixture and the 3 populations allow 9 mixtures with similar compositions. The admissible mixtures are depicted on the ternary plots, red circle indicates solution corresponding to the presented matrix. We exclude mixture v<sub>PR</sub> values that result from deletion of both the variant and reference alleles (empty fields in the matrices).</p></caption>
<graphic xlink:href="fbioe-08-01021-g0005.tif"/>
</fig>
<p>To visualize the admissible mixtures, we use ternary plots, which allow us to illustrate composition of three components in two dimensions. The composition, represented by ratios of the three components, which sum to a constant, is depicted as point inside or on the edge of an equilateral triangle. If the point is on the edges, the composition has only two components. To help interpretation of the ternary plots, we also plot the grid lines that are parallel to the sides of the triangle. These gridlines indicate the directions of constant ratios of the components. Along such direction the ratio of one of the components is fixed and only the other two ratios vary. Examples of visualization of admissible mixtures on ternary plots are shown in <xref ref-type="fig" rid="F5">Figures 5</xref>, <xref ref-type="fig" rid="F6">6</xref>.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Admissible mixtures for increasing mixture complexity. <bold>(A)</bold> Shows admissible mixtures for 3 different mixtures with increasing complexity. The simplest mixture (mixture with the lowest number of components and the simplest set of genetic events) is show within the gray frame. On each ternary plot, the admissible mixtures are indicated by gray dots. The green axis indicates proportion of the normal population (N), the yellow axis indicates proportion of the 1st tumor population (T1), the blue axis shows proportion of the 2nd tumor population (T2), or sum of the 2nd and 3rd tumor populations (T2 &#x0002B; T3). <bold>(B)</bold> Schematic representation of increasing complexity of the mixture models. From a mixture of 1 normal and 1 tumor population in which only deletion is possible to a model with 1 normal and 3 tumor populations and each can have deletions, du-, tri-, and tetraplications.</p></caption>
<graphic xlink:href="fbioe-08-01021-g0006.tif"/>
</fig>
<p>To facilitate analysis of the admissible mixtures returned by the search procedure we introduce mixture complexity. Mixture complexity is a measure that increases with number of populations as well as with variety of genetic events. From the simplest mixture of 1 normal and 1 tumor population in which only deletions are possible to a model with 1 normal and multiple tumor populations where each can have deletions, and any level of multiplications. In practice, we set the limit at 3 tumor populations and tetra-plications. Mixture complexity helps to group and visualize admissible mixtures. Mixtures with higher complexity allow more possible v<sub>PR</sub> values, meaning that it is easier to find the match with the estimated v<sub>PR</sub> values but that the number of admissible mixtures increases (see <xref ref-type="fig" rid="F6">Figure 6</xref>). We, further, observe that proportion of normal population, p<sub>N</sub>, increases with a number of clonal tumor populations included in the model mixture and that, generally, p<sub>N</sub> stays constant with increasing variety of genetic events, for a fixed number of clonal tumor populations. We note that this is just one of many possible ways of deciding which solution should be chosen.</p></sec></sec>
<sec sec-type="results" id="s4">
<title>Results</title>
<p>To evaluate the proposed methodology, we apply it on matched normal and tumor exome and transcriptome sequencing data of 72 breast carcinoma (BRCA) datasets with pre-assessed copy-number and genome admixture estimates acquired through TCGA (see Materials and Methods). We first compare DNA and RNA VAF distributions from matched sequencing datasets and find that both genomic signals give very similar results in terms of segmentation and estimated variant probability values. We further assess the correlations between v<sub>PR</sub> values and copy number alterations (CNA) values and find that they are in agreement with each other. Finally, we use the v<sub>PR</sub> values to estimate tumor purity. The purity estimates based on v<sub>PR</sub> values show good concordance with alternative approaches.</p>
<sec>
<title>Segmentation Results</title>
<p>Segmentation of the data, based on the tumor exome signal, resulted in 2,697 chromosomal segments across the 72 datasets. We excluded from further analysis 294 chromosomal segments where either tumor exome or transcriptome had v<sub>PR</sub> &#x0003E; = 0.58 but their VAF distribution could not be differentiated from the model VAF distributions with v<sub>PR</sub> = 0.5 (<italic>p</italic> &#x0003E; 1e&#x02212;5, Kolmogorov Smirnov test, equivalent to Bonferroni FWER correction for 100,000 comparisons). The 294 excluded chromosomal segments, corresponding to 4% of the total length of the data in base pairs and 4% of all the available data points. This implies these short segments containing few VAF values. In the remaining 2,403 chromosomal segments, we systematically examined the similarity between corresponding VAF<sub>TEX</sub> (tumor exome), VAF<sub>TTR</sub> (tumor transcriptome), and CNA. We obtained several distinct patterns of coordinated RNA-DNA allelic behavior as well as correlations with CNA data.</p>
<p>In 60% of all analyzed chromosomal segments the distributions of VAF<sub>TEX</sub> and VAF<sub>TTR</sub> were statistically concordant (<italic>P</italic> &#x0003E; 1e&#x02212;5, Kolmogorov Smirnov test), and in 40% they were statistically discordant (<italic>P</italic> &#x0003C; 1e&#x02212;5, Kolmogorov Smirnov test). In two chromosomal segments, VAF<sub>TEX</sub> and VAF<sub>TTR</sub>, had the same v<sub>PR</sub>, while having statistically different VAF distributions (P &#x0003C; 1e&#x02212;5, Kolmogorov Smirnov test). We consider such chromosomal segments as concordant. The v<sub>PR</sub> robustly characterizes VAF sample while the Kolmogorov-Smirnov test is very sensitive for differences between distributions that might be caused by to technical variance. In the vast majority of the discordant chromosomal segments v<sub>PR</sub> of the VAF<sub>TTR</sub>, v<sub>PR,TTR</sub>, was higher than v<sub>PR</sub> of the VAF<sub>TEX</sub>, v<sub>PR,TEX</sub>, (only in 21 out of 959 discordant chromosomal segments v<sub>PR,TTR</sub> was lower than v<sub>PR,TEX</sub>).</p></sec>
<sec>
<title>Concurrence of Segmentation Based on DNA and RNA</title>
<p>We next analyzed the concurrence between chromosomal segments resulting from independent segmentations of the tumor exome (VAF<sub>TEX</sub>) and transcriptome (VAF<sub>TTR</sub>) datasets (2,697 and 3,605 chromosomal segments, respectively, across all the samples). We first assessed chromosome-wise alignment of the start and end points of the chromosomal segments. In 45% of the chromosomes both VAF<sub>TEX</sub> and VAF<sub>TTR</sub> signals produce a single segment that contains the whole chromosome. In 33% of chromosomes both signals produced multiple chromosomal segments. These chromosomal segments are well aligned, with 90% of the breakpoints differing &#x0003C;7% of data points in the chromosome, e.g., they are &#x0003C;70 points apart if the chromosome contains 1,000 data points; Q50 = 0.02%, Q75 = 2% of data points in the chromosome. The probability of observing such an alignment by chance is smaller than <italic>p</italic> = 1e&#x02212;5 (100,000 bootstrap samples with breakpoints assigned randomly in all the individual chromosomes where both signals produced multiple chromosomal segments). In 22% of the chromosomes, segments based on VAF<sub>TEX</sub> and VAF<sub>TTR</sub> signals were positionally discordant&#x02014;one signal produced a single segment containing whole chromosome while the other produced multiple chromosomal segments.</p>
<p>To compare the v<sub>PR</sub> values in the 55% of chromosomes where at least one signal produced more than one chromosomal segment, we computed chromosome-wise mean absolute error (MAE) between the v<sub>PR</sub> in two sets of chromosomal segments. To account for different start and end points of the segments we interpolated the v<sub>PR</sub> values (nearest neighbor interpolation) at each data point in the chromosome. We separately compared the v<sub>PR,TEX</sub> and v<sub>PR,TTR</sub> values. Assessment of alignment using MAE showed strong concordance: v<sub>PR,TEX</sub> agreed perfectly in 11% of the chromosomes and had the percentiles of MAE equal to Q50 = 0.012, Q75 = 0.022 and Q97.5 = 0.047, while v<sub>PR,TTR</sub> agreed perfectly in 8% but had slightly higher percentiles of MAE Q50 = 0.019, Q75 = 0.034 and Q97.5 = 0.07. v<sub>PR,TEX</sub> and v<sub>PR,TTR</sub> values had MAE = 0 simultaneously in 4% of the chromosomes. Probability of observing such values of MAE by chance is smaller than <italic>p</italic> = 1e&#x02212;3 (1,000 random assignments of v<sub>PR,TEX</sub> and v<sub>PR,TTR</sub> values to windows in the 873 chromosomes where at least one signal had more than one chromosomal segment). It is noteworthy that MAE Q97.5 &#x0003C; 0.07 is comparable with the confidence interval of single v<sub>PR</sub> estimate based on 50 VAF values. In other words, both signals in a sample (Tex and Ttr) give very similar results in terms of segmentation and estimated values of the v<sub>PR</sub>. Albeit, segmentation of VAF<sub>TTR</sub> generates a higher number of chromosomal segments. The higher number of VAF<sub>TTR</sub> chromosomal segments indicates that transcriptional regulation occurs at a smaller scale than alterations in DNA. <xref ref-type="fig" rid="F7">Figure 7</xref> shows examples of concurrence between chromosomal segments based on VAF<sub>TEX</sub> and VAF<sub>TTR</sub> signals in a positionally concordant chromosome (both signals produced multiple segments).</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Illustration of concurrence between chromosomal segment resulting from independent segmentations of the dataset based on the VAF<sub>TEX</sub> and VAF<sub>TTR</sub> signals. <bold>(A)</bold> Yellow dots, VAF<sub>TTR</sub>; gray circles, v<sub>PR,TTR</sub> interpolated at all data points in segments based on VAF<sub>TEX</sub>; yellow crosses, v<sub>PR,TTR</sub> interpolated at all data points in segments based on VAF<sub>TTR</sub>. <bold>(B)</bold> Bar plot of the absolute difference between the v<sub>PR</sub> values in the two kinds of chromosomal segments. <bold>(C)</bold> orange dots, VAF<sub>TEX</sub>; gray crosses, v<sub>PR,TEX</sub> interpolated at all data points in segments based on VAF<sub>TEX</sub>; orange dots v<sub>PR,TEX</sub> interpolated at all data points in segments based on VAF<sub>TTR</sub>. <bold>(D)</bold> Bar plot of the absolute difference between the v<sub>PR</sub> values in the two kinds of chromosomal segments.</p></caption>
<graphic xlink:href="fbioe-08-01021-g0007.tif"/>
</fig></sec>
<sec>
<title>Correlation Between v<sub>PR</sub> and CNA</title>
<p>Finally, we assess the correlations between v<sub>PR</sub> and CNA in the individual samples. We separately computed correlations for deletions and amplifications. In order to separate deletions and amplifications, for each data set we found CNA<sub>MIN</sub>, value of the CNA in the range &#x02212;0.3 to 0.3 that had the smallest corresponding v<sub>PR,TEX</sub>. To account for observed variability of the CNA values near the CNA<sub>MIN</sub>, we set the threshold for amplifications to CNA<sub>AMPLIFICATION</sub> = CNA<sub>MIN</sub>-0.05, and for deletions we set it to CNA<sub>DELETION</sub> = CNA<sub>MIN</sub> &#x0002B; 0.05 (each data set had a different threshold).</p>
<p>For VAF<sub>TEX</sub>, we observed significant correlations with negative trend between v<sub>PR,TEX</sub> and CNA &#x02264; CNA<sub>DELETION</sub> in 57 datasets and with a positive trend between v<sub>PR,TEX</sub> and CNA &#x02265; CNA<sub>AMPLIFICATION</sub> in 39 datasets (p<sub>FDR</sub> &#x0003C; 0.05, Pearson&#x00027;s correlation with Benjamini Hochberg multiple comparison correction for 72 samples). For VAF<sub>TTR</sub>, we observed significant correlations with a negative trend between v<sub>PR,TTR</sub> and CNA &#x02264; CNA<sub>DELETION</sub> in 62 datasets and with positive trend between v<sub>PR,TTR</sub> and CNA &#x02265; CNA<sub>AMPLIFICATION</sub> in 33 datasets (p<sub>FDR</sub> &#x0003C; 0.05, Pearson correlation with Benjamini Hochberg correction). These correlations indicate that the segmentation and the estimated v<sub>PR</sub> values are concordant with CNA calls. However, the v<sub>PR</sub> values (estimated at the level of chromosomal segments) do not differentiate between positive and negative values of the CNA, meaning it is not possible to use v<sub>PR</sub> alone to call amplifications and deletions.</p>
<p><xref ref-type="fig" rid="F8">Figure 8</xref> shows four typical patterns of correlation between the CNA and v<sub>PR</sub> values observed in the data. In <xref ref-type="fig" rid="F8">Figure 8A</xref>, all the values of CNA are close to CNA<sub>MIN</sub>. In <xref ref-type="fig" rid="F8">Figure 8B</xref>, the relationship between CNA and v<sub>PR</sub> is noisy, only correlations between v<sub>PR,TTR</sub> and CNA &#x02264; CNA<sub>DELETION</sub> are statistically significant (r<sub>TEX,CNA,DEL</sub> = &#x02212;0.29, p<sub>FDR</sub> = 0.063; r<sub>TEX,CNA,DEL</sub> = &#x02212;0.38, p<sub>FDR</sub> = 0.012; r<sub>TEX,CNA,AMPL</sub> = 0.14, p<sub>FDR</sub> = 0.58; r<sub>TEX,CNA,AMPL</sub> = 0.19, p<sub>FDR</sub> = 0.47; Pearson&#x00027;s correlation with Benjamini Hochberg multiple comparison correction for 72 samples). In <xref ref-type="fig" rid="F8">Figure 8C</xref> all the correlations are statistically significant, v<sub>PR,TTR</sub> values (circles) follow closely the v<sub>PR,TEX</sub> (squares) indicating that in most of the windows distributions of the VAF<sub>TEX</sub> and VAF<sub>TTR</sub> are concordant (r<sub>TEX,CNA,D</sub> = &#x02212;0.91, p<sub>FDR</sub> &#x0003C; 1e&#x02212;10; r<sub>TEX,CNA,DEL</sub> = &#x02212;0.96, p<sub>FDR</sub> &#x0003C; 1e&#x02212;10; r<sub>TEX,CNA,AMPL</sub> = 0.92, p<sub>FDR</sub> &#x0003C; 1e&#x02212;10; r<sub>TEX,CNA,AMPL</sub> = 0.95, p<sub>FDR</sub> &#x0003C; 1e&#x02212;10). In <xref ref-type="fig" rid="F8">Figure 8D</xref> correlations between v<sub>PR,TEX</sub>, v<sub>PR,TTR</sub> and CNA &#x02264; CNA<sub>D</sub> are statistically significant, but there is a large difference (with median of 0.18) between v<sub>PR,TEX</sub> and v<sub>PR,TTR</sub> values, indicating that in most of the windows the distributions of the VAF<sub>TEX</sub> and VAF<sub>TTR</sub> in this dataset are discordant (r<sub>TEX,CNA,DEL</sub> = &#x02212;0.44, p<sub>FDR</sub> = 0.047; r<sub>TEX,CNA,DEL</sub> = &#x02212;0.64, p<sub>FDR</sub> = 0.0017; r<sub>TEX,CNA,AMPL</sub> = 0.44 p<sub>FDR</sub> = 0.16; r<sub>TEX,CNA,AMPL</sub> = 0.28, p<sub>FDR</sub> = 0.41). In many of the datasets we observe that the v<sub>PR,TTR</sub> values are higher than the corresponding v<sub>PR,TEX</sub> values (median v<sub>PR,TTR</sub>-v<sub>PR,TEX</sub> = 0.03), likely indicative of preferential transcription of some alleles in the chromosomal segment. Correlations between v<sub>PR</sub> and CNA in all datasets are shown in the <xref ref-type="supplementary-material" rid="SM2">Supplementary Figure 1</xref>.</p>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>Illustration of the correlations between v<sub>PR</sub> and CNA. Orange squares v<sub>PR,TEX</sub>, yellow circles v<sub>PR,TTR</sub>. Lines, least-squares fitted trends for significant correlations (orange correlation with v<sub>PR,TEX</sub>, yellow correlation with v<sub>PR,TTR</sub>). Black, v<sub>PR</sub> for CNA<sub>MIN</sub> &#x000B1; 0.05. Correlations for all the datasets are shown in <xref ref-type="supplementary-material" rid="SM2">Supplementary Figure 1</xref>. <bold>(A)</bold> All the values of CNA are close to CNA<sub>MIN</sub> = 0. <bold>(B)</bold> Relationship between CNA and v<sub>PR</sub> is noisy, only some correlations are statistically significant. <bold>(C)</bold> All the correlations are statistically significant, v<sub>PR,TTR</sub> values (circles) follow closely the v<sub>PR,TEX</sub> (squares) indicating concordance of the VAF<sub>TEX</sub> and VAF<sub>TTR</sub> distributions. <bold>(D)</bold> Only correlations for CNA &#x02264; CNA<sub>D</sub> are statistically significant.</p></caption>
<graphic xlink:href="fbioe-08-01021-g0008.tif"/>
</fig></sec>
<sec>
<title>v<sub>PR</sub> Based Purity Estimation</title>
<p>To demonstrate a practical application of the v<sub>PR</sub> values we use them to estimate tumor purity of the samples. To this end we compared the v<sub>PR</sub> based purity (VBP) estimates with ESTIMATE, ABSOLUTE, LUMP, IHC, and the Consensus Purity Estimation (CPE) (Katkovnik et al., <xref ref-type="bibr" rid="B16">2002</xref>; Pag&#x000E8;s et al., <xref ref-type="bibr" rid="B29">2010</xref>; Carter et al., <xref ref-type="bibr" rid="B4">2012</xref>; Yoshihara et al., <xref ref-type="bibr" rid="B40">2013</xref>; Zheng et al., <xref ref-type="bibr" rid="B41">2014</xref>; Aran et al., <xref ref-type="bibr" rid="B1">2015</xref>).</p>
<p>To obtain the VBP estimate we used v<sub>PR,TEX</sub> values. We, first, selected the v<sub>PR,TEX</sub> values that: 1. are estimated with high confidence, i.e., are based on at least 50 VAF values; 2. are most likely heterozygous in normal exome, i.e., have a corresponding v<sub>PR</sub> value in normal exome v<sub>PR,NEX</sub> &#x0003C;0.58; 3. most likely have v<sub>PR,TEX</sub> &#x0003E; 0.5, i.e., their <italic>p-</italic>value for comparison with v<sub>PR,TEX</sub> = 0.5 is very small <italic>p</italic> &#x0003C; 1e&#x02212;5 (Kolmogorov-Smirnov test).</p>
<p>Next, we used the selected v<sub>PR,TEX</sub> values to find all admissible mixtures (with 1&#x02013;3 tumor populations and allowing for all events, from deletions to tetraplications). To estimate the VBP, out of all the admissible mixtures we chose these with lowest mixture complexity and among these mixtures we take one with the highest p<sub>N</sub> (proportion of the normal population). The VBP, percentage of tumor populations in the sample, is then given as 1-p<sub>N</sub>. Such approach provides rather conservative estimates of VBP (the smallest 1-p<sub>N</sub>). However, GetAllele can be extended to offer alternative methods of employing the admissible mixtures to estimate VBP. Development, analysis and comparison of alternative VBP estimation methods is beyond scope of the current paper.</p>
<p><xref ref-type="fig" rid="F9">Figure 9A</xref> shows violin plots of all considered 1-p<sub>N</sub> values and (x) indicates the smallest value taken as a VBP estimate. In two of the datasets we could not estimate the purity due to lack of suitable v<sub>PR,TEX</sub> values. The VBP estimates shows the best agreement with ABSOLUTE method (y = 0.86 x &#x0002B; 0.02, r = 0.76, <italic>p</italic> &#x0003C; 3.4e&#x02212;14, Pearson&#x00027;s correlation, <xref ref-type="fig" rid="F9">Figure 9B2</xref>). We suppose that this is because the ABSOLUTE method is based on copy number distributions, and our analysis (Section Correlation Between v<sub>PR</sub> and CNA) revealed high correlations between the CNAs and v<sub>PR</sub> values. Similar, to the ABSOLUTE method, VBP estimates are generally lower than the other purity estimates (ESTIMATE, LUMP, IHC, CPE); see <xref ref-type="fig" rid="F9">Figures 9B1</xref>&#x02013;<xref ref-type="fig" rid="F5">5</xref>.</p>
<fig id="F9" position="float">
<label>Figure 9</label>
<caption><p>Illustration of purity estimation based on model mixtures and v<sub>PR,TEX</sub>. Comparison of the Estimate (EST), Absolute (ABS), LUMP, IHC, and the consensus purity estimate (CPE) methods with v<sub>PR</sub> based purity (VBP). <bold>(A)</bold> Violin plots show distributions of purity based on all the admissible proportions of the normal population (x) indicates the lowest value selected as the most conservative estimate; colors corresponding to the different methods are indicated in <bold>(B1&#x02013;5)</bold>. <bold>(B1&#x02013;5)</bold> Correlation of the VBP with individual methods; colored line indicates best fit linear trend. <bold>(C)</bold> Matrix showing significant <italic>p</italic> &#x0003C; 0.05 Pearson&#x00027;s correlation coefficients between all tested methods.</p></caption>
<graphic xlink:href="fbioe-08-01021-g0009.tif"/>
</fig>
<p>The approach presented in this section differs from other methods for inferring genomic mixture composition in that it is based on chromosomal segments with at least 50 VAF values which can extend over millions of base pairs. In contrast, PyClone (Roth et al., <xref ref-type="bibr" rid="B31">2014</xref>) is based on sets of carefully selected individual deeply sequenced VAF values, while SciClone (Miller et al., <xref ref-type="bibr" rid="B24">2014</xref>) and TPES (Locallo et al., <xref ref-type="bibr" rid="B22">2019</xref>) are based on analysis of selected VAF values aggregated from the genome-wide sequences (multiple chromosomes). By using chromosomal segments, v<sub>PR</sub> allows for a more granular description of the VAF distributions than aggregating genome-wide VAF values. At the same time, basing purity estimation on v<sub>PR</sub> values allows for the use of SNVs with a low sequencing depth (3 in the presented analysis). Rigorous comparison of the performance of the different methods is beyond the scope of this demonstration of potential practical applications of v<sub>PR</sub>.</p></sec></sec>
<sec sec-type="discussion" id="s5">
<title>Discussion</title>
<p>We present a novel methodology to assess allele asymmetries in RNA and DNA datasets using VAF. Simultaneous analysis of RNA and DNA VAF is becoming more feasible with the growing accessibility of paired RNA and DNA sequencing datasets from the same individual (ENCODE Project Consortium, <xref ref-type="bibr" rid="B9">2012</xref>; Macaulay et al., <xref ref-type="bibr" rid="B23">2016</xref>; Reuter et al., <xref ref-type="bibr" rid="B30">2016</xref>). Our approach addresses the compatibility between RNA and DNA VAF estimations and the high VAF variability by introducing variant probability, v<sub>PR</sub>, a high-level descriptor of VAF distributions in chromosomal segments (continuous multi-SNV genomic regions).</p>
<p>v<sub>PR</sub> is a parameter of a stochastic model of VAF distributions that allows for the generation of synthetic VAF samples that closely resembles the observed data. The simplicity and transparency of v<sub>PR</sub> is one of the biggest advantages of the presented methodology over other existing methods.</p>
<p>Using variant probability, we analyzed relationships between DNA and RNA VAF estimations and biological processes. We observed that, in chromosomes affected by deletions and amplifications, VAF<sub>RNA</sub> and VAF<sub>DNA</sub> showed highly concordant breakpoint calls. This indicates that VAF<sub>RNA</sub> alone can serve as preliminary indicator for break points of DNA deletions or amplifications if they fall within the regions covered by sequencing, and potential could facilitate the estimation of CNAs from RNA-sequencing data. Furthermore, a large proportion of v<sub>PR</sub> estimates based on VAF<sub>RNA</sub> samples are higher than v<sub>PR</sub> estimates based on VAF<sub>DNA</sub> indicating preferential transcription of some alleles in a number of chromosomal segments. Finally, we showcased that matched v<sub>PR,NEX</sub> and v<sub>PR,TEX</sub> values can be used to model the proportions of normal and tumor populations, thereby providing an estimate of the tumor purity. The purity estimates based on variant probabilities show good concordance with other approaches (Pearson&#x00027;s correlation between 0.44 and 0.76; as illustrated in <xref ref-type="fig" rid="F9">Figure 9</xref>). Additionally, once the mixture composition is estimated, v<sub>PR</sub> values allow for the interrogation of genetic events in each population at a specific chromosomal segment (as illustrated in <xref ref-type="fig" rid="F5">Figure 5</xref>).</p>
<p>Since VAF estimations can be affected by allele mapping bias (Degner et al., <xref ref-type="bibr" rid="B6">2009</xref>) which can lead to overestimation of the reference allele count (Brandt et al., <xref ref-type="bibr" rid="B3">2015</xref>), we suggest that GetAllele input is generated from SNV-aware alignments, which perform better in VAF-based downstream analyses (Spurr et al., <xref ref-type="bibr" rid="B36">2020</xref>). We note that SNV-aware alignments are now facilitated by recent methodological advances, including the implementation of the WASP method (Van De Geijn et al., <xref ref-type="bibr" rid="B38">2015</xref>) in the STAR aligner (Dobin et al., <xref ref-type="bibr" rid="B7">2013</xref>).</p>
<p>Based on our results, variant probabilities can serve as a dependable descriptor of VAF distribution and can be used to assess allele asymmetries or to aid in making matched calls of genomic events in sequencing RNA and DNA datasets without limitations caused by their different molecular nature. Finally, v<sub>PR</sub> provides conceptual and mechanistic insights into relationships between VAF distributions and underlying genetic events.</p>
<p>Methods for estimating and analyzing v<sub>PR</sub> values are implemented in a GeTallele toolbox. GeTallele allows to analyse and visualize patterns observed in the VAF distributions at a desired resolution, such as the chromosome, gene or other custom genomic level.</p></sec>
<sec sec-type="data-availability-statement" id="s6">
<title>Data Availability Statement</title>
<p>The data analyzed in this study is subject to the following licenses/restrictions: The datasets used and/or analyzed during the current study are available from the corresponding author on reasonable request. Requests to access these datasets should be directed to <email>p.m.slowinski&#x00040;exeter.ac.uk</email>.</p></sec>
<sec id="s7">
<title>Author Contributions</title>
<p>PS, ML, PR, NA, LS, CM, KT-A, and AH conception and design of the work. ML data acquisition. PS data analysis. PS, ML, PR, LS, KT-A, and AH interpretation of data. PS creation of new software used in the work. PS, ML, PR, LS, KT-A, and AH have drafted the work or substantively revised it. All authors approved the submitted version. All authors agreed both to be personally accountable for the author&#x00027;s own contributions and to ensure that questions related to the accuracy or integrity of any part of the work, even ones in which the author was not personally involved, are appropriately investigated, resolved, and the resolution documented in the literature.</p>
</sec>
<sec id="s8">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p></sec>
</body>
<back>
<ack><p>This manuscript has been released as a pre-print at bioRxiv <ext-link ext-link-type="uri" xlink:href="https://www.biorxiv.org/content/10.1101/491209v3">https://www.biorxiv.org/content/10.1101/491209v3</ext-link>, (S&#x00142;owi&#x00144;ski et al., <xref ref-type="bibr" rid="B35">2020</xref>).</p>
</ack>
<sec sec-type="supplementary-material" id="s9">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fbioe.2020.01021/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fbioe.2020.01021/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_1.pdf" id="SM2" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Aran</surname> <given-names>D.</given-names></name> <name><surname>Sirota</surname> <given-names>M.</given-names></name> <name><surname>Butte</surname> <given-names>A. J.</given-names></name></person-group> (<year>2015</year>). <article-title>Systematic pan-cancer analysis of tumour purity</article-title>. <source>Nat. Commun</source>. <volume>6</volume>:<fpage>8971</fpage>. <pub-id pub-id-type="doi">10.1038/ncomms9971</pub-id><pub-id pub-id-type="pmid">26634437</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Benjamini</surname> <given-names>Y.</given-names></name> <name><surname>Hochberg</surname> <given-names>Y.</given-names></name></person-group> (<year>1995</year>). <article-title>Controlling the false discovery rate - a practical and powerful approach to multiple testing</article-title>. <source>J. R. Stat. Soc. B</source> <volume>57</volume>, <fpage>289</fpage>&#x02013;<lpage>300</lpage>. <pub-id pub-id-type="doi">10.1111/j.2517-6161.1995.tb02031.x</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brandt</surname> <given-names>D. Y.</given-names></name> <name><surname>Aguiar</surname> <given-names>V. R.</given-names></name> <name><surname>Bitarello</surname> <given-names>B. D.</given-names></name> <name><surname>Nunes</surname> <given-names>K.</given-names></name> <name><surname>Goudet</surname> <given-names>J.</given-names></name> <name><surname>Meyer</surname> <given-names>D.</given-names></name></person-group> (<year>2015</year>). <article-title>Mapping bias overestimates reference allele frequencies at the HLA genes in the 1000 genomes project phase I data</article-title>. <source>G3</source> <volume>5</volume>, <fpage>931</fpage>&#x02013;<lpage>941</lpage>. <pub-id pub-id-type="doi">10.1534/g3.114.015784</pub-id><pub-id pub-id-type="pmid">25787242</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Carter</surname> <given-names>S. L.</given-names></name> <name><surname>Cibulskis</surname> <given-names>K.</given-names></name> <name><surname>Helman</surname> <given-names>E.</given-names></name> <name><surname>McKenna</surname> <given-names>A.</given-names></name> <name><surname>Shen</surname> <given-names>H.</given-names></name> <name><surname>Zack</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>Absolute quantification of somatic DNA alterations in human cancer</article-title>. <source>Nat. Biotechnol</source>. <volume>30</volume>, <fpage>413</fpage>&#x02013;<lpage>421</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.2203</pub-id><pub-id pub-id-type="pmid">22544022</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Corder</surname> <given-names>G. W.</given-names></name> <name><surname>Foreman</surname> <given-names>D. I.</given-names></name></person-group> (<year>2014</year>). <source>Nonparametric Statistics</source>. <publisher-loc>Hoboken, NJ</publisher-loc>: <publisher-name>John Wiley &#x00026; Sons</publisher-name>.</citation></ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Degner</surname> <given-names>J. F.</given-names></name> <name><surname>Marioni</surname> <given-names>J. C.</given-names></name> <name><surname>Pai</surname> <given-names>A. A.</given-names></name> <name><surname>Pickrell</surname> <given-names>J. K.</given-names></name> <name><surname>Nkadori</surname> <given-names>E.</given-names></name> <name><surname>Gilad</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2009</year>). <article-title>Effect of read-mapping biases on detecting allele-specific expression from RNA-sequencing data</article-title>. <source>Bioinformatics</source> <volume>25</volume>, <fpage>3207</fpage>&#x02013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp579</pub-id><pub-id pub-id-type="pmid">19808877</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dobin</surname> <given-names>A.</given-names></name> <name><surname>Davis</surname> <given-names>C. A.</given-names></name> <name><surname>Schlesinger</surname> <given-names>F.</given-names></name> <name><surname>Drenkow</surname> <given-names>J.</given-names></name> <name><surname>Zaleski</surname> <given-names>C.</given-names></name> <name><surname>Jha</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>STAR: ultrafast universal RNA-seq aligner</article-title>. <source>Bioinformatics</source> <volume>29</volume>, <fpage>15</fpage>&#x02013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bts635</pub-id><pub-id pub-id-type="pmid">23104886</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dunn</surname> <given-names>O. J.</given-names></name></person-group> (<year>1961</year>). <article-title>Multiple comparisons among means</article-title>. <source>J. Am. Stat. Assoc</source>. <volume>56</volume>, <fpage>52</fpage>&#x02013;<lpage>64</lpage>. <pub-id pub-id-type="doi">10.1080/01621459.1961.10482090</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><collab>ENCODE Project Consortium</collab></person-group> (<year>2012</year>). <article-title>An integrated encyclopedia of DNA elements in the human genome</article-title>. <source>Nature</source> <volume>489</volume>:<fpage>57</fpage>. <pub-id pub-id-type="doi">10.1038/nature11247</pub-id><pub-id pub-id-type="pmid">22955616</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ferreira</surname> <given-names>E.</given-names></name> <name><surname>Shaw</surname> <given-names>D. M.</given-names></name> <name><surname>Oddo</surname> <given-names>S.</given-names></name></person-group> (<year>2016</year>). <article-title>Identification of learning-induced changes in protein networks in the hippocampi of a mouse model of Alzheimer&#x00027;s disease</article-title>. <source>Transl. Psychiatry</source>. <volume>6</volume>:<fpage>e849</fpage>. <pub-id pub-id-type="doi">10.1038/tp.2016.114</pub-id><pub-id pub-id-type="pmid">27378549</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ha</surname> <given-names>G.</given-names></name> <name><surname>Roth</surname> <given-names>A.</given-names></name> <name><surname>Lai</surname> <given-names>D.</given-names></name> <name><surname>Bashashati</surname> <given-names>A.</given-names></name> <name><surname>Ding</surname> <given-names>J.</given-names></name> <name><surname>Goya</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>Integrative analysis of genome-wide loss of heterozygosity and monoallelic expression at nucleotide resolution reveals disrupted pathways in triple-negative breast cancer</article-title>. <source>Genome Res</source>. <volume>22</volume>, <fpage>1995</fpage>&#x02013;<lpage>2007</lpage>. <pub-id pub-id-type="doi">10.1101/gr.137570.112</pub-id><pub-id pub-id-type="pmid">22637570</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han</surname> <given-names>L.</given-names></name> <name><surname>Vickers</surname> <given-names>K. C.</given-names></name> <name><surname>Samuels</surname> <given-names>D. C.</given-names></name> <name><surname>Guo</surname> <given-names>Y.</given-names></name></person-group> (<year>2015</year>). <article-title>Alternative applications for distinct RNA sequencing strategies</article-title>. <source>Brief. Bioinform</source>. <volume>16</volume>, <fpage>629</fpage>&#x02013;<lpage>639</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbu032</pub-id><pub-id pub-id-type="pmid">25246237</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hardy</surname> <given-names>G. H.</given-names></name> <name><surname>Wright</surname> <given-names>E. M.</given-names></name></person-group> (<year>2008</year>). <source>An Introduction to the Theory of Numbers</source>. <publisher-loc>Oxford, NY</publisher-loc>: <publisher-name>Oxford University Press</publisher-name>.</citation></ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hollander</surname> <given-names>M.</given-names></name> <name><surname>Wolfe</surname> <given-names>D. A.</given-names></name> <name><surname>Chicken</surname> <given-names>E.</given-names></name></person-group> (<year>2013</year>). <source>Nonparametric Statistical Methods</source>. <publisher-loc>Hoboken, NJ</publisher-loc>: <publisher-name>John Wiley &#x00026; Sons</publisher-name>.</citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kantorovich</surname> <given-names>L. V.</given-names></name> <name><surname>Rubinstein</surname> <given-names>G. S.</given-names></name></person-group> (<year>1958</year>). <article-title>On a space of completely additive functions</article-title>. <source>Vestnik Leningrad. Univ</source>. <volume>13</volume>, <fpage>52</fpage>&#x02013;<lpage>59</lpage>.</citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Katkovnik</surname> <given-names>V.</given-names></name> <name><surname>Kgiazarian</surname> <given-names>K.</given-names></name> <name><surname>Astola</surname> <given-names>J.</given-names></name></person-group> (<year>2002</year>). <article-title>Adaptive window size image de-noising based on intersection of confidence intervals (ICI) rule</article-title>. <source>J. Math. Imaging Vis</source>. <volume>16</volume>, <fpage>223</fpage>&#x02013;<lpage>235</lpage>. <pub-id pub-id-type="doi">10.1023/A:1020329726980</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Killick</surname> <given-names>R.</given-names></name> <name><surname>Fearnhead</surname> <given-names>P.</given-names></name> <name><surname>Eckley</surname> <given-names>I. A.</given-names></name></person-group> (<year>2012</year>). <article-title>Optimal detection of changepoints with a linear computational cost</article-title>. <source>J. Am. Stat. Assoc</source>. <volume>107</volume>, <fpage>1590</fpage>&#x02013;<lpage>1598</lpage>. <pub-id pub-id-type="doi">10.1080/01621459.2012.737745</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lavielle</surname> <given-names>M.</given-names></name></person-group> (<year>2005</year>). <article-title>Using penalized contrasts for the change-point problem</article-title>. <source>Signal Process</source> <volume>85</volume>, <fpage>1501</fpage>&#x02013;<lpage>1510</lpage>. <pub-id pub-id-type="doi">10.1016/j.sigpro.2005.01.012</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Levina</surname> <given-names>E.</given-names></name> <name><surname>Bickel</surname> <given-names>P.</given-names></name></person-group> (<year>2001</year>). <article-title>The earth mover&#x00027;s distance is the mallows distance: some insights from statistics</article-title>, in <source>IEEE International Conference on Computer Vision</source> (<publisher-loc>Vancouver, BC</publisher-loc>), <fpage>251</fpage>&#x02013;<lpage>256</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV.2001.937632</pub-id></citation></ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H.</given-names></name></person-group> (<year>2011</year>). <article-title>A statistical framework for SNP calling, mutation discovery, association mapping and population genetical parameter estimation from sequencing data</article-title>. <source>Bioinformatics</source> <volume>27</volume>, <fpage>2987</fpage>&#x02013;<lpage>2993</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btr509</pub-id><pub-id pub-id-type="pmid">21903627</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Handsaker</surname> <given-names>B.</given-names></name> <name><surname>Wysoker</surname> <given-names>A.</given-names></name> <name><surname>Fennell</surname> <given-names>T.</given-names></name> <name><surname>Ruan</surname> <given-names>J.</given-names></name> <name><surname>Homer</surname> <given-names>N.</given-names></name> <etal/></person-group>. (<year>2009</year>). <article-title>The sequence alignment/map format and SAMtools</article-title>. <source>Bioinformatics</source> <volume>25</volume>, <fpage>2078</fpage>&#x02013;<lpage>2079</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp352</pub-id><pub-id pub-id-type="pmid">19505943</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Locallo</surname> <given-names>A.</given-names></name> <name><surname>Prandi</surname> <given-names>D.</given-names></name> <name><surname>Fedrizzi</surname> <given-names>T.</given-names></name> <name><surname>Demichelis</surname> <given-names>F.</given-names></name></person-group> (<year>2019</year>). <article-title>TPES: tumor purity estimation from SNVs</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>4433</fpage>&#x02013;<lpage>4435</lpage> <pub-id pub-id-type="doi">10.1093/bioinformatics/btz406</pub-id><pub-id pub-id-type="pmid">31099386</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Macaulay</surname> <given-names>I. C.</given-names></name> <name><surname>Teng</surname> <given-names>M. J.</given-names></name> <name><surname>Haerty</surname> <given-names>W.</given-names></name> <name><surname>Kumar</surname> <given-names>P.</given-names></name> <name><surname>Ponting</surname> <given-names>C. P.</given-names></name> <name><surname>Voet</surname> <given-names>T.</given-names></name></person-group> (<year>2016</year>). <article-title>Separation and parallel sequencing of the genomes and transcriptomes of single cells using G&#x00026;T-seq</article-title>. <source>Nat. Protoc</source>. <volume>11</volume>, <fpage>2081</fpage>&#x02013;<lpage>2103</lpage>. <pub-id pub-id-type="doi">10.1038/nprot.2016.138</pub-id><pub-id pub-id-type="pmid">27685099</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Miller</surname> <given-names>C. A.</given-names></name> <name><surname>White</surname> <given-names>B. S.</given-names></name> <name><surname>Dees</surname> <given-names>N. D.</given-names></name> <name><surname>Griffith</surname> <given-names>M.</given-names></name> <name><surname>Welch</surname> <given-names>J. S.</given-names></name> <name><surname>Griffith</surname> <given-names>O. L.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>SciClone: inferring clonal architecture and tracking the spatial and temporal patterns of tumor evolution</article-title>. <source>PLOS Comput. Biol</source>. <volume>10</volume>:<fpage>e1003665</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1003665</pub-id><pub-id pub-id-type="pmid">25102416</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Morin</surname> <given-names>R. D.</given-names></name> <name><surname>Mungall</surname> <given-names>K.</given-names></name> <name><surname>Pleasance</surname> <given-names>E.</given-names></name> <name><surname>Mungall</surname> <given-names>A. J.</given-names></name> <name><surname>Goya</surname> <given-names>R.</given-names></name> <name><surname>Huff</surname> <given-names>R. D.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Mutational and structural analysis of diffuse large B-cell lymphoma using whole-genome sequencing</article-title>. <source>Blood</source> <volume>122</volume>, <fpage>1256</fpage>&#x02013;<lpage>1265</lpage>. <pub-id pub-id-type="doi">10.1182/blood-2013-02-483727</pub-id><pub-id pub-id-type="pmid">23699601</pub-id></citation></ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Movassagh</surname> <given-names>M.</given-names></name> <name><surname>Alomran</surname> <given-names>N.</given-names></name> <name><surname>Mudvari</surname> <given-names>P.</given-names></name> <name><surname>Dede</surname> <given-names>M.</given-names></name> <name><surname>Dede</surname> <given-names>C.</given-names></name> <name><surname>Kowsari</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>RNA2DNAlign: nucleotide resolution allele asymmetries through quantitative assessment of RNA and DNA paired sequencing data</article-title>. <source>Nucleic Acids Res</source>. <volume>44</volume>:<fpage>e161</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw757</pub-id><pub-id pub-id-type="pmid">27576531</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oesper</surname> <given-names>L.</given-names></name> <name><surname>Mahmoody</surname> <given-names>A.</given-names></name> <name><surname>Raphael</surname> <given-names>B. J.</given-names></name></person-group> (<year>2013</year>). <article-title>THetA: inferring intra-tumor heterogeneity from high-throughput DNA sequencing data</article-title>. <source>Genome Biol</source>. <volume>14</volume>:<fpage>R80</fpage>. <pub-id pub-id-type="doi">10.1186/gb-2013-14-7-r80</pub-id><pub-id pub-id-type="pmid">23895164</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oesper</surname> <given-names>L.</given-names></name> <name><surname>Satas</surname> <given-names>G.</given-names></name> <name><surname>Raphael</surname> <given-names>B. J.</given-names></name></person-group> (<year>2014</year>). <article-title>Quantifying tumor heterogeneity in whole-genome and whole-exome sequencing data</article-title>. <source>Bioinformatics</source> <volume>30</volume>, <fpage>3532</fpage>&#x02013;<lpage>3540</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btu651</pub-id><pub-id pub-id-type="pmid">25297070</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pag&#x000E8;s</surname> <given-names>F.</given-names></name> <name><surname>Galon</surname> <given-names>J.</given-names></name> <name><surname>Dieu-Nosjean</surname> <given-names>M. C.</given-names></name> <name><surname>Tartour</surname> <given-names>E.</given-names></name> <name><surname>Sautes-Fridman</surname> <given-names>C.</given-names></name> <name><surname>Fridman</surname> <given-names>W. H.</given-names></name></person-group> (<year>2010</year>). <article-title>Immune infiltration in human tumors: a prognostic factor that should not be ignored</article-title>. <source>Oncogene</source> <volume>29</volume>, <fpage>1093</fpage>&#x02013;<lpage>1102</lpage>. <pub-id pub-id-type="doi">10.1038/onc.2009.416</pub-id><pub-id pub-id-type="pmid">21245428</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Reuter</surname> <given-names>J. A.</given-names></name> <name><surname>Spacek</surname> <given-names>D. V.</given-names></name> <name><surname>Pai</surname> <given-names>R. K.</given-names></name> <name><surname>Snyder</surname> <given-names>M. P.</given-names></name></person-group> (<year>2016</year>). <article-title>Simul-seq: combined DNA and RNA sequencing for whole-genome and transcriptome profiling</article-title>. <source>Nat. Methods</source> <volume>13</volume>, <fpage>953</fpage>&#x02013;<lpage>958</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.4028</pub-id><pub-id pub-id-type="pmid">27723755</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Roth</surname> <given-names>A.</given-names></name> <name><surname>Khattra</surname> <given-names>J.</given-names></name> <name><surname>Yap</surname> <given-names>D.</given-names></name> <name><surname>Wan</surname> <given-names>A.</given-names></name> <name><surname>Laks</surname> <given-names>E.</given-names></name> <name><surname>Biele</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>PyClone: statistical inference of clonal population structure in cancer</article-title>. <source>Nat. Methods</source> <volume>11</volume>, <fpage>396</fpage>&#x02013;<lpage>398</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2883</pub-id><pub-id pub-id-type="pmid">24633410</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shah</surname> <given-names>S. P.</given-names></name> <name><surname>Roth</surname> <given-names>A.</given-names></name> <name><surname>Goya</surname> <given-names>R.</given-names></name> <name><surname>Oloumi</surname> <given-names>A.</given-names></name> <name><surname>Ha</surname> <given-names>G.</given-names></name> <name><surname>Zhao</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>The clonal and mutational evolution spectrum of primary triple-negative breast cancers</article-title>. <source>Nature</source> <volume>486</volume>, <fpage>395</fpage>&#x02013;<lpage>399</lpage>. <pub-id pub-id-type="doi">10.1038/nature10933</pub-id><pub-id pub-id-type="pmid">22495314</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shi</surname> <given-names>L.</given-names></name> <name><surname>Guo</surname> <given-names>Y.</given-names></name> <name><surname>Dong</surname> <given-names>C.</given-names></name> <name><surname>Huddleston</surname> <given-names>J.</given-names></name> <name><surname>Yang</surname> <given-names>H.</given-names></name> <name><surname>Han</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Long-read sequencing and <italic>de novo</italic> assembly of a Chinese genome</article-title>. <source>Nat. Commun</source>. <volume>7</volume>:<fpage>12065</fpage>. <pub-id pub-id-type="doi">10.1038/ncomms12065</pub-id><pub-id pub-id-type="pmid">27356984</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shlien</surname> <given-names>A.</given-names></name> <name><surname>Raine</surname> <given-names>K.</given-names></name> <name><surname>Fuligni</surname> <given-names>F.</given-names></name> <name><surname>Arnold</surname> <given-names>R.</given-names></name> <name><surname>Nik-Zainal</surname> <given-names>S.</given-names></name> <name><surname>Dronov</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Direct transcriptional consequences of somatic mutation in breast cancer</article-title>. <source>Cell Rep</source>. <volume>16</volume>, <fpage>2032</fpage>&#x02013;<lpage>2046</lpage>. <pub-id pub-id-type="doi">10.1016/j.celrep.2016.07.028</pub-id><pub-id pub-id-type="pmid">27498871</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>S&#x00142;owi&#x00144;ski</surname> <given-names>P.</given-names></name> <name><surname>Li</surname> <given-names>M.</given-names></name> <name><surname>Restrepo</surname> <given-names>P.</given-names></name> <name><surname>Alomran</surname> <given-names>N.</given-names></name> <name><surname>Spurr</surname> <given-names>L. F.</given-names></name> <name><surname>Miller</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>GeTallele: a mathematical model and a toolbox for integrative analysis and visualization of DNA and RNA allele frequencies</article-title>. <source>bioRxiv [Preprint]</source>. <pub-id pub-id-type="doi">10.1101/491209</pub-id></citation></ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Spurr</surname> <given-names>L. F.</given-names></name> <name><surname>Alomran</surname> <given-names>N.</given-names></name> <name><surname>Bousounis</surname> <given-names>P.</given-names></name> <name><surname>Reece-Stremtan</surname> <given-names>D.</given-names></name> <name><surname>Prashant</surname> <given-names>N. M.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>ReQTL: identifying correlations between expressed SNVs and gene expression using RNA-sequencing data</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>1351</fpage>&#x02013;<lpage>1359</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz750</pub-id><pub-id pub-id-type="pmid">31589315</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thorvaldsd&#x000F3;ttir</surname> <given-names>H.</given-names></name> <name><surname>Robinson</surname> <given-names>J. T.</given-names></name> <name><surname>Mesirov</surname> <given-names>J. P.</given-names></name></person-group> (<year>2013</year>). <article-title>Integrative genomics viewer (IGV): high-performance genomics data visualization and exploration</article-title>. <source>Brief. Bioinform</source>. <volume>14</volume>, <fpage>178</fpage>&#x02013;<lpage>192</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbs017</pub-id><pub-id pub-id-type="pmid">22517427</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Van De Geijn</surname> <given-names>B.</given-names></name> <name><surname>McVicker</surname> <given-names>G.</given-names></name> <name><surname>Gilad</surname> <given-names>Y.</given-names></name> <name><surname>Pritchard</surname> <given-names>J. K.</given-names></name></person-group> (<year>2015</year>). <article-title>WASP: allele-specific software for robust molecular quantitative trait locus discovery</article-title>. <source>Nat. Methods</source> <volume>12</volume>, <fpage>1061</fpage>&#x02013;<lpage>1063</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.3582</pub-id><pub-id pub-id-type="pmid">26366987</pub-id></citation></ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>S.</given-names></name> <name><surname>Mercante</surname> <given-names>D. E.</given-names></name> <name><surname>Zhang</surname> <given-names>K.</given-names></name> <name><surname>Fang</surname> <given-names>Z.</given-names></name></person-group> (<year>2016</year>). <article-title>An integrated approach for RNA-seq data normalization</article-title>. <source>Cancer Inform</source>. <volume>15</volume>, <fpage>129</fpage>&#x02013;<lpage>141</lpage>. <pub-id pub-id-type="doi">10.4137/CIN.S39781</pub-id><pub-id pub-id-type="pmid">27385909</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yoshihara</surname> <given-names>K.</given-names></name> <name><surname>Shahmoradgoli</surname> <given-names>M.</given-names></name> <name><surname>Mart&#x000ED;nez</surname> <given-names>E.</given-names></name> <name><surname>Vegesna</surname> <given-names>R.</given-names></name> <name><surname>Kim</surname> <given-names>H.</given-names></name> <name><surname>Torres-Garcia</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Inferring tumour purity and stromal and immune cell admixture from expression data</article-title>. <source>Nat. Commun</source>. <volume>4</volume>:<fpage>2612</fpage>. <pub-id pub-id-type="doi">10.1038/ncomms3612</pub-id><pub-id pub-id-type="pmid">24113773</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname> <given-names>X.</given-names></name> <name><surname>Zhao</surname> <given-names>Q.</given-names></name> <name><surname>Wu</surname> <given-names>H. J.</given-names></name> <name><surname>Li</surname> <given-names>W.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Meyer</surname> <given-names>C. A.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>MethylPurify: tumor purity deconvolution and differential methylation detection from single tumor DNA methylomes</article-title>. <source>Genome Biol</source>. <volume>15</volume>:<fpage>419</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-014-0419-x</pub-id><pub-id pub-id-type="pmid">25103624</pub-id></citation></ref>
</ref-list>
<glossary>
<def-list>
<title>Abbreviations</title>
<def-item><term>BRCA</term>
<def><p>breast invasive carcinoma</p></def></def-item>
<def-item><term>CDF</term>
<def><p>cumulative distribution function</p></def></def-item>
<def-item><term>CNA</term>
<def><p>copy number alterations</p></def></def-item>
<def-item><term>CNA<sub>DELETION</sub></term>
<def><p>copy number alterations corresponding to deletions (see section Correlation between v<sub>PR</sub> and CNA)</p></def></def-item>
<def-item><term>CNA<sub>AMPLIFICATION</sub></term>
<def><p>copy number alterations corresponding to amplifications (see section Correlation between v<sub>PR</sub> and CNA)</p></def></def-item>
<def-item><term>CPE</term>
<def><p>consensus purity estimate</p></def></def-item>
<def-item><term>DNA</term>
<def><p>genome</p></def></def-item>
<def-item><term>EMD</term>
<def><p>earth mover&#x00027;s distance</p></def></def-item>
<def-item><term>FWER</term>
<def><p>family-wise error rate</p></def></def-item>
<def-item><term>FDR</term>
<def><p>false discovery rate</p></def></def-item>
<def-item><term>MEA</term>
<def><p>mean absolute error</p></def></def-item>
<def-item><term>Nex</term>
<def><p>normal exome</p></def></def-item>
<def-item><term>Ntr</term>
<def><p>normal transcriptome</p></def></def-item>
<def-item><term>r<sub>TEX,CNA,DEL</sub></term>
<def><p>Pearson&#x00027;s correlation coefficient between v<sub>PR,TEX</sub> and CNA<sub>DELETION</sub></p></def></def-item>
<def-item><term>r<sub>TEX,CNA,AMPl</sub></term>
<def><p>Pearson&#x00027;s correlation coefficient between v<sub>PR,TEX</sub> and CNA<sub>AMPLIFICATION</sub></p></def></def-item>
<def-item><term>r<sub>TTR,CNA,DEL</sub></term>
<def><p>Pearson&#x00027;s correlation coefficient between v<sub>PR,TTR</sub> and CNA<sub>DELETION</sub></p></def></def-item>
<def-item><term>r<sub>TTR,CNA,AMPL</sub></term>
<def><p>Pearson&#x00027;s correlation coefficient between v<sub>PR,TTR</sub> and CNA<sub>AMPLIFICATION</sub></p></def></def-item>
<def-item><term>p<sub>FDR</sub></term>
<def><p><italic>p</italic>-value after multiple comparisons Benjamini and Hochberg false discovery rate correction</p></def></def-item>
<def-item><term>PDF</term>
<def><p>probability density function</p></def></def-item>
<def-item><term>QN (e.g., Q50)</term>
<def><p>N-th percentile</p></def></def-item>
<def-item><term>RNA</term>
<def><p>transcriptome</p></def></def-item>
<def-item><term>SNV</term>
<def><p>single-nucleotide variant</p></def></def-item>
<def-item><term>TCGA</term>
<def><p>the cancer genome atlas</p></def></def-item>
<def-item><term>Tex</term>
<def><p>tumor exome</p></def></def-item>
<def-item><term>Ttr</term>
<def><p>tumor transcriptome</p></def></def-item>
<def-item><term>VAF</term>
<def><p>variant allele frequency</p></def></def-item>
<def-item><term>VAF<sub>TEX</sub></term>
<def><p>variant allele frequency in tumor exome sequence</p></def></def-item>
<def-item><term>VAF<sub>TTR</sub></term>
<def><p>variant allele frequency in tumor transcriptome sequence</p></def></def-item>
<def-item><term>VBP</term>
<def><p>v<sub>PR</sub> based purity</p></def></def-item>
<def-item><term>v<sub>PR</sub></term>
<def><p>variant probability</p></def></def-item>
<def-item><term>v<sub>PR,TEX</sub></term>
<def><p>variant probability estimated from tumor exome sequence</p></def></def-item>
<def-item><term>v<sub>PR,TTR</sub></term>
<def><p>variant probability estimated from tumor transcriptome sequence</p></def></def-item>
<def-item><term>WES</term>
<def><p>whole exome sequencing.</p></def></def-item>
</def-list>
</glossary>
<fn-group>
<fn fn-type="financial-disclosure"><p><bold>Funding.</bold> This work was supported by McCormick Genomic and Proteomic Center (MGPC), The George Washington University; [MGPC_PG2018 to AH]. Work of PS was generously supported by the Wellcome Trust Institutional Strategic Support Award [204909/Z/16/Z]. KT-A gratefully acknowledges the financial support of the EPSRC via grant EP/N014391/1.</p>
</fn>
</fn-group>
</back>
</article>