<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article article-type="methods-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">850804</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.850804</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>seGMM: A New Tool for Gender Determination From Massively Parallel Sequencing Data</article-title>
<alt-title alt-title-type="left-running-head">Liu et&#x20;al.</alt-title>
<alt-title alt-title-type="right-running-head">Tool for Gender Determination</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Sihan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1601221/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zeng</surname>
<given-names>Yuanyuan</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1134719/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Chao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Qian</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1589790/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Meilin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Xiaolu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Lanchen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Lu</surname>
<given-names>Yu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Guo</surname>
<given-names>Hui</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/861770/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Bu</surname>
<given-names>Fengxiao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1258614/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Institute of Rare Diseases</institution>, <institution>West China Hospital of Sichuan University</institution>, <addr-line>Chengdu</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Medicine</institution>, <institution>National Institute for Data Science in Health and Medicine</institution>, <institution>Xiamen University</institution>, <addr-line>Xiamen</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Center for Medical Genetics and Hunan Provincial Key Laboratory of Medical Genetics</institution>, <institution>School of Life Sciences</institution>, <institution>Central South University</institution>, <addr-line>Changsha</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/552766/overview">Tao Huang</ext-link>, Shanghai Institute of Nutrition and Health (CAS), China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/541194/overview">Ram Vinay Pandey</ext-link>, Karolinska University Hospital, Sweden</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1251218/overview">Luca Ermini</ext-link>, Luxembourg Institute of Health, Luxembourg</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1633670/overview">Uday Shanker Evani</ext-link>, New York Genome Center, United&#x20;States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1376611/overview">Ying Cai</ext-link>, Albert Einstein College of Medicine, United&#x20;States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Yu Lu, <email>samuelluyu@163.com</email>; Hui Guo, <email>guohui@sklmg.edu.cn</email>; Fengxiao Bu, <email>bufengxiao@wchscu.cn</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Computational Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>03</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>850804</elocation-id>
<history>
<date date-type="received">
<day>08</day>
<month>01</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>10</day>
<month>02</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Liu, Zeng, Wang, Zhang, Chen, Wang, Wang, Lu, Guo and Bu.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Liu, Zeng, Wang, Zhang, Chen, Wang, Wang, Lu, Guo and Bu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these&#x20;terms.</p>
</license>
</permissions>
<abstract>
<p>In clinical genetic testing, checking the concordance between self-reported gender and genotype-inferred gender from genomic data is a significant quality control measure because mismatched gender due to sex chromosomal abnormalities or misregistration of clinical information can significantly affect molecular diagnosis and treatment decisions. Targeted gene sequencing (TGS) is widely recommended as a first-tier diagnostic step in clinical genetic testing. However, the existing gender-inference tools are optimized for whole genome and whole exome data and are not adequate and accurate for analyzing TGS data. In this study, we validated a new gender-inference tool, seGMM, which uses unsupervised clustering (Gaussian mixture model) to determine the gender of a sample. The seGMM tool can also identify sex chromosomal abnormalities in samples by aligning the sequencing reads from the genotype data. The seGMM tool consistently demonstrated&#x2009;&#x3e;99% gender-inference accuracy in a publicly available 1,000-gene panel dataset from the 1,000 Genomes project, an in-house 785 hearing loss gene panel dataset of 16,387 samples, and a 187 autism risk gene panel dataset from the Autism Clinical and Genetic Resources in China (ACGC) database. The performance and accuracy of seGMM was significantly higher for the targeted gene sequencing (TGS), whole exome sequencing (WES), and whole genome sequencing (WGS) datasets compared to the other existing gender-inference tools such as PLINK, seXY, and XYalign. The results of seGMM were confirmed by the short tandem repeat analysis of the sex chromosome marker gene, amelogenin. Furthermore, our data showed that seGMM accurately identified sex chromosomal abnormalities in the samples. In conclusion, the seGMM tool shows great potential in clinical genetics by determining the sex chromosomal karyotypes of samples from massively parallel sequencing data with high accuracy.</p>
</abstract>
<kwd-group>
<kwd>massively parallel sequencing data</kwd>
<kwd>Gaussian mixture model</kwd>
<kwd>gender</kwd>
<kwd>sex chromosomal abnormality</kwd>
<kwd>aneuploidy</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>The next-generation sequencing (NGS) technology has revolutionized human biology and medicine in the last decade. NGS is routinely used in clinical genetic testing for molecular diagnosis of hereditary disorders, infectious diseases, and immune disorders, non-invasive prenatal genetic testing, and personalized precision medicine, especially for cancer patients (<xref ref-type="bibr" rid="B21">Phillips and Douglas, 2018</xref>; <xref ref-type="bibr" rid="B22">Phillips et&#x20;al., 2020</xref>). Clinical genetic testing is a diagnostic tool that involves genome sequencing to identify pathogenic gene mutations (genetic variants) in human diseases (<xref ref-type="bibr" rid="B17">McPherson, 2006</xref>). This may involve targeted gene sequencing (TGS) of single or multiple genes, whole exome sequencing (WES), or whole genome sequencing (WGS) (<xref ref-type="bibr" rid="B5">Di Resta et&#x20;al., 2018</xref>). TGS is highly accurate, robust, and cost-effective. Therefore, TGS has been used for the diagnosis of several human diseases including hearing loss, vision loss, cardiovascular disorders, neurologic disorders, cancer risk, and renal disorders (<xref ref-type="bibr" rid="B14">Lin et&#x20;al., 2012</xref>; <xref ref-type="bibr" rid="B26">Saudi Mendeliome, 2015</xref>).</p>
<p>Parallelized TGS analysis of large patient cohorts requires rigorous quality control (QC) and preprocessing to identify the pathogenic gene variants (<xref ref-type="bibr" rid="B11">Lee et&#x20;al., 2017</xref>). Verification of the concordance between self-reported gender and genetically inferred gender is an essential QC step because misregistration of clinical information, sample swaps, sample pollution, or sex chromosomal abnormalities can result in wrong conclusions and affect treatment decisions (<xref ref-type="bibr" rid="B31">Taylor et&#x20;al., 2015</xref>; <xref ref-type="bibr" rid="B34">Webster et&#x20;al., 2019</xref>). Sex chromosomal abnormalities are reported in approximately 1 in 448 newborn children (<xref ref-type="bibr" rid="B18">Nielsen and Wohlert, 1990</xref>). Therefore, there is a higher probability of gender inconsistencies in larger cohorts. Cytogenetic karyotyping is the gold standard method for confirming the gender of an individual and identifying chromosomal abnormalities. The highly conserved sex chromosomal marker gene, amelogenin, is widely used for identifying gender using short tandem repeat (STR) typing (<xref ref-type="bibr" rid="B32">Thangaraj et&#x20;al., 2002</xref>; <xref ref-type="bibr" rid="B15">Ma et&#x20;al., 2012</xref>). The 6&#xa0;bp deletion within intron 1 of the amelogenin gene in the X chromosome is used to distinguish the PCR amplified products of the amelogenin gene in the X and Y chromosomes (<xref ref-type="bibr" rid="B29">Sullivan et&#x20;al., 1993</xref>). However, these methods are time- and labor-consuming.</p>
<p>Several computational tools such as PLINK, seXY, and XYalign, have been developed for gender inference based on genome-wide WES or WGS data. PLINK inferred gender by calculating F coefficients from the genotyping array data using X chromosome homozygosity/heterozygosity rates; samples with F coefficient values of more than 0.8 were designated as males and samples with F coefficient values of less than 0.2 were considered as females (<xref ref-type="bibr" rid="B23">Purcell et&#x20;al., 2007</xref>). The seXY tool is based on the logistic regression model and identifies gender by considering X chromosome heterozygosity and Y chromosome missingness in the genotyping array data (<xref ref-type="bibr" rid="B24">Qian et&#x20;al., 2017</xref>). XYalign tool identifies gender from both WES and WGS datasets by extracting the read counts mapped to the sex chromosomes and calculating the ratio of X and Y read counts in a scatter plot (<xref ref-type="bibr" rid="B34">Webster et&#x20;al., 2019</xref>). However, none of these tools are optimized for analyzing TGS panel data, which contains significantly reduced information compared to the whole genomic or exome data. As shown in <xref ref-type="table" rid="T1">Table&#x20;1</xref>, the performance of the existing tools was not satisfactory in reporting gender using the TGS data. Furthermore, sex chromosomal abnormalities were not clearly identified by the PLINK, seXY, and XYalign tools. Few studies reported the sex chromosomal abnormalities of individuals based on the ratio of sequencing reads that were mapped to the X and Y chromosomes from the genotyping array and WGS data (<xref ref-type="bibr" rid="B2">Bycroft et&#x20;al., 2018</xref>; <xref ref-type="bibr" rid="B33">Turro et&#x20;al., 2020</xref>). However, this methodology has not been automated. Therefore, there is an urgent need to construct highly accurate bioinformatics tools for gender inference from TGS data and reporting sex chromosomal abnormalities.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Gender prediction accuracy of different methods for samples in dataset 1.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Tools</th>
<th align="center">Accuracy for all samples (%)</th>
<th align="center">Accuracy for male samples (%)</th>
<th align="center">Accuracy for female samples (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">PLINK</td>
<td align="char" char=".">81.44</td>
<td align="char" char=".">48.28</td>
<td align="char" char=".">100</td>
</tr>
<tr>
<td align="left">seXY</td>
<td align="char" char=".">62.5</td>
<td align="char" char=".">45.45</td>
<td align="char" char=".">81.63</td>
</tr>
<tr>
<td align="left">XYalign</td>
<td align="char" char=".">98.08</td>
<td align="char" char=".">100</td>
<td align="char" char=".">95.92</td>
</tr>
<tr>
<td align="left">seGMM</td>
<td align="char" char=".">99.52</td>
<td align="char" char=".">100</td>
<td align="char" char=".">98.98</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In this study, we verified the performance and accuracy of the new gender inference tool, seGMM, using both in-house and publicly available TGS, WES, and WGS datasets. The seGMM tool used unsupervised learning to integrate the information of the X and Y chromosomes from the TGS, WES, or WGS datasets and classified the samples into one of the six sex chromosomal karyotypes (XX, XY, XYY, XXY, XXX, and&#x20;X).</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and Methods</title>
<sec id="s2-1">
<title>Data</title>
<p>We compared the performances of three existing gender-inferring methods and seGMM using the TGS data from the following 3 datasets: 1) Dataset 1: exon-targeted sequencing data of 1,000 genes (34 X chromosomal genes and two Y chromosomal genes) for a cohort of 110 males and 98 females from the 1,000 Genomes Project (<xref ref-type="sec" rid="s11">Supplementary Table S1</xref>) (<xref ref-type="bibr" rid="B7">Genomes Project et&#x20;al., 2010</xref>); 2) Dataset 2 (in-house): massive parallel sequencing of 785&#x20;deafness-related genes (eight genes in the X chromosome) for an in-house cohort of 8,805 males and 7,582 females; and 3) Dataset 3: targeted sequencing data of 187 autism risk genes (13 genes in the X chromosome) for a cohort of 42 females and 205 males from the Autism Clinical and Genetic Resources in China (ACGC) (<xref ref-type="bibr" rid="B10">Guo et&#x20;al., 2018</xref>).</p>
<p>We also used the following two publicly available datasets (<xref ref-type="sec" rid="s11">Supplementary Tables S2, S3</xref>) and one in-house dataset for analyzing the performance of seGMM in determining gender using WES and WGS data: 1) Dataset 4: exome sequencing data of 164 males and 118 females from the 1,000 Genomes Project (<xref ref-type="bibr" rid="B8">Genomes Project et&#x20;al., 2015</xref>); 2) Dataset 5 (in-house): exome sequencing data of 1,257 males and 1,136 females; and 3) Dataset 6: high-coverage whole genome sequencing data of 11 males and 16 females from the 1,000 Genomes Project (<xref ref-type="bibr" rid="B8">Genomes Project et&#x20;al., 2015</xref>).</p>
<p>The publicly available BAM files were previously mapped to the reference genome (GRCh37) and directly used for downstream analyses. For the in-house datasets, Fastp was used to remove the adapters and low-quality reads, and the quality of sequencing data was evaluated using measures such as Q20, sequence duplication levels, coverage, and GC content (<xref ref-type="bibr" rid="B3">Chen et&#x20;al., 2018</xref>). Clean DNA sequencing reads were mapped to the human reference genome (GRCh37) using the BWA-MEM algorithm (<xref ref-type="bibr" rid="B12">Li and Durbin, 2009</xref>). Duplicated reads in the BAM files from the public and in-house datasets were removed using the sambamba tool (<xref ref-type="bibr" rid="B30">Tarasov et&#x20;al., 2015</xref>). The variants were identified based on the Genome Analysis Toolkit best practices recommendations (<xref ref-type="bibr" rid="B16">McKenna et&#x20;al., 2010</xref>) and filtered with VCFtools (<xref ref-type="bibr" rid="B4">Danecek et&#x20;al., 2011</xref>) using parameters such as missing data in more than 50% of samples, minor allele count &#x3c;3, overall SNP quality (QUAL) score &#x3c;30, and read depth &#x3c;5.</p>
</sec>
<sec id="s2-2">
<title>Gender Inference Using seGMM</title>
<p>The model for seGMM included five gender-associated features, namely, X chromosome heterozygosity (XH), reads mapped to the X chromosome (Xmap), reads mapped to the Y chromosome (Ymap), the ratio of X/Y counts (XYratio), and the mean depth of the sex-determining region of the Y chromosome (<italic>SRY</italic>) gene (SRY_dep). The seGMM tool computed XH as the fraction of all genotypes on the X chromosome with two different allele calls, excluding the missing genotypes. Xmap/Ymap was computed as the fraction of high-quality reads (mapq &#x3e; 30) that mapped to the X/Y chromosome divided by the total number of high-quality reads that mapped to the genome using the samtools algorithm (<xref ref-type="bibr" rid="B13">Li et&#x20;al., 2009</xref>). XYratio was computed as the ratio of Xmap to Ymap (Xmap/Ymap). SRY_dep was determined using the mosdepth tool (<xref ref-type="bibr" rid="B20">Pedersen and Quinlan, 2018</xref>). The seGMM tool allows the users to customize feature selection for the GMM model because different TGS panel designs may only provide some features. For example, if the gene panel contains only genes located on the X chromosome, the relevant features on the X chromosome (XH and Xmap) are extracted and put into the model for gender determination.</p>
<p>The features extracted from the BAM and VCF files were normalized to the same level using the scale function in R 4.1.2 (<xref ref-type="bibr" rid="B25">R Core Team., 2021</xref>). Then, the mclust (v.5.4.9) R package was used to perform model-based clustering with the expectation-maximization (EM) algorithm and the samples were classified into two clusters (<xref ref-type="bibr" rid="B27">Scrucca et&#x20;al., 2016</xref>). The gender was inferred based on the cluster results for a group of samples. The outliers were identified when uncertainty (probability of being assigned to two different clusters) was greater than 0.1. When a single sample was submitted, gender was inferred using the reference data that was analyzed with the same features as those in the seGMM model (<xref ref-type="fig" rid="F1">Figure&#x20;1</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Schematic diagram of seGMM. The seGMM tool automatically collects features from the input VCF and BAM files and builds the GMM model. The output of seGMM includes gender prediction results and identification of samples with abnormal sex chromosomes.</p>
</caption>
<graphic xlink:href="fgene-13-850804-g001.tif"/>
</fig>
</sec>
<sec id="s2-3">
<title>Identifying Potential Sex Chromosomal Abnormalities in the Sequenced Samples</title>
<p>We defined the gates to classify individual karyotypes. The distribution of Xmap and Ymap in the females and males of the large cohort was normal. The ratio of samples with sex chromosomal abnormalities was 0.022% (<xref ref-type="bibr" rid="B18">Nielsen and Wohlert, 1990</xref>). This data was in agreement with the empirical rule, which states that 99.7% of normally distributed data lies within 3 standard deviations (sd) of the mean. Hence, we defined the normal gates as mean&#xb1;3sd. The fold changes in Xmap or Ymap values indicated sex chromosomal aneuploidy.</p>
<p>To identify sex chromosomal abnormalities in the samples, we first calculated the mean value and standard deviation values of Xmap (mean_xmap and sd_xmap) and Ymap (mean_ymap, and sd_ymap) in the genetically determined male and female samples. The values for the males and females were denoted as m and f, respectively. The following six gates were then used to classify the karyotypes of individuals:<list list-type="simple">
<list-item>
<label>&#x2022;</label>
<p>XY&#x20;Gate</p>
<list list-type="simple">
<list-item>
<p>&#x25cb; mean_xmap_m - 3 sd_xmap_m &#x3c; x &#x3c; mean_xmap_m &#x2b; 3 sd_xmap_m</p>
</list-item>
<list-item>
<p>&#x25cb; mean_ymap_m - 3 sd_ymap_m &#x3c; y &#x3c; mean_ymap_m &#x2b; 3 sd_ymap_m</p>
</list-item>
</list>
</list-item>
<list-item>
<label>&#x2022;</label>
<p>XYY gate:</p>
<list list-type="simple">
<list-item>
<p>&#x25cb; mean_xmap_m - 3 sd_xmap_m &#x3c; x &#x3c; mean_xmap_m &#x2b; 3 sd_xmap_m</p>
</list-item>
<list-item>
<p>&#x25cb; y &#x3e; 2 mean_ymap_m</p>
</list-item>
</list>
</list-item>
<list-item>
<label>&#x2022;</label>
<p>XX gate:</p>
<list list-type="simple">
<list-item>
<p>&#x25cb; mean_xmap_f - 3 sd_xmap_f &#x3c; x &#x3c; mean_xmap_f &#x2b; 3 sd_xmap_f</p>
</list-item>
<list-item>
<p>&#x25cb; mean_ymap_f - 3 sd_ymap_f &#x3c; y &#x3c; mean_ymap_f &#x2b; 3 sd_ymap_f</p>
</list-item>
</list>
</list-item>
<list-item>
<label>&#x2022;</label>
<p>XXY gate:</p>
<list list-type="simple">
<list-item>
<p>&#x25cb; x &#x3e; 2 mean_xmap_f</p>
</list-item>
<list-item>
<p>&#x25cb; mean_ymap_m - 3 sd_ymap_m &#x3c; y &#x3c; mean_ymap_m &#x2b; 3 sd_ymap_m</p>
</list-item>
</list>
</list-item>
<list-item>
<label>&#x2022;</label>
<p>XXX gate:</p>
<list list-type="simple">
<list-item>
<p>&#x25cb; x &#x3e; 3 mean_xmap_f</p>
</list-item>
<list-item>
<p>&#x25cb; mean ymap_f - 3 sd_ymap_f &#x3c; y &#x3c; mean_ymap_f &#x2b; 3 sd_ymap_f</p>
</list-item>
</list>
</list-item>
<list-item>
<label>&#x2022;</label>
<p>X gate:</p>
<list list-type="simple">
<list-item>
<p>&#x25cb; x &#x3c; 0.5 mean_xmap_f</p>
</list-item>
<list-item>
<p>&#x25cb; mean_ymap_f - 3 sd_ymap_f &#x3c; y &#x3c; mean_ymap_f &#x2b; 3 sd_ymap_f</p>
</list-item>
</list>
</list-item>
</list>
</p>
</sec>
<sec id="s2-4">
<title>Comparing the Performance of seGMM With Other Existing Gender-Inference Methods</title>
<p>The performance of seGMM was compared to PLINK 1.9, XYalign (v.1.1.6), and seXY (v.20170316). For PLINK 1.9, the pseudoautosomal region of the X chromosome was first split off with the parameter--split-x. Then, the parameter--check-sex was run without parameters. After reviewing the distribution of F estimates, the parameter--check-sex was rerun with parameters corresponding to the empirical gap. XYalign was performed following the method described in the original literature. The CHROM_STATS module was used to obtain the depths of chromosomes 1, X, and Y. The depths of X and Y chromosomes were normalized relative to the depth of chromosome 1. Then, a scatter plot of normalized X and Y chromosomes depth was plotted to assess gender in samples. Gender of the samples was inferred with seXY using the X.ped and Y.ped data that was derived from PLINK. The training dataset was provided by seXY. We expected to compare seGMM and other existing tools for all the six datasets. However, target gene panel data for datasets 2 and 3 did not contain genes on the Y chromosome. Therefore, the performance of XYalign and seXY was not available for these two datasets.</p>
</sec>
<sec id="s2-5">
<title>STR Analysis for Verifying Gender</title>
<p>The STR analysis was performed using the customized multiplex PowerPlex&#xae; 16 System, which allowed co-amplification and four-color detection of amelogenin and other gene loci. The following primers were used for amplifying amelogenin: forward, 5&#x2032;- GTT&#x200b;AG&#x200b;ACG&#x200b;TGT&#x200b;GCT&#x200b;TCA&#x200b;ACT&#x200b;TCA&#x200b;GCT&#x200b;ATG&#x200b;AGG&#x200b;TAA&#x200b;TTT&#x200b;TTC&#x2014;3&#x2032;; reverse, 5&#x2032;- ATC&#x200b;CGA&#x200b;CGG&#x200b;TAG&#x200b;TGT&#x200b;CCA&#x200b;ACC&#x200b;ATC&#x200b;AGA&#x200b;GCT&#x200b;TAA&#x200b;ACT&#x200b;GG-3&#x2032;. All genetic loci were amplified simultaneously in a single tube and analyzed in a single lane. One of the primers for the amelogenin gene was labeled with carboxyrhodamine (ROX). The amplicons were separated in the ABI 3730XL Genetic Analyzer and the data was extracted using GeneMapper ID v3.2. The gender was inferred according to the peaks for the amelogenin gene. If only one peak was observed for the amelogenin locus, the gender was designated as female. If two distinct peaks differing by 6&#xa0;bp were observed in the amelogenin locus, the gender was designated as&#x20;male.</p>
</sec>
<sec id="s2-6">
<title>Quantitative Determination of Y Chromosome Copy Number</title>
<p>Genomic DNA (gDNA) was extracted using the MagMAX High Purity Free DNA Separation Kit (Magen, China). DNA concentration of the samples was measured using the NanoDrop One spectrophotometer (Thermo Fisher Scientific, United&#x20;States). The working concentration of all DNA samples was 20&#xa0;ng/&#xb5;l. The primers targeting <italic>SRY</italic>, zinc finger protein Y-linked (<italic>ZFY</italic>), and deleted in azoospermia 1 (<italic>DAZ1</italic>) genes were designed using the Primer-BLAST online tool to determine the Y chromosome copy number (<xref ref-type="bibr" rid="B35">Ye et&#x20;al., 2012</xref>). <italic>RPP30</italic> was used as the internal control. All the primers used in this study are listed in <xref ref-type="sec" rid="s11">Supplementary Table S4</xref>. The qPCR reaction mix included 0.6&#xa0;&#xb5;l of gDNA, 0.4&#xa0;&#xb5;l of each primer, 5&#xa0;&#xb5;l of iTaq&#x2122; Universal SYBR&#xae; Green Supermix (Bio&#x2013;Rad, United&#x20;States), and 3.6&#xa0;&#xb5;l of double-distilled water. Each sample was analyzed with three replicates. The quantitative real-time PCR assay (RT&#x2013;qPCR) was performed in the QuantStudio 5&#x20;Real-Time PCR system (Thermo Fisher Scientific) using the following conditions: initial hot start cycle at 98&#xb0;C for 2&#xa0;min followed by 40 cycles consisting of denaturation at 98&#xb0;C for 10&#xa0;s, annealing at 60&#xb0;C for 10&#xa0;s, and the final extension step of 30&#xa0;s at 72&#xb0;C.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec id="s3-1">
<title>The seGMM Tool Shows Better Performance Compared to Other Tools for the TGS Data</title>
<p>The distribution of XH, Xmap, Ymap, and XYratio values for dataset 1 (<italic>n</italic>&#x20;&#x3d; 208) shown in <xref ref-type="fig" rid="F2">Figures 2A&#x2013;D</xref>. The accuracy of seGMM was 99.52% and none of the samples were outliers (<xref ref-type="fig" rid="F2">Figure&#x20;2E</xref>; <xref ref-type="table" rid="T1">Table&#x20;1</xref>). The accuracy of seGMM in females and males was 98.98 and 100%, respectively. The XYratio of one female sample (NA19054) resembled that of males and was incorrectly classified as male by seGMM. The gender-inference performance of seGMM for dataset 1 was superior to PLINK, seXY and XYalign. The PLINK tool analysis showed that the F coefficients for the dataset1 samples ranged from 0 to 0.9 and gap of F coefficients was not observed (<xref ref-type="sec" rid="s11">Supplementary Figure S1</xref>). The accuracy of PLINK was 81.44% by running --check-sex without parameters. The accuracy of seXY for the dataset 1 was only 62.5% (<xref ref-type="table" rid="T1">Table&#x20;1</xref>). XYalign does not directly indicate predicted gender. Therefore, plotting the normalized sequence depth of the sex chromosomes and cluster samples along two ellipses using the stat_ellipse function resulted in a confidence level of 99.99%. XYalign plot showed that one female sample was located along with the male samples and three female samples were located between the two ellipses. Hence, the predicted gender of these four female samples was ambiguous (<xref ref-type="sec" rid="s11">Supplementary Figure S2</xref>) and the accuracy of XYalign is 98.08%.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The performance of seGMM in the TGS datasets. <bold>(A&#x2013;D)</bold> Distribution of features collected from dataset 1. <bold>(E&#x2013;G)</bold> Sample classification results of datasets 1, 2, and 3 based on seGMM. The colors represent different sample clusters. Dir1 and Dir2 represent the eigenvectors that specify the discriminant subspace generated from the features included in the GMM&#x20;model.</p>
</caption>
<graphic xlink:href="fgene-13-850804-g002.tif"/>
</fig>
<p>The performance of seGMM for the target gene panel data was validated using dataset 2 (<italic>n</italic>&#x20;&#x3d; 16,387) and dataset 3 (<italic>n</italic>&#x20;&#x3d; 247). The read counts, base quality, and GC distribution of the sequencing data of all the 16,387 subjects in dataset 2 was assessed. The average total number of sequence read per sample was 10.72&#xa0;million. The average quality score for all bases was above 30 and the average GC content was 50.11% per subject. The average targeted sequence coverage was 90.43%, and unique mapping rate of each sample was 99.26%. We identified 16,988 variants in eight genes located on the X chromosome. Because the target gene panel for dataset 2 did not contain genes on the Y chromosome, we only used XH and Xmap to analyze the performance of the seGMM model. XH and Xmap plots showed distinct clusters for males and females (<xref ref-type="sec" rid="s11">Supplementary Figure S3</xref>). The overall accuracy of seGMM and PLINK was 99.92 and 87.10%, respectively (<xref ref-type="fig" rid="F2">Figure&#x20;2F</xref>; <xref ref-type="table" rid="T2">Table&#x20;2</xref>). The accuracy of seGMM in females and males was both 99.98%. One self-reported female sample (HL-001200) and two self-reported male samples (CTRL-002692 and CTRL-002753) were misclassified. Therefore, we performed STR analysis with the sex chromosome marker gene, amelogenin, to verify the gender of these three ambiguous samples. All these&#x20;three samples were identified as males because two&#x20;distinct peaks were observed with a difference of 6&#x20;bp for the amelogenin gene (<xref ref-type="fig" rid="F3">Figures 3A&#x2013;C</xref>). CTRL-002692 and CTRL-002753 were misclassified because all other male samples&#x20;had a XH value of 0, while these two samples had a non-zero value (XH<sub>CTRL-002692</sub> &#x3d; 0.0025; XH<sub>CTRL-002753</sub> &#x3d; 0.0046), which could be caused by the individual variation in targeting region.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Gender prediction accuracy of different methods for samples in datasets 2 and 3.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Methods</th>
<th align="center">Dataset 2</th>
<th align="center">Dataset 3</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">PLINK</td>
<td align="char" char=".">87.10</td>
<td align="char" char=".">38.87</td>
</tr>
<tr>
<td align="left">seGMM</td>
<td align="char" char=".">99.92</td>
<td align="char" char=".">92.31</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Experimentally verified gender of HL-001200 <bold>(A)</bold>, CTRL-002692 <bold>(B)</bold> and CTRL-002753 <bold>(C)</bold>. The green box shows the location of the amelogenin loci.</p>
</caption>
<graphic xlink:href="fgene-13-850804-g003.tif"/>
</fig>
<p>The overall accuracy of seGMM for dataset 3 was 92.31% (97.56% for females and 84.88% for males, <xref ref-type="fig" rid="F2">Figure&#x20;2G</xref> and <xref ref-type="table" rid="T2">Table&#x20;2</xref>). The accuracy of PLINK was only 38.87% for dataset 3. The performances of seGMM and PLINK were significantly better for datasets 1 and 2 compared to dataset 3 because the number of X chromosome SNPs (81) were lower and sequencing data for the Y chromosome was absent in dataset 3, thereby affecting the distribution of XH values from the male and female samples (<xref ref-type="sec" rid="s11">Supplementary Figure S4A</xref>). In contrast to PLINK, seGMM collected additional information for the reads mapped to the X chromosome, thereby enabling better separation between the female and male samples (<xref ref-type="sec" rid="s11">Supplementary Figure S4B</xref>). Furthermore, we assessed the performance of seGMM using features only&#x20;extracted from the X chromosome in dataset 1. The seGMM tool showed that 59 samples were outliers and the accuracy for the remaining samples was only 84.56%. We then&#x20;evaluated the&#x20;computation time of different methods using 1 core, 10 cores and 20 cores on a server with 64 Intel(R) Xeon(R) CPU E7-8895 v3 at 2.60&#xa0;GHz. The analysis time for the seGMM tool was longer than PLINK and seXY because it collected additional features such as reads mapped to&#x20;the X and Y chromosomes. Moreover, the analysis time for&#x20;seGMM with 1 core was longer than XYalign and 10&#x20;times&#x20;faster than XYalign with 20 cores (<xref ref-type="sec" rid="s11">Supplementary Table&#x20;S5</xref>).</p>
</sec>
<sec id="s3-2">
<title>The seGMM Tool Shows Better Accuracy Than Other Known Tools for the WES and WGS Data</title>
<p>We then evaluated the performance of the seGMM tool for the WES and WGS data. First, we analyzed the publicly available WES data (dataset 4). The accuracy of seGMM was 100% for the samples in dataset 4 (<italic>n</italic>&#x20;&#x3d; 282) (<xref ref-type="table" rid="T3">Table&#x20;3</xref> and <xref ref-type="sec" rid="s11">Supplementary Figure S5</xref>). The accuracy of PLINK and seXY was also 100%. The accuracy of XYalign was 99.65% (<xref ref-type="sec" rid="s11">Supplementary Figure&#x20;S6</xref>).</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Gender prediction accuracy of different methods for the WES and WGS datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Datasets</th>
<th align="center">PLINK</th>
<th align="center">XYalign</th>
<th align="center">seXY</th>
<th align="center">seGMM</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1000G phase3 WES data</td>
<td align="char" char=".">100</td>
<td align="char" char=".">99.65</td>
<td align="char" char=".">100</td>
<td align="char" char=".">100</td>
</tr>
<tr>
<td align="left">1000G phase3 high quality WGS data</td>
<td align="char" char=".">100</td>
<td align="char" char=".">100</td>
<td align="char" char=".">100</td>
<td align="char" char=".">100</td>
</tr>
<tr>
<td align="left">In-house WES data</td>
<td align="char" char=".">99.79</td>
<td align="char" char=".">99.91</td>
<td align="char" char=".">49.23</td>
<td align="char" char=".">100</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Next, we analyzed the in-house WES data (dataset 5, <italic>n</italic>&#x20;&#x3d; 2,393) using seGMM and other tools. In dataset 5, the average number of sequencing reads per sample was 114.43&#xa0;million. The average Q20, Q30 and GC content of the reads per subject was 97.36%, 93.21%, and 51.23%, respectively. Furthermore, the average unique mapping rate for each individual sample was 99.92%. We identified 89,273 variants on the X chromosome and 4,866 variants on the Y chromosome. The concordance between inferred gender and self-reported gender using the seGMM tool based on the five features for the in-house WES dataset 5 was 99.75% (99.76% for males and 99.74% for females, <xref ref-type="fig" rid="F4">Figure&#x20;4A</xref>, <xref ref-type="sec" rid="s11">Supplementary Figure S7</xref>). Six mismatched samples (HL-005584, HL-006009, HL-006904, HL-007335, HL-007935 and HL-012246) were identified by comparing SNP-inferred gender and self-reported gender. This indicated misregistration of clinical information for some samples. Therefore, we performed STR analysis to validate the gender of these six samples. Three samples were classified as females because they showed only one peak for the amelogenin locus, whereas the remaining three samples&#x20;showed two distinct peaks with a difference of 6&#xa0;bp and were classified as males (<xref ref-type="table" rid="T4">Table&#x20;4</xref>). The results demonstrated that the actual accuracy of seGMM prediction was 100%. We also evaluated the correlation between age and reads mapped to the Y chromosome in the male samples (<xref ref-type="sec" rid="s11">Supplementary Figure S8A</xref>) and reads mapped to the X chromosome in the&#x20;female samples from the in-house WES dataset 5 (<xref ref-type="sec" rid="s11">Supplementary Figure S8B</xref>). The results showed significant negative correlation (<italic>p</italic>&#x20;&#x3d; 6.899e-09; correlation coefficient: &#x2212;0.17) between reads mapped to the Y chromosome and age, thereby indicating loss of Y chromosome during&#x20;aging.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>The prediction accuracy of seGMM in inferring the gender of samples from the in-house WES dataset. <bold>(A)</bold> Sample clustering results of seGMM. The colors represent different sample clusters. Dir1 and Dir2 represent eigenvectors that specify the discriminant subspace generated from the features included in the GMM model. <bold>(B)</bold> Scatter plot shows the reads mapped to the X and Y chromosomes. As shown, we identified three samples (HL-029620, HL-009382 and HL-019110) with XYY sex chromosome karyotypes.</p>
</caption>
<graphic xlink:href="fgene-13-850804-g004.tif"/>
</fig>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Experimental verification of gender prediction results for samples in the in-house WES&#x20;data.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Sample ID</th>
<th align="center">Size of PCR products in the amelogenin loci (bp)</th>
<th align="center">Self-reported gender</th>
<th align="center">seGMM inferred gender</th>
<th align="center">Experimentally validated gender</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="left">HL-005584</td>
<td align="char" char=".">209.15</td>
<td rowspan="2" align="left">Male</td>
<td rowspan="2" align="left">Female</td>
<td rowspan="2" align="left">Female</td>
</tr>
<tr>
<td align="center">-</td>
</tr>
<tr>
<td rowspan="2" align="left">HL-006009</td>
<td align="char" char=".">209.06</td>
<td rowspan="2" align="left">Male</td>
<td rowspan="2" align="left">Female</td>
<td rowspan="2" align="left">Female</td>
</tr>
<tr>
<td align="center">-</td>
</tr>
<tr>
<td rowspan="2" align="left">HL-006904</td>
<td align="char" char=".">209.04</td>
<td rowspan="2" align="left">Female</td>
<td rowspan="2" align="left">Male</td>
<td rowspan="2" align="left">Male</td>
</tr>
<tr>
<td align="char" char=".">214.8</td>
</tr>
<tr>
<td rowspan="2" align="left">HL-007335</td>
<td align="char" char=".">209.06</td>
<td rowspan="2" align="left">Female</td>
<td rowspan="2" align="left">Male</td>
<td rowspan="2" align="left">Male</td>
</tr>
<tr>
<td align="char" char=".">214.85</td>
</tr>
<tr>
<td rowspan="2" align="left">HL-007935</td>
<td align="char" char=".">209.11</td>
<td rowspan="2" align="left">Male</td>
<td rowspan="2" align="left">Female</td>
<td rowspan="2" align="left">Female</td>
</tr>
<tr>
<td align="center">-</td>
</tr>
<tr>
<td rowspan="2" align="left">HL-012246</td>
<td align="char" char=".">209.18</td>
<td rowspan="2" align="left">Female</td>
<td rowspan="2" align="left">Male</td>
<td rowspan="2" align="left">Male</td>
</tr>
<tr>
<td align="char" char=".">214.92</td>
</tr>
<tr>
<td rowspan="2" align="left">HL-033182</td>
<td align="char" char=".">209.02</td>
<td rowspan="2" align="left">Female</td>
<td rowspan="2" align="left">Female</td>
<td rowspan="2" align="left">Female</td>
</tr>
<tr>
<td align="center">-</td>
</tr>
<tr>
<td rowspan="2" align="left">HL-020292</td>
<td align="char" char=".">209.19</td>
<td rowspan="2" align="left">Female</td>
<td rowspan="2" align="left">Female</td>
<td rowspan="2" align="left">Female</td>
</tr>
<tr>
<td align="center">-</td>
</tr>
<tr>
<td rowspan="2" align="left">HL-011500</td>
<td align="char" char=".">209.25</td>
<td rowspan="2" align="left">Male</td>
<td rowspan="2" align="left">Male</td>
<td rowspan="2" align="left">Male</td>
</tr>
<tr>
<td align="char" char=".">215.07</td>
</tr>
<tr>
<td rowspan="2" align="left">HL-019211</td>
<td align="char" char=".">209.19</td>
<td rowspan="2" align="left">Male</td>
<td rowspan="2" align="left">Male</td>
<td rowspan="2" align="left">Male</td>
</tr>
<tr>
<td align="char" char=".">215.04</td>
</tr>
<tr>
<td rowspan="2" align="left">HL-009389</td>
<td align="char" char=".">209.27</td>
<td rowspan="2" align="left">Female</td>
<td rowspan="2" align="left">Female</td>
<td rowspan="2" align="left">Female</td>
</tr>
<tr>
<td align="center">-</td>
</tr>
<tr>
<td rowspan="2" align="left">HL-012554</td>
<td align="char" char=".">209.26</td>
<td rowspan="2" align="left">Male</td>
<td rowspan="2" align="left">Male</td>
<td rowspan="2" align="left">Male</td>
</tr>
<tr>
<td align="char" char=".">215.03</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We then compared the performances of PLINK, seXY and XYalign for dataset 5 using the corrected gender information. The accuracy of PLINK was 99.79% with five mismatched samples (HL-033182, HL-020292, HL-011500, HL-019211 and HL-012554) (<xref ref-type="table" rid="T3">Table&#x20;3</xref>). The accuracy of XYalign was 99.91% with two mismatched samples (HL-009389 and HL-012554). Overall, six samples were mismatched, as predicted by PLINK and XYalign. STR analysis showed that the gender of these samples was consistent with their self-reported gender and matched the predicted results of seGMM analysis (<xref ref-type="table" rid="T4">Table&#x20;4</xref>). Furthermore, the accuracy of seXY was 49.23%. The loss of accuracy in seXY for dataset 5 was because the distribution of Y&#x20;chromosome missingness in the male and female samples were&#x20;confounded (<xref ref-type="sec" rid="s11">Supplementary Figure S9</xref>). Finally, the performances of these tools were assessed using the WGS data (dataset 6, <italic>n</italic>&#x20;&#x3d; 27). The accuracy of all tools was 100% for dataset 6 (<xref ref-type="table" rid="T3">Table&#x20;3</xref>).</p>
</sec>
<sec id="s3-3">
<title>The seGMM Tool Identifies Samples With Sex Chromosomal Abnormalities</title>
<p>The seGMM tool can identify six sex chromosomal karyotypes (XX, XY, XYY, XXY, XXX, and X) using Xmap and Ymap. In a large cohort, the distribution of Xmap and Ymap was normal in females and males. The Xmap or Ymap values of samples with sex chromosome abnormalities such as XYY and XXY were significantly different and were recognized as outliers compared to samples with normal sex chromosomes. Three samples in dataset 5 (HL-029620, HL-009382 and HL-019110) were classified as the XYY karyotype. In dataset 5, the average rate of reads mapping to the X and Y chromosomes in the female samples were 0.035&#x20;&#xb1; 0.0020 and 1.61e-05&#x20;&#xb1; 3.36e-05, respectively, and 0.018&#x20;&#xb1; 0.0011 and 0.00067 &#xb1; 0.00014, respectively, for the male samples. The rate of reads mapping to the Y chromosome for the three outlier samples was twice as high as the mean value of Ymap in all the male samples (Ymap<sub>HL-029620</sub> &#x3d; 0.0015, Ymap<sub>HL-009382</sub> &#x3d; 0.0018, and Ymap<sub>HL-019110</sub> &#x3d; 0.0015), thereby suggesting a XYY karyotype by seGMM (<xref ref-type="fig" rid="F4">Figure&#x20;4B</xref>). Furthermore, although HL-009389 and HL-012554 samples were located close together in the&#x20;middle of the plot, they were correctly predicted by seGMM as female and male, respectively (<xref ref-type="fig" rid="F4">Figure&#x20;4B</xref>). This is because features such as Xmap and SRY_dep, which are not&#x20;shown in <xref ref-type="fig" rid="F4">Figure&#x20;4B</xref>, clearly separated all female and male&#x20;samples (<xref ref-type="sec" rid="s11">Supplementary Figure S10</xref>). This demonstrated the significance of incorporating key features to improve the accuracy of the gender prediction model. In the other datasets, sex chromosomal abnormalities were not identified.</p>
<p>Next, we evaluated the accuracy of the data-based sex chromosome karyotype of these three samples by analyzing the copy number ratios of Y chromosome-specific genes (<italic>SRY</italic>, <italic>ZFY</italic> and <italic>DAZ1</italic>) by RT&#x2013;qPCR. We used HL-007935 and HL-012246 samples as controls for females and males based on the STR analysis results. The copy number ratio for normal females was 0. The copy number ratio for normal male samples was 1. We analyzed the copy number ratios of HL-029620, HL-009382 and HL-019110 samples in dataset 5 and found that the copy number ratio of HL-019110 was 2 (<xref ref-type="fig" rid="F5">Figure&#x20;5</xref>). This confirmed that the karyotype for the HL-019110 sample was&#x20;XYY.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Quantitative determination of Y chromosome copy number.</p>
</caption>
<graphic xlink:href="fgene-13-850804-g005.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>In this study, we characterized the performance of the new gender inference tool, seGMM, in comparison with the other established gender inference tools using NGS data, especially TGS panel data. The seGMM tool used unsupervised clustering to classify samples based on X and Y sex chromosomal features. The performance and accuracy of the seGMM tool were significantly better than other existing gender inference tools using TGS, WES, and WGS data. Furthermore, seGMM accurately predicted six different sex chromosomal karyotypes, including those with sex chromosome abnormalities. The mean and standard deviation values of Xmap and Ymap were used to determine potential sex chromosome aneuploidy in the male and female samples by seGMM. Previous studies have identified sex chromosomal aneuploidy in samples by measuring the intensities of X and Y chromosomes (<xref ref-type="bibr" rid="B2">Bycroft et&#x20;al., 2018</xref>; <xref ref-type="bibr" rid="B33">Turro et&#x20;al., 2020</xref>). A similar strategy was incorporated into the seGMM tool and used to validate a sample with sex chromosome karyotype XYY in the in-house WES dataset. Samples with sex chromosomal abnormalities may result in false calling of the genotype. This can affect identification of pathogenic variants in the sex chromosomes. Therefore, samples with sex chromosome abnormalities should be removed or recalled genotypes to ensure accuracy of the clinical diagnosis.</p>
<p>The seGMM tool applies unsupervised learning algorithm to infer gender of samples from the TGS panel data to overcome the pitfalls of existing tools. The TGS panel consists of a select set of genes with known or suspected association with the disease under study. The advantage of TGS in clinical genetic testing includes high sequencing depth of the genes of interest, which allows identification of rare and causative variants (<xref ref-type="bibr" rid="B6">Eggers et&#x20;al., 2016</xref>; <xref ref-type="bibr" rid="B1">Bewicke-Copley et&#x20;al., 2019</xref>). The data size of TGS depends on the number of genes included in the panel and the methods used for targeted sequencing including target enrichment by hybridization capture and amplicon sequencing. Hence, the number of variants and sequencing depth of the X and Y chromosomes varies for different TGS panels. The accuracy of existing methods in inferring gender using TGS data is unsatisfactory because the algorithms are either based on a data-dependent threshold or supervised learning on a fixed sample set (<xref ref-type="bibr" rid="B23">Purcell et&#x20;al., 2007</xref>; <xref ref-type="bibr" rid="B24">Qian et&#x20;al., 2017</xref>). PLINK uses a data-dependent threshold strategy that determines gender by computing the F coefficients based on the observed and expected number of homozygous markers and requires reasonable minor allele frequency estimates. However, variants detected in the TGS datasets tend to have lower minor allele frequency and the number of variants detected in the X chromosome are limited. Therefore, F coefficient of the male and female samples based on the TGS data is ambiguous. Furthermore, the logistic regression classifier for the seXY tool was based on GWAS data collected from prostate cancer and ovarian cancer samples, and was not suitable for TGS panels because the distribution of X chromosome heterozygosity and Y chromosome missingness varied between the TGS panel dataset and the training dataset. In contrast, seGMM applied a Gaussian mixture model to infer gender. Therefore, the performance and accuracy of seGMM were higher for data with different covariance structures and were adaptable to include fresh samples.</p>
<p>Our study also demonstrated that the gender-inference accuracy of seGMM improved when the information from both X and Y chromosomes was available. For example, the accuracy of seGMM for dataset 1 was 84.56% when the data included only X chromosomal features, but the accuracy increased to 99.52% upon adding Y chromosomal features. Moreover, the accuracy of seGMM was lower for male samples compared to female samples in datasets 2 and 3 because the sequencing data did not contain information on genes in the Y chromosome. Our data also suggested that addition of probes that target unique regions of the Y chromosome such as the <italic>SRY</italic> exon, which is involved in typical male sex development (<xref ref-type="bibr" rid="B9">Gubbay et&#x20;al., 1990</xref>; <xref ref-type="bibr" rid="B19">Parma and Radi, 2012</xref>), is helpful for inferring genders using the TGS panel&#x20;data.</p>
<p>DNA sequencing data from the lymphoblastoid cell lines (LCLs) established from the EBV-infected peripheral blood mononuclear cells (PBMCs) may confound the prediction of sex chromosomal karyotypes. A previous study demonstrated that EBV transformation adversely affected the genomic DNA stability; mosaic loss of X chromosome was observed in 7% (2/29) of the samples analyzed (<xref ref-type="bibr" rid="B28">Shirley et&#x20;al., 2012</xref>). The false-positive rates due to EBV-induced mutations in LCLs may reduce the accuracy of predicting the sex chromosomal karyotypes. The majority of samples in the 1000G WES data were derived from LCLs, but we did not identify any sample in this dataset with abnormal sex chromosomal aneuploidy. The box plots of reads mapped to the Y chromosome showed a much lower value for one male sample (NA12413) compared to the others, thereby indicating potential loss of chromosome Y (<xref ref-type="sec" rid="s11">Supplementary Figure S11</xref>). However, we could not confirm if the loss of Y chromosome was due to LCLs or as a result of authentic sex chromosome abnormalities since experimental validation is required for further analysis.</p>
<p>A few critical considerations are necessary while applying seGMM. First, seGMM is not applicable when the targeted sequencing data does not include genes located on the X and Y chromosomes. Secondly, seGMM requires a sufficient sample size to train an accurate model. Therefore, prediction accuracy should be enhanced for small sample datasets by including reference data (using --reference function parameter). We have provided two reference datasets that were generated from the 1000G WES and WGS datasets. In addition, samples sequenced with the same version of TGS panel can be used to build a user&#x2019;s own reference to maximize the accuracy of gender prediction. When applying seGMM, the experimental and analytical methods between reference data and testing data need to be consistent to prevent bias. Thirdly, parallel computing (using --num_threshold function parameter) is recommended to speed up the analysis since the seGMM tool collects more features than the other existing tools. Lastly, the use of Empirical Rule to classify individual karyotypes improves the recall rate, but may magnify false positives rate, as has been reported in previous study using this strategy (<xref ref-type="bibr" rid="B33">Turro et&#x20;al., 2020</xref>). In addition, many factors may contribute to false positive predicition results, including the copy number variations such as large deletions or insertions on the sex chromosomes or genetic chimerism. Therefore, to overcome this limitation, karyotyping of predicted abnormal samples is recommended to confirm the sample karyotype.</p>
<p>In conclusion, we demonstrate that the performance and accuracy of seGMM, a new tool to infer sex chromosomal karyotypes based on a Gaussian mixture model, was significantly higher and satisfactory for TGS, WES, and WGS datasets, including those with samples containing sex chromosomal abnormalities compared to other existing tools. Hence, seGMM is a promising tool for inferring the gender of samples in TGS, WES, and WGS datasets.</p>
</sec>
</body>
<back>
<sec id="s5">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s11">Supplementary Material</xref>, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s6">
<title>Ethics Statement</title>
<p>Written informed consent was obtained from the individual(s), and minor(s)&#x2019; legal guardian/next of kin, for the publication of any potentially identifiable images or data included in this article.</p>
</sec>
<sec id="s7">
<title>Author Contributions</title>
<p>SL developed the seGMM tool and drafted the manuscript; YZ, QZ, and MC participated in data pre-processing and testing; CW, XW, and LW performed the experiments; YL and HG reviewed and revised the manuscript; FB designed and supervised the study; FB reviewed the manuscript; All authors contributed to the article and approved the final submitted version.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>This work was financially supported by the 1&#xb7;3&#xb7;5 project for disciplines of excellence, West China Hospital, Sichuan University.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2022.850804/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2022.850804/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bewicke-Copley</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Arjun Kumar</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Palladino</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Korfi</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Applications and Analysis of Targeted Genomic Sequencing in Cancer Studies</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>17</volume>, <fpage>1348</fpage>&#x2013;<lpage>1359</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2019.10.004</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bycroft</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Freeman</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Petkova</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Band</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Elliott</surname>
<given-names>L. T.</given-names>
</name>
<name>
<surname>Sharp</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>The UK Biobank Resource with Deep Phenotyping and Genomic Data</article-title>. <source>Nature</source> <volume>562</volume> (<issue>7726</issue>), <fpage>203</fpage>&#x2013;<lpage>209</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-018-0579-z</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Fastp: an Ultra-fast All-In-One FASTQ Preprocessor</article-title>. <source>Bioinformatics</source> <volume>34</volume> (<issue>17</issue>), <fpage>i884</fpage>&#x2013;<lpage>i890</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty560</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Danecek</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Auton</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abecasis</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Albers</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Banks</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>DePristo</surname>
<given-names>M. A.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>The Variant Call Format and VCFtools</article-title>. <source>Bioinformatics</source> <volume>27</volume> (<issue>15</issue>), <fpage>2156</fpage>&#x2013;<lpage>2158</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btr330</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Di Resta</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Galbiati</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Carrera</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ferrari</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Next-generation Sequencing Approach for the Diagnosis of Human Diseases: Open Challenges and New Opportunities</article-title>. <source>EJIFCC</source> <volume>29</volume> (<issue>1</issue>), <fpage>4</fpage>&#x2013;<lpage>14</lpage>. </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eggers</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sadedin</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>van den Bergen</surname>
<given-names>J.&#x20;A.</given-names>
</name>
<name>
<surname>Robevska</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Ohnesorg</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hewitt</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Disorders of Sex Development: Insights from Targeted Gene Sequencing of a Large International Patient Cohort</article-title>. <source>Genome Biol.</source> <volume>17</volume> (<issue>1</issue>), <fpage>243</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-016-1105-y</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Genomes Project</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Abecasis</surname>
<given-names>G. R.</given-names>
</name>
<name>
<surname>Altshuler</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Auton</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Brooks</surname>
<given-names>L. D.</given-names>
</name>
<name>
<surname>Durbin</surname>
<given-names>R. M.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>A Map of Human Genome Variation from Population-Scale Sequencing</article-title>. <source>Nature</source> <volume>467</volume> (<issue>7319</issue>), <fpage>1061</fpage>&#x2013;<lpage>1073</lpage>. <pub-id pub-id-type="doi">10.1038/nature09534</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Genomes Project</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Auton</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Brooks</surname>
<given-names>L. D.</given-names>
</name>
<name>
<surname>Durbin</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Garrison</surname>
<given-names>E. P.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>H. M.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>A Global Reference for Human Genetic Variation</article-title>. <source>Nature</source> <volume>526</volume> (<issue>7571</issue>), <fpage>68</fpage>&#x2013;<lpage>74</lpage>. <pub-id pub-id-type="doi">10.1038/nature15393</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gubbay</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Collignon</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Koopman</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Capel</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Economou</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>M&#xfc;nsterberg</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>1990</year>). <article-title>A Gene Mapping to the Sex-Determining Region of the Mouse Y Chromosome Is a Member of a Novel Family of Embryonically Expressed Genes</article-title>. <source>Nature</source> <volume>346</volume> (<issue>6281</issue>), <fpage>245</fpage>&#x2013;<lpage>250</lpage>. <pub-id pub-id-type="doi">10.1038/346245a0</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Long</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Coe</surname>
<given-names>B. P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Inherited and Multiple De Novo Mutations in Autism/developmental Delay Risk Genes Suggest a Multifactorial Model</article-title>. <source>Mol. Autism</source> <volume>9</volume>, <fpage>64</fpage>. <pub-id pub-id-type="doi">10.1186/s13229-018-0247-z</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bae</surname>
<given-names>J.&#x20;S.</given-names>
</name>
<name>
<surname>Ryu</surname>
<given-names>G. H.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>N. K. D.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>A Method to Evaluate the Quality of Clinical Gene-Panel Sequencing Data for Single-Nucleotide Variant Detection</article-title>. <source>J.&#x20;Mol. Diagn.</source> <volume>19</volume> (<issue>5</issue>), <fpage>651</fpage>&#x2013;<lpage>658</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmoldx.2017.06.001</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Durbin</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Fast and Accurate Short Read Alignment with Burrows-Wheeler Transform</article-title>. <source>Bioinformatics</source> <volume>25</volume> (<issue>14</issue>), <fpage>1754</fpage>&#x2013;<lpage>1760</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp324</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Handsaker</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wysoker</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Fennell</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ruan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Homer</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>The Sequence Alignment/Map Format and SAMtools</article-title>. <source>Bioinformatics</source> <volume>25</volume> (<issue>16</issue>), <fpage>2078</fpage>&#x2013;<lpage>2079</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp352</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ahmad</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Colby</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Applications of Targeted Gene Capture and Next-Generation Sequencing Technologies in Studies of Human Deafness and Other Genetic Disabilities</article-title>. <source>Hearing Res.</source> <volume>288</volume> (<issue>1-2</issue>), <fpage>67</fpage>&#x2013;<lpage>76</lpage>. <pub-id pub-id-type="doi">10.1016/j.heares.2012.01.004</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kuang</surname>
<given-names>J.-Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>G.-M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.-J.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>W.-M.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Y Chromosome Interstitial Deletion Induced Y-STR Allele Dropout in AMELY-Negative Individuals</article-title>. <source>Int. J.&#x20;Leg. Med</source> <volume>126</volume> (<issue>5</issue>), <fpage>713</fpage>&#x2013;<lpage>724</lpage>. <pub-id pub-id-type="doi">10.1007/s00414-012-0720-8</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McKenna</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hanna</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Banks</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Sivachenko</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cibulskis</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kernytsky</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>The Genome Analysis Toolkit: a MapReduce Framework for Analyzing Next-Generation DNA Sequencing Data</article-title>. <source>Genome Res.</source> <volume>20</volume> (<issue>9</issue>), <fpage>1297</fpage>&#x2013;<lpage>1303</lpage>. <pub-id pub-id-type="doi">10.1101/gr.107524.110</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McPherson</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Genetic Diagnosis and Testing in Clinical Practice</article-title>. <source>Clin. Med. Res.</source> <volume>4</volume> (<issue>2</issue>), <fpage>123</fpage>&#x2013;<lpage>129</lpage>. <pub-id pub-id-type="doi">10.3121/cmr.4.2.123</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nielsen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wohlert</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>1990</year>). <article-title>Sex Chromosome Abnormalities Found&#x20;Among 34,910 Newborn Children: Results from a 13-year Incidence Study in Arhus, Denmark</article-title>. <source>Birth Defects Orig Artic Ser.</source> <volume>26</volume> (<issue>4</issue>), <fpage>209</fpage>&#x2013;<lpage>223</lpage>. </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Parma</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Radi</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Molecular Mechanisms of Sexual Development</article-title>. <source>Sex. Dev.</source> <volume>6</volume> (<issue>1-3</issue>), <fpage>7</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1159/000332209</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pedersen</surname>
<given-names>B. S.</given-names>
</name>
<name>
<surname>Quinlan</surname>
<given-names>A. R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Mosdepth: Quick Coverage Calculation for Genomes and Exomes</article-title>. <source>Bioinformatics</source> <volume>34</volume> (<issue>5</issue>), <fpage>867</fpage>&#x2013;<lpage>868</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx699</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Phillips</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Douglas</surname>
<given-names>M. P.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>The Global Market for Next-Generation Sequencing Tests Continues its Torrid Pace</article-title>. <source>J.&#x20;Precis Med.</source> <volume>4</volume>. </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Phillips</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Douglas</surname>
<given-names>M. P.</given-names>
</name>
<name>
<surname>Marshall</surname>
<given-names>D. A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Expanding Use of Clinical Genome Sequencing and the Need for More Data on Implementation</article-title>. <source>JAMA</source> <volume>324</volume> (<issue>20</issue>), <fpage>2029</fpage>&#x2013;<lpage>2030</lpage>. <pub-id pub-id-type="doi">10.1001/jama.2020.19933</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Purcell</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Neale</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Todd-Brown</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Thomas</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>M. A. R.</given-names>
</name>
<name>
<surname>Bender</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>PLINK: a Tool Set for Whole-Genome Association and Population-Based Linkage Analyses</article-title>. <source>Am. J.&#x20;Hum. Genet.</source> <volume>81</volume> (<issue>3</issue>), <fpage>559</fpage>&#x2013;<lpage>575</lpage>. <pub-id pub-id-type="doi">10.1086/519795</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qian</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Busam</surname>
<given-names>J.&#x20;A.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>O&#x2019;Mara</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Eeles</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Schumacher</surname>
<given-names>F.&#x20;R.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>seXY: a Tool for Sex Inference from Genotype Arrays</article-title>.&#x20;<source>Bioinformatics</source> <volume>33</volume> (<issue>4</issue>), <fpage>btw696</fpage>&#x2013;<lpage>563</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btw696</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<collab>R Core Team</collab> (<year>2021</year>). <source>R: A Language and Environment for Statistical Computing</source>. <publisher-loc>Vienna, Austria</publisher-loc>: <publisher-name>R Foundation for Statistical Computing</publisher-name>. <comment>URL <ext-link ext-link-type="uri" xlink:href="https://www.R-project.org/">https://www.R-project.org/</ext-link>.</comment> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saudi Mendeliome</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Comprehensive Gene Panels Provide Advantages over Clinical Exome Sequencing for Mendelian Diseases</article-title>. <source>Genome Biol.</source> <volume>16</volume>, <fpage>134</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-015-0693-2</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Scrucca</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Fop</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Murphy</surname>
<given-names>T. B.</given-names>
</name>
<name>
<surname>Raftery</surname>
<given-names>A. E.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Mclust 5: Clustering, Classification and Density Estimation Using Gaussian Finite Mixture Models</article-title>. <source>R. J.</source> <volume>8</volume> (<issue>1</issue>), <fpage>289</fpage>&#x2013;<lpage>317</lpage>. <pub-id pub-id-type="doi">10.32614/rj-2016-021</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shirley</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Baugher</surname>
<given-names>J.&#x20;D.</given-names>
</name>
<name>
<surname>Stevens</surname>
<given-names>E. L.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gerry</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Beiswanger</surname>
<given-names>C. M.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Chromosomal Variation in Lymphoblastoid Cell Lines</article-title>. <source>Hum. Mutat.</source> <volume>33</volume> (<issue>7</issue>), <fpage>1075</fpage>&#x2013;<lpage>1086</lpage>. <pub-id pub-id-type="doi">10.1002/humu.22062</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sullivan</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Mannucci</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kimpton</surname>
<given-names>C. P.</given-names>
</name>
<name>
<surname>Gill</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>1993</year>). <article-title>A Rapid and Quantitative DNA Sex Test: Fluorescence-Based PCR Analysis of X-Y&#x20;Homologous Gene Amelogenin</article-title>. <source>Biotechniques</source> <volume>15</volume> (<issue>4</issue>), <fpage>636</fpage>&#x2013;<lpage>638</lpage>. </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tarasov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Vilella</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Cuppen</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Nijman</surname>
<given-names>I. J.</given-names>
</name>
<name>
<surname>Prins</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Sambamba: Fast Processing of NGS Alignment Formats</article-title>. <source>Bioinformatics</source> <volume>31</volume> (<issue>12</issue>), <fpage>2032</fpage>&#x2013;<lpage>2034</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv098</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Taylor</surname>
<given-names>J.&#x20;C.</given-names>
</name>
<name>
<surname>Martin</surname>
<given-names>H. C.</given-names>
</name>
<name>
<surname>Lise</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Broxholme</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cazier</surname>
<given-names>J.&#x20;B.</given-names>
</name>
<name>
<surname>Rimmer</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Factors Influencing success of Clinical Genome Sequencing across a Broad Spectrum of Disorders</article-title>. <source>Nat. Genet.</source> <volume>47</volume> (<issue>7</issue>), <fpage>717</fpage>&#x2013;<lpage>726</lpage>. <pub-id pub-id-type="doi">10.1038/ng.3304</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thangaraj</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Reddy</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Is the Amelogenin Gene Reliable for Gender Identification in Forensic Casework and Prenatal Diagnosis?</article-title> <source>Int. J.&#x20;Leg. Med</source> <volume>116</volume> (<issue>2</issue>), <fpage>121</fpage>&#x2013;<lpage>123</lpage>. <pub-id pub-id-type="doi">10.1007/s00414-001-0262-y</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Turro</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Astle</surname>
<given-names>W. J.</given-names>
</name>
<name>
<surname>Megy</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Gr&#xe4;f</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Greene</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Shamardina</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Whole-genome Sequencing of Patients with Rare Diseases in a National Health System</article-title>. <source>Nature</source> <volume>583</volume> (<issue>7814</issue>), <fpage>96</fpage>&#x2013;<lpage>102</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-020-2434-2</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Webster</surname>
<given-names>T. H.</given-names>
</name>
<name>
<surname>Couse</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Grande</surname>
<given-names>B. M.</given-names>
</name>
<name>
<surname>Karlins</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Phung</surname>
<given-names>T. N.</given-names>
</name>
<name>
<surname>Richmond</surname>
<given-names>P. A.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Identifying, Understanding, and Correcting Technical Artifacts on the Sex Chromosomes in Next-Generation Sequencing Data</article-title>. <source>Gigascience</source> <volume>8</volume> (<issue>7</issue>). <pub-id pub-id-type="doi">10.1093/gigascience/giz074</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Coulouris</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zaretskaya</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Cutcutache</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Rozen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Madden</surname>
<given-names>T. L.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Primer-BLAST: a Tool to Design Target-specific Primers for Polymerase Chain Reaction</article-title>. <source>BMC Bioinformatics</source> <volume>13</volume>, <fpage>134</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-13-134</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>
