<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article article-type="methods-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1635734</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2025.1635734</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A likelihood ratio framework for inferring close kinship from dynamically selected SNPs</article-title>
<alt-title alt-title-type="left-running-head">Ge et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2025.1635734">10.3389/fgene.2025.1635734</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Ge</surname>
<given-names>Jianye</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Budowle</surname>
<given-names>Bruce</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cariaso</surname>
<given-names>Michael</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/65697/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mittelman</surname>
<given-names>Kristen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Mittelman</surname>
<given-names>David</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3080615/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Othram Inc.</institution>, <addr-line>The Woodlands</addr-line>, <addr-line>TX</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Forensic Medicine</institution>, <institution>University of Helsinki</institution>, <addr-line>Helsinki</addr-line>, <country>Finland</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Forensic Science Institute</institution>, <institution>Radford University</institution>, <addr-line>Radford</addr-line>, <addr-line>VA</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/142023/overview">Ka-Chun Wong</ext-link>, City University of Hong Kong, Hong Kong SAR, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2270089/overview">Guanju Ma</ext-link>, Hebei Medical University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3089975/overview">Jiaming Xue</ext-link>, Sichuan University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: David Mittelman, <email>david@othram.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>23</day>
<month>07</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1635734</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>24</day>
<month>06</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Ge, Budowle, Cariaso, Mittelman and Mittelman.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Ge, Budowle, Cariaso, Mittelman and Mittelman</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Forensic genetic genealogy (FGG) is a force-multiplier for human identification, leveraging dense single nucleotide polymorphism (SNP) data to infer relationships through identity by descent (IBD) segment analysis. Although powerful for investigative lead generation, broad adoption of SNP-based identification methods by the forensic community, especially medical examiners and crime laboratories, necessitates likelihood ratio (LR)-based relationship testing, to align with traditional kinship testing standards. To address this gap, a novel method was developed that incorporates LR calculations into FGG and SNP testing workflows. This approach is unique in that it dynamically selects unlinked, highly informative SNPs based on configurable thresholds for minor allele frequency (MAF) and minimum genetic distance for a robust and reliable analysis. Employing a curated panel of 222,366 SNPs from gnomAD v4 and data from the 1,000 genomes project, high accuracy in resolving relationships up to second-degree relatives can be achieved. For example, a subset of 126 SNPs (MAF &#x3e; 0.4, minimum genetic distance of 30&#xa0;cM) yielded 96.8% accuracy and a weighted F1 score of 0.975 across 2,244 tested pairs. This LR-based methodology enables forensic laboratories to select informative SNPs and integrate modern genomic data with existing accredited relationship testing frameworks, providing critical statistical support for close-relationship comparisons and enhances the rigor of FGG- and SNP-based human identification applications.</p>
</abstract>
<kwd-group>
<kwd>forensic genetic genealogy</kwd>
<kwd>single nucleotide polymorphism</kwd>
<kwd>likelihood ratio</kwd>
<kwd>kinship analysis</kwd>
<kwd>identity by descent</kwd>
<kwd>relationship testing</kwd>
<kwd>whole genome sequencing</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>Advances in genomic analyses, particularly massively parallel sequencing (MPS), have enabled unprecedented high throughput SNP analyses which have fostered the development of Forensic Genetic Genealogy (FGG). This subdiscipline of forensic genetics exploits genome-wide dense single nucleotide polymorphism (SNP) data generated by whole genome sequencing (WGS) or microarrays to determine near and distant kinship relationships. While targeted sequencing, in theory, can be used in forensic contexts to interrogate selected SNPs at lower cost (although current commercial forensic options are relatively costly), whole genome sequencing (WGS) offers a more comprehensive approach, capturing significantly more genetic variation and enabling deeper kinship inference. As sequencing costs continue to decline, WGS is becoming increasingly preferred in FGG for its broader utility and richer dataset (<xref ref-type="bibr" rid="B31">Mandape, et al., 2024</xref>). The kinship relationships can facilitate investigative leads for identification of human remains and source attribution of donors of crime scene DNA evidence more effectively than current forensic genetic methods (<xref ref-type="bibr" rid="B14">Dowdeswell, 2023</xref>; <xref ref-type="bibr" rid="B25">Kling, 2019</xref>; <xref ref-type="bibr" rid="B26">Kling and Tillmar, 2019</xref>; <xref ref-type="bibr" rid="B31">Mandape et al., 2024</xref>; <xref ref-type="bibr" rid="B8">Budowle et al., 2024</xref>). Kinship associations typically are determined using either IBS (identity by State) or IBD (identity by Descent) segmented-based methods (<xref ref-type="bibr" rid="B22">Henn et al., 2012</xref>; <xref ref-type="bibr" rid="B5">Browning and Browning, 2011</xref>; <xref ref-type="bibr" rid="B6">Browning and Browning, 2013</xref>; <xref ref-type="bibr" rid="B23">Huff et al., 2011</xref>; <xref ref-type="bibr" rid="B29">Lipatov et al., 2015</xref>; <xref ref-type="bibr" rid="B37">Staples et al., 2016</xref>; <xref ref-type="bibr" rid="B24">Kaplanis et al., 2018</xref>; <xref ref-type="bibr" rid="B35">Ramstetter et al., 2018</xref>; <xref ref-type="bibr" rid="B32">Morimoto et al., 2018</xref>; <xref ref-type="bibr" rid="B36">Seidman et al., 2020</xref>). While these approaches provide high accuracy, they differ from the likelihood ratio (LR)-based framework traditionally used in kinship analysis, which underpins the validity and acceptance of short tandem repeat (STR) interpretation methods in forensic applications. LR-based statistics are routinely employed to support identifications, making them a critical component of forensic practice. To enable the forensic community to leverage SNP data, an analogous LR-based framework is needed that offers the flexibility, rigor, interpretability, and validity required for broad acceptance. Accordingly, an LR-based interpretation method tailored to whole genome sequencing (WGS) data for SNP selection and relationship inference was developed, with a focus on pairwise comparisons up to second degree relatives.</p>
<p>Like traditional kinship analysis, the approach described herein calculates the likelihoods of each selected SNP for specific relationships, with LRs obtained by comparing the likelihoods of genetic data under alternative relationships. Assuming independence among SNPs, the cumulative LR is calculated by multiplying the LR values for each individual SNP. Therefore, selecting the most informative SNPs is critical to the success of the method. Our approach is unique in that it enables dynamic SNP selection in tandem with likelihood ratio (LR) calculations compared with traditional kinship software that relies on fixed, pre-selected markers. This dynamic integration allows for greater flexibility and improved performance when working with WGS data. A straightforward approach was used to refine a subset of SNPs from the hundreds of thousands to millions identified by WGS, prioritizing high minor allele frequency (MAF) markers observed in each case. These markers offer high discrimination power for relationship inference. Subsequently, the data are further subset to include SNPs located in genomic regions identified by Genome-in-a-Bottle as being easy to sequence or genotype (<xref ref-type="bibr" rid="B15">Dwarshuis et al., 2024</xref>). This subset is then further refined by selecting only SNPs that exhibit nominal or no linkage and linkage disequilibrium (LD). This approach provides a curated panel of SNPs with high MAF for robust LR support for true close kinship relationships, particularly up to the second degree.</p>
<p>Indeed, <xref ref-type="bibr" rid="B39">Yousefi et al. (2018)</xref> demonstrated in a Dutch sample population that 50 independent SNPs (MAF &#x3e; 0.2; average MAF &#x3d; 0.35) are sufficient to yield low probabilities of identity (or population match probabilities) of 6.9 &#xd7; 10<sup>&#x2212;20</sup> and 1.2 &#xd7; 10<sup>&#x2212;10</sup> for unrelated individuals and siblings, respectively. <xref ref-type="bibr" rid="B12">Chakraborty et al. (1999)</xref> estimated that approximately 40 SNPs (MAF &#x223c; 0.5) could achieve random match probabilities around 10<sup>&#x2212;15</sup> and 33 SNPs (MAF &#x223c; 0.5) would be required to reach an exclusion probability of 99.9% in a typical trio paternity case. Furthermore, the higher the MAF for a suite of binary SNPs (i.e., approaching the highest heterozygosity of 0.5) is globally, the lower are the effects of population substructure. Moreover, with high heterozygosity any positive predictive power with binary SNPs for associations with private genetic information would be negligible (<xref ref-type="bibr" rid="B9">Budowle and van Daal, 2008</xref>).</p>
<p>Here, the validation of a statistical approach, KinSNP-LR (version 1.1), is described for computing LRs based on WGS generated SNP data. Instead of <italic>a priori</italic> selecting a fixed panel of informative SNPs, the first SNP on an end of a chromosome passing the MAF threshold is selected from a large candidate panel and the next SNP at a specified genetic distance (e.g., 30&#x2013;50 centimorgans (cM)) and meeting the MAF criterion is selected and so on across the genome. The LRs for multiple relationships are calculated based on the methods described in <xref ref-type="bibr" rid="B38">Thompson (1975)</xref>, <xref ref-type="bibr" rid="B20">Ge et al. (2010)</xref>, and <xref ref-type="bibr" rid="B21">Ge et al. (2011)</xref>. This approach maximizes the number of SNPs with little to no linkage and LD selected in a case-specific manner.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and methods/implementation</title>
<sec id="s2-1">
<title>Empirical genomic data</title>
<p>A large, preselected SNP panel (222,366 SNPs) from gnomAD v4 (<xref ref-type="bibr" rid="B13">Chen et al., 2024</xref>) was used as the data foundation for this validation study (details on this panel can be found in <xref ref-type="sec" rid="s11">Supplementary Material A</xref>). The SNP allele frequencies and genetic distances between the SNPs in this panel will be used in SNP selection and likelihood calculation in the kinship analysis. The SNPs in the panel were filtered and obtained by quality control, MAF, and &#x201c;Not in all difficult regions&#x201d; regions. The data contain nine populations: Admixed America, African, Ashkenazi Jewish, East Asian, Finnish, Non-Finnish European, Middle Eastern, South Asian, and Remaining Individuals (includes Amish). Five major populations: African (AFR), Admixed American (AMR), East Asian (EAS), South Asian (SAS), and Non-Finnish European (NFE), were selected in validating KinSNP-LR. More details about preparing this SNP panel can be found in the <xref ref-type="sec" rid="s11">Supplementary Material</xref>.</p>
<p>In addition, the 1,000 genomes project data contain 3,202 whole genome sequenced samples with many closely related pairs, and these related pairs were used to validate the methodology. Each sample in the 1,000 genomes data was converted using the GRCh38 coordinate positions into a tab-separated text format, but only for the SNPs contained in the preselected gnomAD panel. After removing uncertain relationships, there are 1,200 parent-child, 12 full-sibling, and 32&#xa0;second degree pairs in the 1,000 genome project. Unrelated pairs were randomly selected in the populations. The validation study described herein used data in GRCh38 coordinates.</p>
</sec>
<sec id="s2-2">
<title>Simulation data</title>
<p>Pedigrees and phased genotypes simulations were performed using Ped-sim (v1.4) (<xref ref-type="bibr" rid="B10">Caballero et al., 2019</xref>) with the unrelated individuals in the ASW (Americans of African Ancestry in SW United States), CEU Utah Residents with Northern and Western European Ancestry), CHB (Han Chinese in Beijing, China), and MXL (Mexican Ancestry from Los Angeles, United States) populations in the 1,000 genomes project. Each population represents one of four major continental groups (African, East Asian, European, and Admixed American); thus a broad genetic diversity is captured as opposed to simulating all 26 subpopulations, many of which are closely related groups.</p>
<p>Only the SNPs in the preselected SNP panel from gnomAD v4 were used in the simulations, and only unrelated individuals in each of the populations were used as founders in the simulations. The unrelated relationships were confirmed with IBIS (<xref ref-type="bibr" rid="B36">Seidman et al., 2020</xref>) with a maximum total shared IBD segment of 40&#xa0;cM.</p>
<p>Briefly, Ped-sim simulated 50 families as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, in which there were three generations with second degree being the most distant relationship. Each family has 22 parent-child pairs, 20 sibling pairs, 40&#xa0;second degree pairs, and 22 unrelated pairs. In the simulation, descendants inherit recombinant chromosomes, one from each parent, according to the specified pedigree. Recombination follows an interference model and applies sex-average genetic maps. Chromosome segments are tracked through the inheritance process, and genetic markers from the founders are overlaid on the corresponding segments to produce whole genome data. Simulations were conducted with true IBD segments and founder sample identifiers recorded with zero missing genotype call rate, zero opposite homozygote errors, and various genotyping errors (i.e., 0.001, 0.01, and 0.05). In the simulation, the keep_phase flag was enabled, using high-resolution sex average genetic maps from the Broad Institute (see Data availability for details) and chromosome interference maps (nu_p_campbell_X.tsv) from <xref ref-type="bibr" rid="B11">Campbell et al. (2015)</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The simulated family tree using unrelated founders in the 1,000 genomes data. This pedigree includes first to second degree relationships and four unrelated founders. The same format was used for all populations studied herein.</p>
</caption>
<graphic xlink:href="fgene-16-1635734-g001.tif">
<alt-text content-type="machine-generated">Pedigree chart showing a family tree. Two generations are illustrated. The top generation includes two sets of parents, each with a male and female, indicating individuals as squares and circles respectively. The second generation displays multiple offspring, all represented as filled squares, indicating they are male. One square is marked with a filled circle, suggesting a condition or trait. Each figure is labeled with identifiers like F_g1-b1-s1 or F_g2-b3-i1, indicating relationships and generational positions.</alt-text>
</graphic>
</fig>
<p>For LR calculations, the allele frequencies from the corresponding gnomAD major population were used. For example, gnomAD Non-Finnish European frequencies were used for the CEU pairwise LR calculations.</p>
</sec>
<sec id="s2-3">
<title>Candidate SNP selection</title>
<p>To maximize information content, the SNPs with MAF higher than a threshold (e.g., 0.4) in each individual population, were selected for relationship tests. Further, only unlinked SNPs were selected with a Minimum Genetic Distance (MGD) cM threshold, such as 30&#xa0;cM. This selection is necessary for following traditional LR-based kinship analyses that require independent markers to calculate cumulative likelihoods by directly multiplying the likelihood of each marker. LD tends to decay over distances of approximately 1&#xa0;cM or less (<xref ref-type="bibr" rid="B2">Ardlie et al., 2002</xref>), and thus, if the MGD is greater than 1&#xa0;cM, LD between SNPs may be ignored.</p>
<p>Genetic distances between SNP pairs were calculated based on the sex averaged genetic map. Genetic distance maps of both GRCh38/HG38 and GRCh37/HG19 are available at Broad Institute (see Data availability for details). Since the input profile only contains the SNP names and physical positions in the chromosomes, the SNPs&#x2019; physical positions were mapped to their genetic positions, so genetic distance in cM can be obtained between two SNPs on the same chromosome. The positions of the SNPs that are not in the genetic map were interpolated linearly.</p>
<p>To maximize the number of selected markers, the first SNP on an end of a chromosome passing the MAF threshold is selected from a large candidate panel, and the next SNP at a specified genetic distance (i.e., MGD) and meeting the MAF criterion is selected and so on across the genome. This dynamic SNP-selection algorithm applies a high MAF filter to a large candidate panel, then sweeps each chromosome once from an end inward, greedily choosing the first marker and every subsequent marker that lies at least a certain centimorgans (defined by the selected MGD) distance from the last one selected. Because all retained SNPs have similarly high MAF, information content per marker is nearly uniform, so simply maximizing the count within the distance constraint maximizes cumulative power. The procedure runs in linear time after sorting, yet it approximates the optimal distance-constrained subset-selection problem, which is NP-hard for exact solutions, and sufficient for forensic and association applications. The pseudocode of this dynamic SNP selection algorithm can be found in the <xref ref-type="sec" rid="s11">Supplementary Material B</xref>.</p>
</sec>
<sec id="s2-4">
<title>Kinship analysis</title>
<p>Once the SNPs are selected, the pairwise likelihood(s) given a specific relationship(s) is calculated following the methods described in <xref ref-type="bibr" rid="B38">Thompson (1975)</xref>, <xref ref-type="bibr" rid="B20">Ge et al. (2010)</xref>, and <xref ref-type="bibr" rid="B21">Ge et al. (2011)</xref>. This method relies on 1) Identity by Descent (IBD) Coefficients (<xref ref-type="table" rid="T1">Table 1</xref>) and 2) joint genotype probabilities given IBD (<xref ref-type="table" rid="T2">Table 2</xref>).</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Identity by Descent (IBD) Coefficients derived from <xref ref-type="bibr" rid="B27">Li and Sacks (1954)</xref>.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Relationship</th>
<th align="left">&#x3c6;<sub>2</sub>
</th>
<th align="left">&#x3c6;<sub>1</sub>
</th>
<th align="left">&#x3c6;<sub>0</sub>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Identical twins</td>
<td align="left">1</td>
<td align="left">0</td>
<td align="left">0</td>
</tr>
<tr>
<td align="left">Parent-child</td>
<td align="left">0</td>
<td align="left">1</td>
<td align="left">0</td>
</tr>
<tr>
<td align="left">Full sibs</td>
<td align="left">&#xbc;</td>
<td align="left">&#xbd;</td>
<td align="left">&#xbc;</td>
</tr>
<tr>
<td align="left">Half sibs</td>
<td align="left">0</td>
<td align="left">&#xbd;</td>
<td align="left">&#xbd;</td>
</tr>
<tr>
<td align="left">1st cousins</td>
<td align="left">0</td>
<td align="left">&#xbc;</td>
<td align="left">&#xbe;</td>
</tr>
<tr>
<td align="left">Unrelated</td>
<td align="left">0</td>
<td align="left">0</td>
<td align="left">1</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Joint genotype probabilities for two genotypes X and Y given IBD (i.e., &#x3c6;).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Ordered genotypes (X, Y)</th>
<th colspan="3" align="left">Joint probabilities: Pr(X,Y&#x7c; &#x3c6;)</th>
</tr>
<tr>
<th align="left">IBD &#x3d; 2 (&#x3c6;<sub>2</sub>)</th>
<th align="left">IBD &#x3d; 1 (&#x3c6;<sub>1</sub>)</th>
<th align="left">IBD &#x3d; 0 (&#x3c6;<sub>0</sub>)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">A<sub>i</sub>A<sub>i</sub>, A<sub>i</sub>A<sub>i</sub>
</td>
<td align="left">
<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<sup>
<italic>2</italic>
</sup>
</td>
<td align="left">
<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<sup>
<italic>3</italic>
</sup>
</td>
<td align="left">
<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<sup>
<italic>4</italic>
</sup>
</td>
</tr>
<tr>
<td align="left">A<sub>i</sub>A<sub>i</sub>, A<sub>j</sub>A<sub>j</sub>
</td>
<td align="left">0</td>
<td align="left">0</td>
<td align="left">
<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<sup>
<italic>2</italic>
</sup>
<italic>p</italic>
<sub>
<italic>j</italic>
</sub>
<sup>
<italic>2</italic>
</sup>
</td>
</tr>
<tr>
<td align="left">A<sub>i</sub>A<sub>i</sub>, A<sub>i</sub>A<sub>j</sub>
</td>
<td align="left">0</td>
<td align="left">
<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<sup>
<italic>2</italic>
</sup>
<italic>p</italic>
<sub>
<italic>j</italic>
</sub>
</td>
<td align="left">2<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<sup>
<italic>3</italic>
</sup>
<italic>p</italic>
<sub>
<italic>j</italic>
</sub>
</td>
</tr>
<tr>
<td align="left">A<sub>i</sub>A<sub>j</sub>, A<sub>i</sub>A<sub>i</sub>
</td>
<td align="left">0</td>
<td align="left">
<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<sup>
<italic>2</italic>
</sup>
<italic>p</italic>
<sub>
<italic>j</italic>
</sub>
</td>
<td align="left">2<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<sup>
<italic>3</italic>
</sup>
<italic>p</italic>
<sub>
<italic>j</italic>
</sub>
</td>
</tr>
<tr>
<td align="left">A<sub>i</sub>A<sub>i</sub>, A<sub>j</sub>A<sub>k</sub>
</td>
<td align="left">0</td>
<td align="left">0</td>
<td align="left">2<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<sup>
<italic>2</italic>
</sup>
<italic>p</italic>
<sub>
<italic>j</italic>
</sub>
<italic>p</italic>
<sub>
<italic>k</italic>
</sub>
</td>
</tr>
<tr>
<td align="left">A<sub>j</sub>A<sub>k</sub>, A<sub>i</sub>A<sub>i</sub>
</td>
<td align="left">0</td>
<td align="left">0</td>
<td align="left">2<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<sup>
<italic>2</italic>
</sup>
<italic>p</italic>
<sub>
<italic>j</italic>
</sub>
<italic>p</italic>
<sub>
<italic>k</italic>
</sub>
</td>
</tr>
<tr>
<td align="left">A<sub>i</sub>A<sub>j</sub>, A<sub>i</sub>A<sub>j</sub>
</td>
<td align="left">2<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<italic>p</italic>
<sub>
<italic>j</italic>
</sub>
</td>
<td align="left">
<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<italic>p</italic>
<sub>
<italic>j</italic>
</sub> (<italic>p</italic>
<sub>
<italic>i</italic>
</sub> &#x2b; <italic>p</italic>
<sub>
<italic>j</italic>
</sub>)</td>
<td align="left">4<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<sup>
<italic>2</italic>
</sup>
<italic>p</italic>
<sub>
<italic>j</italic>
</sub>
<sup>
<italic>2</italic>
</sup>
</td>
</tr>
<tr>
<td align="left">A<sub>i</sub>A<sub>j</sub>, A<sub>i</sub>A<sub>k</sub>
</td>
<td align="left">0</td>
<td align="left">
<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<italic>p</italic>
<sub>
<italic>j</italic>
</sub>
<italic>p</italic>
<sub>
<italic>k</italic>
</sub>
</td>
<td align="left">4<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<sup>
<italic>2</italic>
</sup>
<italic>p</italic>
<sub>
<italic>j</italic>
</sub>
<italic>p</italic>
<sub>
<italic>k</italic>
</sub>
</td>
</tr>
<tr>
<td align="left">A<sub>i</sub>A<sub>j</sub>, A<sub>k</sub>A<sub>l</sub>
</td>
<td align="left">0</td>
<td align="left">0</td>
<td align="left">4<italic>p</italic>
<sub>
<italic>i</italic>
</sub>
<italic>p</italic>
<sub>
<italic>j</italic>
</sub>
<italic>p</italic>
<sub>
<italic>k</italic>
</sub>
<italic>p</italic>
<sub>
<italic>l</italic>
</sub>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>A<sub>
<italic>i</italic>
</sub>, <italic>A</italic>
<sub>
<italic>j</italic>
</sub>, <italic>A</italic>
<sub>
<italic>k</italic>
</sub> and <italic>A</italic>
<sub>
<italic>l</italic>
</sub> are alleles at the locus with allele frequencies <italic>p</italic>
<sub>
<italic>i</italic>
</sub>, <italic>p</italic>
<sub>
<italic>j</italic>
</sub>, <italic>p</italic>
<sub>
<italic>k</italic>
</sub>, and <italic>p</italic>
<sub>
<italic>l</italic>
</sub>, respectively.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The likelihoods of the observed genotypes given two mutually exclusive hypotheses are compared. The likelihoods of the genetic data under each hypothesis are calculated as follows,<disp-formula id="equ1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>Pr</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>Pr</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">Y</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="normal">&#x3c6;</mml:mi>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="normal">&#x3c6;</mml:mi>
<mml:mn>0</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>Pr</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3c6;</mml:mi>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="normal">&#x3c6;</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>Pr</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">Y</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="normal">&#x3c6;</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="normal">&#x3c6;</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</disp-formula>where X and Y are two genotypes at a single locus, R is the hypothesized relationship, &#x3c6;0, &#x3c6;1, and &#x3c6;2 are IBD &#x3d; 0, 1, and 2, respectively, and Pr(X,Y&#x7c;R) is the likelihood of the genotypes given R. The joint genotype probabilities of X and Y given IBD are shown in <xref ref-type="table" rid="T2">Table 2</xref> (<xref ref-type="bibr" rid="B20">Ge et al., 2010</xref>; <xref ref-type="bibr" rid="B21">Ge et al., 2011</xref>). The cumulative likelihoods across multiple loci are the product of the likelihoods of each locus, assuming the loci are independent.</p>
<p>To avoid zero likelihoods due to mutations, a simple mutation model was implemented for the parent-child relationship. In this model, a constant mutation rate is set for any nucleotide difference; transitions (A&#x2194;G or C&#x2194;T) and transversions (A&#x2194;C, A&#x2194;T, G&#x2194;C, or G&#x2194;T) are treated as equally probable. Since the sequencing error rate likely is much higher than the mutation rate (e.g., 10<sup>&#x2212;8</sup>), the default mutation rate is set at 10<sup>&#x2212;3</sup> (based on Q30 filtering, although it can be configured). While a mutation rate parameter is used herein, one could consider this rate as a genotyping error parameter of which the mutation rate is subsumed. This mutation model was applied only for the parent-child relationship. The impact of mutation for other relationships was considered negligible due to zero probability of IBD &#x3d; 0. With the mutation model, the likelihood of a pair of genotypes given a parent-child relationship would be the product of the parent and the transmission probability from parent to child (the details can be found in Equation 6 in <xref ref-type="bibr" rid="B20">Ge et al. (2010)</xref>.</p>
</sec>
<sec id="s2-5">
<title>Software</title>
<p>KinSNP-LR was implemented in Python (version 3.10&#x2b;). Input genotype data of tested individuals for KinSNP-LR were formatted as a tab-separated text format. Configurable parameters, such as MAF, MGD, mutation rate, genomic build (GRCh38 or GRCh37), and populations allowed for the exploration of a broad parameter space. KinSNP-LR outputs two files: a file contains the cumulative likelihoods and LRs for each population selected, and a file contains the details about each marker and the likelihoods for each population and relationship.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec id="s3-1">
<title>Empirical genomic data</title>
<p>The accuracy of traditional LR-based kinship analysis is directly dependent on the information content of each SNP and the total number of SNPs measured. To maximize information content of a SNP panel, a higher MAF is necessary. <xref ref-type="fig" rid="F2">Figure 2</xref> shows the number of selected SNPs with MAF thresholds, in which the allele frequencies in gnomAD were used. Only SNPs that had minor allele frequencies higher than a MAF threshold in each of the five populations (AFR, AMR, EAS, SAS, and NFE) were selected. With the MAF thresholds of 0.4 and 0.45, only 25,664 and 1,441 SNPs remained, respectively, which are appropriate for the threshold range for SNP selections.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Number of selected SNPs with Minor Allele Frequency (MAF). Minimum Genetic Distance is not considered in this Figure.</p>
</caption>
<graphic xlink:href="fgene-16-1635734-g002.tif">
<alt-text content-type="machine-generated">Line graph showing the decline in the number of SNPs from 103,670 to 11 as the minimum allele frequency (MAF) increases from 0.35 to 0.48.</alt-text>
</graphic>
</fig>
<p>In the 1,000 genomes data, there were 1,200 parent-child, 12 full-sibling, and 32&#xa0;second degree pairs. In addition, 1,000 unrelated pairs were randomly selected with 200 in each of the five populations. In total, 2,244 pairs were selected to validate the LR method. For each pair, the likelihoods are calculated given each of the four relationships (unrelated, parent-child, sibling, and second degree), and each of the five populations were calculated. Assuming the population is known, the relationship is determined as the one with the maximum likelihood in that population.</p>
<p>
<xref ref-type="fig" rid="F3">Figure 3</xref> shows the confusion matrix of determining relationships given MAF &#x3d; 0.4 and MGD &#x3d; 30&#xa0;cM with 126 SNPs selected. Based on this confusion matrix, the accuracy and the weighted F1 score could be calculated as 0.9679 and 0.9751, respectively, which were reasonably high given the limited number of SNPs available with the selection criteria described. As expected, most of the false associations were either with the first degree relationship (parent-child and sibling) or between unrelated and second degree.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Confusion matrix of determining relationships given Minor Allele Frequency (MAF) &#x3d; 0.4 and Minimum Genetic Distance (MGD) &#x3d; 30&#xa0;cM, with 126 selected SNPs. The rows are the true relationships, and the columns are the predicted relationships.</p>
</caption>
<graphic xlink:href="fgene-16-1635734-g003.tif">
<alt-text content-type="machine-generated">Confusion matrix showing true versus predicted relationships. Rows represent true labels: Parent-Child, Sibling, Second-Degree, and Unrelated. Columns represent predicted labels. Values are as follows: 1196 Parent-Child, 11 Sibling, 28 Second-Degree, and 937 Unrelated correctly classified; 3 Parent-Child as Sibling, 1 Parent-Child as Second-Degree, 63 Second-Degree as Unrelated; minimal misclassifications elsewhere.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="sec" rid="s11">Supplementary Table S1</xref> gives more details of each of the false identifications in <xref ref-type="fig" rid="F3">Figure 3</xref>. All the truly related but falsely identified pairs had the first-to-second LR less than 10, except HG00578-HG00582 (i.e., a ratio of 85), and many of them are about 1 or 2. But HG00578-HG00582 was correctly determined with lower MGD thresholds (and more SNPs), such as 10&#xa0;cM (346 SNPs) and 20&#xa0;cM (184 SNPs). The pairs truly unrelated but incorrectly identified as second degree all had first-to-second LRs less than 60, and 77.78% of them had ratios less than 10. Thus, the LR difference could be easily overcome by adding more SNPs. With a lower MGD of 10&#xa0;cM, most of the errors were eliminated (<xref ref-type="sec" rid="s11">Supplementary Table S2</xref>), except a few unrelated pairs were incorrectly determined as second degree. This observation is reasonable as many of the unrelated pairs in the 1,000 genomes data are not truly unrelated but distantly related to a certain degree (<xref ref-type="bibr" rid="B19">Fedorova et al., 2016</xref>), which resembles many actual relationship testing cases.</p>
<p>The ratio of the maximum and second maximum likelihoods among the populations were compared to investigate how confident a conclusion may be made to determine a relationship. As shown in <xref ref-type="fig" rid="F4">Figure 4</xref>, in which MGD &#x3d; 30&#xa0;cM and MAF &#x3d; 0.4 were applied, the parent-child and sibling relationships usually had several magnitudes higher likelihoods than the unrelated and second degree relationships. For parent-child and sibling, 53 out of 1,212 pairs (4.37%) had ratios less than 10, and 230 out of 1,212 pairs (18.98%) had ratios less than 100. For unrelated and second degree, 250 out of 1,032 pairs (24.22%) had the ratio less than 10, and 524 out of 1,032 pairs (50.78%) had ratios less than 100. In addition, all incorrectly identified pairs had the ratios less than 100, and 79.17% (15/72) of them were less than 10. Thus, with these 1,000 genomes data, a conclusion might be made with high confidence if the maximum likelihood of a relationship is 100 times greater than the second maximum likelihood.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>The ratios of the maximum likelihood and the second maximum likelihood among the populations for each pair for Minimum Genetic Distance (MGD) &#x3d; 30&#xa0;cM and Minor Allele Frequency (MAF) &#x3d; 0.4. The circles and triangles represent the pairs with correct and incorrect identifications, respectively.</p>
</caption>
<graphic xlink:href="fgene-16-1635734-g004.tif">
<alt-text content-type="machine-generated">Scatter plot showing maximum to second maximum likelihood ratios versus maximum likelihood, both on a log scale. Data points represent different relationships: unrelated, parent-child, sibling, and second-degree, with correct and incorrect classifications indicated by shapes and colors. Blue circles show parent-child (correct), black circles show unrelated (correct), and other relationships use triangles in varying colors.</alt-text>
</graphic>
</fig>
<p>Further, for a fixed MGD &#x3d; 30&#xa0;cM, the effect of MAF also was investigated. As shown in <xref ref-type="fig" rid="F5">Figure 5</xref>, the highest accuracy was reached with MAF &#x3d; 0.4. With MAF &#x3d; 0.35, more SNPs (i.e., 130) were selected but with the accuracy decreased, because more SNPs with lower information content were selected. MAF &#x3d; 0.38, 0.4, and 0.42 all had the same number of SNPs selected but with different sets of SNPs. MAF &#x3d; 0.4 had a higher accuracy than MAF &#x3d; 0.38 probably because more informative SNPs were selected. MAF &#x3d; 0.42 had a lower accuracy compared with MAF &#x3d; 0.4, probably due to the randomness in the selection process. In general, a reasonable MAF threshold may be between 0.4 and 0.42 based on the current marker selection algorithm.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Distributions of accuracy and F1 score (weighted) with Minor Allele Frequency (MAF) given Minimum Genetic Distance (MGD) &#x3d; 30&#xa0;cM.</p>
</caption>
<graphic xlink:href="fgene-16-1635734-g005.tif">
<alt-text content-type="machine-generated">Line graph showing Accuracy and F1 Score against Minimum Allele Frequency (MAF). Both scores increase, peaking at MAF 0.40, then decline. SNP counts are noted: Accuracy peak at 126 SNPs, with 130 SNPs at MAF 0.35 and 103 at MAF 0.45.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="table" rid="T3">Table 3</xref> displays the number of selected SNPs, the accuracies and weighted F1 scores of relationship identification given various MAF and MGD. The results were consistent with <xref ref-type="fig" rid="F5">Figure 5</xref> that MAF &#x3d; 0.45 could lead to smaller numbers of selected SNPs and lower accuracies of relationship identifications with the current SNP selection method. As expected, lower MAFs and MGDs led to higher number of selected SNPs and higher relationship identification accuracies. With MAF &#x3d; 0.4 and MGD &#x3d; 10&#xa0;cM, 346 SNPs were selected, and the accuracies and F1 scores were close to 100%. However, with MGD less than 50&#xa0;cM, the selected SNPs may be considered linked to some degree. In practice, the linkage of pairs of markers separated by 30 or 40&#xa0;cM may be sufficiently small to have little effect on the match probabilities or likelihood calculations (<xref ref-type="bibr" rid="B7">Buckleton and Triggs, 2006</xref>; <xref ref-type="bibr" rid="B28">Li et al., 2021</xref>). However, closer SNPs with MGDs of 10 or 20&#xa0;cM may have substantial linkage effects on marker independence.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>The numbers of selected SNPs, accuracies, and weighted F1 scores given various Minor Allele Frequency (MAF) and Minimum Genetic Distance (MGD).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="3" align="left">MGD</th>
<th colspan="6" align="left">MAF</th>
</tr>
<tr>
<th colspan="2" align="left">Number of selected SNPs</th>
<th colspan="2" align="left">Accuracy</th>
<th colspan="2" align="left">F1 score (weighted)</th>
</tr>
<tr>
<th align="left">0.40</th>
<th align="left">0.45</th>
<th align="left">0.40</th>
<th align="left">0.45</th>
<th align="left">0.40</th>
<th align="left">0.45</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">10&#xa0;cM</td>
<td align="left">346</td>
<td align="left">211</td>
<td align="left">0.9955</td>
<td align="left">0.9799</td>
<td align="left">0.9958</td>
<td align="left">0.9831</td>
</tr>
<tr>
<td align="left">20&#xa0;cM</td>
<td align="left">184</td>
<td align="left">140</td>
<td align="left">0.9733</td>
<td align="left">0.9541</td>
<td align="left">0.9792</td>
<td align="left">0.9661</td>
</tr>
<tr>
<td align="left">30&#xa0;cM</td>
<td align="left">126</td>
<td align="left">103</td>
<td align="left">0.9679</td>
<td align="left">0.9563</td>
<td align="left">0.9751</td>
<td align="left">0.9671</td>
</tr>
<tr>
<td align="left">40&#xa0;cM</td>
<td align="left">99</td>
<td align="left">85</td>
<td align="left">0.9407</td>
<td align="left">0.9256</td>
<td align="left">0.9573</td>
<td align="left">0.9477</td>
</tr>
<tr>
<td align="left">50&#xa0;cM</td>
<td align="left">80</td>
<td align="left">71</td>
<td align="left">0.9305</td>
<td align="left">0.9100</td>
<td align="left">0.9509</td>
<td align="left">0.9382</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-2">
<title>Simulation data</title>
<p>Similar studies were conducted with simulation data. The number of SNPs after MAF filtering (as shown in <xref ref-type="fig" rid="F6">Figure 6</xref>) were very similar to the numbers in <xref ref-type="fig" rid="F2">Figure 2</xref> (i.e., empirical data), in which the allele frequencies in the 1,000 genomes data were used. The differences were due to the slight allele frequency differences between gnomAD and the 1,000 genomes data.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Number of selected SNPs with Minor Allele Frequency (MAF). Minimum Genetic Distance is not considered in this Figure.</p>
</caption>
<graphic xlink:href="fgene-16-1635734-g006.tif">
<alt-text content-type="machine-generated">Line graph showing the relationship between the minimum allele frequency (MAF) on the x-axis and the number of SNPs on the y-axis. The data points indicate a decrease in SNPs from 90,278 at MAF 0.35 to 23 at MAF 0.48.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="fig" rid="F7">Figure 7</xref> shows the confusion matrix of determining relationships given MAF &#x3d; 0.4 and MGD &#x3d; 30&#xa0;cM with 130 SNPs selected. Based on this confusion matrix, the accuracy and the weighted F1 score were both 0.9282, which were slightly lower than those with empirical data. This observation is likely due to the simulated data having more sibling and second degree relationships and that more distant relationships usually have lower accuracies in testing. The confusion matrices of ASW, CHB, and MXL can be found in the <xref ref-type="sec" rid="s11">Supplementary Material C</xref>.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Confusion matrix of determining relationships given a Minor Allele Frequency (MAF &#x3d; 0.4) and Minimum Genetic Distance (MGD) &#x3d; 30&#xa0;cM, with 130 selected SNPs. The rows are the true relationships, and the columns are the predicted relationships.</p>
</caption>
<graphic xlink:href="fgene-16-1635734-g007.tif">
<alt-text content-type="machine-generated">Confusion matrix showing predicted versus true relationships. Categories include parent-child, sibling, second-degree, and unrelated. The diagonal contains the highest values: 1089, 941, 1774, and 1069, indicating accurate predictions. Other values represent misclassifications.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="sec" rid="s11">Supplementary Table S3</xref> gives more details of each of the false identifications in <xref ref-type="fig" rid="F7">Figure 7</xref>. 97.1% (i.e., 366/377) of the pairs were falsely identified with a first-to-second LR ratio greater than 100, and 82.5% (311/377) of ratios were less than 10. Like the empirical data, when MGD reduces to 10&#xa0;cM (<xref ref-type="sec" rid="s11">Supplementary Table S4</xref>), the accuracy substantially increases due to more SNPs being selected.</p>
<p>The ratio of the maximum and second maximum likelihoods for the CEU population with MGD &#x3d; 30&#xa0;cM and MAF &#x3d; 0.4 were plotted in <xref ref-type="fig" rid="F8">Figure 8</xref>. Like the results with empirical data, the parent-child and sibling relationships usually had several magnitudes higher likelihoods than the unrelated and second degree relationships. For parent-child and sibling, 226 out of 2,100 pairs (10.76%) had ratios less than 10, and 554 out of 2,100 pairs (26.38%) had ratios less than 100. For unrelated and second degree, 971 out of 3,150 pairs (30.86%) had the ratio less than 10, and 2,200 out of 3,150 pairs (70.48%) had ratios less than 100.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>The ratios of the maximum likelihood and the second maximum likelihood among the populations for each pair for Minimum Genetic Distance (MGD) &#x3d; 30&#xa0;cM and Minor Allele Frequency (MAF) &#x3d; 0.4. The circles and triangles represent the pairs with correct and incorrect identifications, respectively.</p>
</caption>
<graphic xlink:href="fgene-16-1635734-g008.tif">
<alt-text content-type="machine-generated">Scatter plot showing relationships classified by maximum likelihood (log scale) on the x-axis and maximum to second maximum likelihood ratio (log scale) on the y-axis. Data points include various relationships: unrelated, parent-child, sibling, and second-degree, indicated by different colored circles and triangles for correct and incorrect identifications. Black represents unrelated, blue is parent-child, green is sibling, and red is second-degree, with shapes indicating accuracy.</alt-text>
</graphic>
</fig>
<p>In addition, 82.49% (311/377) of the incorrectly identified pairs had ratios less than 10, 97.08% (366/377) of them were less than 100, and only 1 out of 377 had the ratio &#x3e;1,000. Thus, within unrelated and these tested close relationships, it is highly confident to conclude a relationship if the maximum likelihood of a relationship is 1,000 times greater than the second maximum likelihood. Otherwise, an inconclusive interpretation could be made.</p>
<p>The effect of MAF also was investigated with the simulation data. The peak accuracy was obtained with MAF &#x3d; 0.3, which is slightly different from the distributions generated from empirical data, likely due to the different allele frequencies in the gnomAD and the 1,000 genomes data. The accuracy difference across different MAFs may not be substantial, and the distribution is rather random due to the availability of the SNPs and allele frequencies used in analysis. For samples from real cases, the SNP profiles generated may have a much smaller number of SNPs with high quality. Thus, multiple MAFs between 0.3 and 0.4 may be tried to achieve the highest likelihood and/or LR.</p>
<p>It is worth noting that the gnomAD panel was used to filter the SNPs in the 1,000 genomes data in the simulation study. However, many SNPs in the gnomAD panel have lower MAF than 0.3 than in the 1,000 genomes data. Because of this difference, MAF &#x3d; 0.2 is in <xref ref-type="fig" rid="F9">Figure 9</xref>.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Distributions of accuracy and F1 score (weighted) with Minor Allele Frequency (MAF) given Minimum Genetic Distance (MGD) &#x3d; 30&#xa0;cM. The accuracy and F1 score (weighted) are almost identical due to the relatively even numbers of samples for each relationship.</p>
</caption>
<graphic xlink:href="fgene-16-1635734-g009.tif">
<alt-text content-type="machine-generated">Line graph showing Accuracy and F1 Score (weighted) plotted against Minimum Allele Frequency (MAF). Both metrics generally increase from 0.20 to 0.30, then fluctuate between 0.30 and 0.42, with a decline at 0.45. Data points indicate the number of SNPs, ranging from 112 to 130.</alt-text>
</graphic>
</fig>
<p>The effect of genotyping errors also was evaluated. LRs with various genotyping error rates in simulation and in LR calculation were calculated, and the accuracies and F1 Scores (weighted) were summarized in <xref ref-type="table" rid="T4">Table 4</xref> (the confusion matrices can be found in <xref ref-type="sec" rid="s11">Supplementary Material D</xref>). Apparently, lower error rate increases the accuracy. An error rate of 0.01 may still be acceptable for LR calculations (i.e., the accuracy is reduced from 0.9282 to 0.9198 with error rates increasing from 0.001 to 0.01). Error rate in LR calculations also is important. With 0.05 simulation error, LR calculations with the same error rate (i.e., 0.05) can substantially increase the relationship test accuracy (i.e., 0.8663 and 0.7907 with 0.05 and 0.01 LR calculation errors, respectively). In general, low genotyping error is extremely important for LR based relationship testing. It is necessary to filter out low quality variants with certain measures (such as read depth, variant quality scores, etc.) before calculating LRs.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>The accuracies and F1 scores (weighted) for various simulation genotyping error rates and LR calculation error rates.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Simulation error</th>
<th align="left">LR calculation error</th>
<th align="left">Accuracy</th>
<th align="left">F1 score (weighted)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">0.001</td>
<td align="left">0.001</td>
<td align="left">0.9282</td>
<td align="left">0.9282</td>
</tr>
<tr>
<td align="left">0.01</td>
<td align="left">0.001</td>
<td align="left">0.9110</td>
<td align="left">0.9114</td>
</tr>
<tr>
<td align="left">0.01</td>
<td align="left">0.01</td>
<td align="left">0.9198</td>
<td align="left">0.9198</td>
</tr>
<tr>
<td align="left">0.05</td>
<td align="left">0.001</td>
<td align="left">0.7907</td>
<td align="left">0.7805</td>
</tr>
<tr>
<td align="left">0.05</td>
<td align="left">0.05</td>
<td align="left">0.8663</td>
<td align="left">0.8660</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>CEU population, Minor Allele Frequency (MAF &#x3d; 0.4), and Minimum Genetic Distance (MGD) &#x3d; 30&#xa0;cM were used.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec id="s4">
<title>Discussion and conclusion</title>
<p>This paper describes the development and validation of a LR-based methodology designed to conduct close kinship relationship testing by dynamically selecting informative SNPs based on MAF and MGD from genome-wide SNP data. Our approach is unique in that it integrates dynamic SNP selection based on output data, rather than relying on a fixed, pre-selected SNP set. The described approach bridges modern genomic datasets with the traditional LR-based framework, allowing kinship testing laboratories to leverage comprehensive genomic information while maintaining established LR-based interpretive methodologies. KinSNP-LR, as implemented in this study, does not incorporate comprehensive population parameters, such as Fst or linkage between markers; these are features that can be considered in future iterations. This initial implementation is intended to demonstrate the core logic and workflow of dynamically integrating SNP selection with LR calculations for close kinship analysis. The system is evaluated for accuracy using empirical and simulated datasets.</p>
<p>However, the traditional LR framework assumes marker independence, necessitating the selection of only a limited subset of SNPs from the total typed data to minimize the confounding effects of linkage and LD<sup>31</sup>. The current implementation of the methodology supports parent-child, full-sibling, and second degree relationships, as well as distinguishing unrelated individuals. These close relationships are most commonly encountered in missing and unidentified persons cases handled by medical examiner offices, where the goal is to confirm identity rather than generate investigative leads. While more distant relationships could be incorporated in future iterations by including linked SNPs, such comparisons generally are less fruitful for identity confirmation due to the higher number of potential matches and reduced specificity.</p>
<p>This LR methodology assumes access to high-quality genotyping data. When genotypes are derived from low-quality or low-quantity DNA samples, rigorous quality control measures are necessary to ensure accurate relationship testing, which may include the use of metrics such as genotype quality scores, read depth, or other laboratory-specific thresholds for SNP quality. The default mutation rate in the methodology is set at 0.001, corresponding to the Q30 threshold, as genotyping errors are significantly more frequent than the actual biological mutation rate (&#x223c;10<sup>&#x2212;8</sup>). For datasets of lower quality, the mutation rate can be adjusted to higher values (e.g., 0.01) to account for increased error rates. In this study, the mutation rate (or better stated typing error rate) was performed only on parent-child relationships as is typically done in such cases. However, one could consider applying the typing error rate to all relationships tested herein since typing error will impact all sequence data.</p>
<p>Additionally, targeted panels may be considered, as they can be enriched to increase sensitivity of detection. The LR approach would remain similar, though the number of candidate SNPs will be fewer unless the DNA is highly degraded. For close relationships up to second degree, the number of SNPs may be sufficient, as only a limited number are needed. However, enrichment approaches still may suffer from differential detection of SNP states, and overall data quality will impact success. While the choice between targeted panels and WGS has historically been influenced by cost and laboratory preference, declining sequencing costs are making WGS increasingly attractive due to its broader genomic coverage and greater utility in distant kinship inference. This LR tool was developed to accommodate WGS data, but if one would like to use it with a targeted panel, the panel should be evaluated for its ability to include a maximum number of unlinked (or loosely linked) SNPs.</p>
<p>Substructure, which likely exists at some level, was not included in this iteration of the software because these SNPs likely exhibit low Fst values, particularly for each individual population. The impact of Fst on calculations involving high frequency SNP alleles should minimize their impact of substructure on LR calculations and support robust relationship testing across populations. Future versions of KinSNP-LR could explore the impact of a Fst correction based as described by <xref ref-type="bibr" rid="B3">Balding and Nichols (1994)</xref>.</p>
<p>The selection of an appropriate large set of candidate SNPs for relationship testing was derived from a preselected gnomAD panel containing 222,366 SNPs. These SNPs were filtered based on a MAF threshold of 0.3. Subsequently, the data were further subset to include SNPs located in genomic regions identified by Genome-in-a-Bottle as being easy to sequence or genotype. One might consider a much lower MAF, such 0.01, as rare alleles may increase the LR; however, such SNPs would not apply to all kinship comparisons as they are quite uncommon and have little value in most cases. A high MAF ensures that the SNPs from the large, preselected candidate panel of 222,366 SNPs would be applicable routinely. The high MAF threshold enhances the informativeness of selected SNPs, while reducing risk of unintended medical interpretations. This filtering strategy ensures a robust and unbiased dataset, providing a solid foundation for FGG as well as for other forensic applications, such as mixture interpretation. These selection steps provided SNPs that are amenable to validation studies performed herein, supporting reliability and reproducibility in forensic applications.</p>
<p>There have been reports by <xref ref-type="bibr" rid="B33">Mostad et al. (2023)</xref> and <xref ref-type="bibr" rid="B1">Andersen et al. (2025)</xref> that also provide approaches to use LRs with kinship analyses based on SNP data. They both have some positive features and limitations, as does KinSNP-LR. <xref ref-type="bibr" rid="B33">Mostad et al. (2023)</xref> use the Lander-Green algorithm, basically a Hidden Markov model (HMM), like the methods described by <xref ref-type="bibr" rid="B18">Epstein et al. (2000)</xref> and <xref ref-type="bibr" rid="B4">Boehnke and Cox (1997)</xref>. Their method can exploit larger amounts of SNP data with the assumption of linkage equilibrium and no genetic interference, in which the likelihood of individuals given a defined relationship is calculated using the allele frequencies, the recombination fractions between the markers, and transition probabilities of the IBD values of linked markers. Thus, instead of using a maximum of a few hundred SNPs with the assumption of little or no linkage between the markers, HMM-based methods in theory could use all available SNPs to increase the accuracy of the relationship tests. However, the <xref ref-type="bibr" rid="B33">Mostad et al. (2023)</xref> approach does not address mutation or genotyping error which could be a problem with low quality data, although mutation rates could be added to their model. The benefits of HMM may be limited for close relationships, as shown herein and the accuracy is already very high for close relationships with a few hundred SNPs. However, if LRs are desired for more distant relationships, HMM-based methods can be utilized in subsequent versions of KinSNP-LR. <xref ref-type="bibr" rid="B1">Andersen et al. (2025)</xref> used a preselected panel of 43 SNPs with relatively high MAF for LR calculations accounting for genotype errors for direct comparison of single source samples to persons of interest. Our approach is in some ways similar to that of <xref ref-type="bibr" rid="B1">Andersen et al. (2025)</xref> but extended to indirect kinship comparisons and makes use of more SNPs. KinSNP-LR provides a model for possible solutions to address LR/SNP-based calculations by dynamically selecting high MAF SNPs common to several population groups, MGD to account for linkage effects, incorporation of a basic mutation (or genotyping error) rate model, and being flexible on a case-by-case perspective.</p>
<p>In addition, including multiple reference family members instead of pairwise comparisons (i.e., joint probabilities), which are not described herein because the current genetic genealogy applications are mostly with pairwise relationship tests, can be considered in future studies. For cases with more than one family reference sample, a pedigree LR can be calculated capturing all genetic information from all the references. <xref ref-type="bibr" rid="B20">Ge et al. (2010)</xref>, based on the Elston-Stewart algorithm (<xref ref-type="bibr" rid="B17">Elston and Stewart, 1971</xref>), provide the methods for calculating a pedigree likelihood which incorporated genetic mutation, population substructure and accommodations for missing genetic data.</p>
<p>There are other ways to implement algorithms other than the one used herein which is a traditional, case-specific approach for LR calculation based on genotype combinations, where the formulae differ according to the genotype combination of the participants. For example, unified LR calculation formulae could be considered to address all genotype combinations with a single, more generalizable equation (<xref ref-type="bibr" rid="B16">Egeland, et al., 2017</xref>; <xref ref-type="bibr" rid="B30">Ma, et al., 2024</xref>). This would reduce reliance on complex code with numerous conditional statements which in turn potentially could reduce computational efficiency and increase risk of implementation errors, especially when dealing with high-density SNP data. KinSNP-LR is designed for close relationship testing for a specific targeted audience, such as medical examiners. The code for this software is only a few hundred lines. Therefore, these concerns that can be obviated to some degree with a unified LR calculation formulation are less likely an issue with KinSNP-LR which has been tested under the conditions shown in this study. In future iterations, where more distant and complex relationships would be addressed, the unified approach will be considered.</p>
<p>The validation results from both empirical and simulation data show high accuracies of close relationships, particularly for first degree relationships. One potential limitation is that the sample size of 50 pedigrees may be considered small. However, the data shown in <xref ref-type="fig" rid="F6">Figures 6</xref>&#x2013;<xref ref-type="fig" rid="F9">9</xref> and <xref ref-type="table" rid="T4">Table 4</xref> support that the sample size is sufficient to evaluate the performance of the software tool for the tested relationships. On another point, the ratios for selecting relationships with maximum and second maximum likelihoods differed for empirical data (&#x3e;100) and simulated data (&#x3e;1,000). Sampling is a potential explanation for the observed differences between empirical and simulation results. The empirical data have less samples, which tend to result in lower ratios overall. With more samples, as with the simulated data, larger ratios will be observed. These differences between the maximum and the second maximum likelihoods are merely observations and may not be construed as a recommended threshold(s) for casework. Threshold selection will be done at the laboratory level and based on the risk profile that a laboratory or jurisdiction defines operationally.</p>
<p>Comparisons with other kinship software were not conducted. The reasons were the input data format and requirements, or even assumptions and methods, are different for different software programs. To seamlessly integrate with WGS-based genetic genealogy application, KinSNP-LR can accept any VCF file without pre-defined panels, which is unique among all LR-based relationship testing software programs. Due to the different input data, assumptions, methods, and/or software implementations, different software programs may use different sets of variants, population features, and methods in relationship testing, which will lead to different LR results or even different conclusions but not provide insight into accuracy of any given software. Thus, the evaluation (e.g., accuracy) of any software program should be conducted with ground truth data, which is what was done in this study, instead of comparing with other software programs.</p>
<p>In addition, the WGS data generated for some forensic samples may yield 0.5X to 5X coverage. If, for example, DeepVariant (<xref ref-type="bibr" rid="B34">Poplin et al., 2018</xref>) were used to call variants with additional high-quality filters, such as Genotype Quality (GD) &#x2265; 30 (equivalent to 99.9% genotyping accuracy) and a minimum read depth &#x2265;5, such samples may yield limited data. However, for higher quality samples, typically &#x223c;1,600 to &#x223c;350,000 high quality variants could be detected for relationship testing. Even with further MAF and MGD filtering, there should be enough variants for close relationship tests, which usually only require dozens to a few hundred SNPs. On the other hand, relationship test software programs that require pre-selected panels may not work well for WGS data generated from typical forensic samples, as there may not be enough SNPs recovered with smaller pre-selected SNP panels after stringent quality filtering. A pre-selected panel may start with upwards of &#x223c;3,500 to &#x223c;10,000 compared with millions or tens of millions of SNPs by WGS.</p>
<p>In summary, the methodology presented in this study marks a significant step forward in bridging traditional forensic relationship testing with modern genomic technologies. By applying a LR-based framework with dynamically selected high-MAF SNPs, forensic laboratories can harness the data generated by WGS while maintaining compliance with accredited relationship testing standards. The use of unlinked markers and case-specific SNP selection distinguishes this method from existing LR-based kinship tools, making it especially applicable for FGG and other applications requiring statistically robust assessments of close relationships. This approach is particularly important for medical examiners, who are focused on confirming identity, which is a process that typically relies on comparisons of close relatives. In cases where STR testing is not possible due to degraded or limited DNA, this SNP-based LR approach offers a reliable and scientifically defensible alternative. As the field evolves, these advancements provide a strong foundation for broader adoption of relationship testing approaches that combine the rigor of traditional statistical models with the scalability of modern genomics.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s11">Supplementary Material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>JG: Software, Writing &#x2013; review and editing, Writing &#x2013; original draft, Conceptualization, Visualization, Formal Analysis, Data curation, Validation. BB: Conceptualization, Formal Analysis, Writing &#x2013; original draft, Investigation, Supervision, Methodology, Writing &#x2013; review and editing. MC: Writing &#x2013; review and editing, Data curation, Conceptualization, Writing &#x2013; original draft. KM: Writing &#x2013; original draft, Writing &#x2013; review and editing. DM: Writing &#x2013; original draft, Supervision, Writing &#x2013; review and editing, Conceptualization.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research and/or publication of this article.</p>
</sec>
<ack>
<p>The authors would like to especially thank Justin M. Zook and Michael D. Edge for their input and suggestions regarding this work.</p>
</ack>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>Authors JG, BB, MC, KM, and DM were employed by Othram Inc.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2025.1635734/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2025.1635734/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Supplementaryfile1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table1.xlsx" id="SM2" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andersen</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Kampmann</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Jepsen</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Morling</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Eriksen</surname>
<given-names>P. S.</given-names>
</name>
<name>
<surname>B&#xf8;rsting</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Shotgun DNA sequencing for human identification: dynamic SNP selection and likelihood ratio calculations accounting for errors</article-title>. <source>Forensic Sci. Int. Genet.</source> <volume>74</volume>, <fpage>103146</fpage>. <pub-id pub-id-type="doi">10.1016/j.fsigen.2024.103146</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ardlie</surname>
<given-names>K. G.</given-names>
</name>
<name>
<surname>Kruglyak</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Seielstad</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Patterns of linkage disequilibrium in the human genome</article-title>. <source>Nat. Rev. Genet.</source> <volume>3</volume> (<issue>4</issue>), <fpage>299</fpage>&#x2013;<lpage>309</lpage>. <pub-id pub-id-type="doi">10.1038/nrg777</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Balding</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Nichols</surname>
<given-names>R. A.</given-names>
</name>
</person-group> (<year>1994</year>). <article-title>DNA profile match probability calculation: how to allow for population stratification, relatedness, database selection and single bands</article-title>. <source>Forensic Sci. Int.</source> <volume>64</volume> (<issue>2-3</issue>), <fpage>125</fpage>&#x2013;<lpage>140</lpage>. <pub-id pub-id-type="doi">10.1016/0379-0738(94)90222-4</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boehnke</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cox</surname>
<given-names>N. J.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Accurate inference of relationships in sib-pair linkage studies</article-title>. <source>Am. J. Hum. Genet.</source> <volume>61</volume> (<issue>2</issue>), <fpage>423</fpage>&#x2013;<lpage>429</lpage>. <pub-id pub-id-type="doi">10.1086/514862</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Browning</surname>
<given-names>B. L.</given-names>
</name>
<name>
<surname>Browning</surname>
<given-names>S. R.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>A fast, powerful method for detecting identity by descent</article-title>. <source>Am. J. Hum. Genet.</source> <volume>88</volume> (<issue>2</issue>), <fpage>173</fpage>&#x2013;<lpage>182</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2011.01.010</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Browning</surname>
<given-names>B. L.</given-names>
</name>
<name>
<surname>Browning</surname>
<given-names>S. R.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Detecting identity by descent and estimating genotype error rates in sequence data</article-title>. <source>Am. J. Hum. Genet.</source> <volume>93</volume> (<issue>5</issue>), <fpage>840</fpage>&#x2013;<lpage>851</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2013.09.014</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Buckleton</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Triggs</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>The effect of linkage on the calculation of DNA match probabilities for siblings and half siblings</article-title>. <source>Forensic Sci. Int.</source> <volume>160</volume> (<issue>2-3</issue>), <fpage>193</fpage>&#x2013;<lpage>199</lpage>. <pub-id pub-id-type="doi">10.1016/j.forsciint.2005.10.004</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Budowle</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Baker</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Sajantila</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mittelman</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Mittelman</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Prioritizing privacy and presentation of supportable hypothesis testing in forensic genetic genealogy investigations</article-title>. <source>BioTechniques</source> <volume>76</volume> (<issue>9</issue>), <fpage>425</fpage>&#x2013;<lpage>431</lpage>. <pub-id pub-id-type="doi">10.1080/07366205.2024.2386218</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Budowle</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>van Daal</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Forensically relevant SNP classes</article-title>. <source>BioTechniques</source> <volume>44</volume> (<issue>5</issue>), <fpage>603</fpage>&#x2013;<lpage>608</lpage>. <pub-id pub-id-type="doi">10.2144/000112806</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Caballero</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Seidman</surname>
<given-names>D. N.</given-names>
</name>
<name>
<surname>Qiao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sannerud</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Dyer</surname>
<given-names>T. D.</given-names>
</name>
<name>
<surname>Lehman</surname>
<given-names>D. M.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Crossover interference and sex-specific genetic maps shape identical by descent sharing in close relatives</article-title>. <source>PLoS Genet.</source> <volume>15</volume> (<issue>12</issue>), <fpage>e1007979</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pgen.1007979</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Campbell</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Furlotte</surname>
<given-names>N. A.</given-names>
</name>
<name>
<surname>Eriksson</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hinds</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Auton</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Escape from crossover interference increases with maternal age</article-title>. <source>Nat. Comm.</source> <volume>6</volume> (<issue>1</issue>), <fpage>6260</fpage>. <pub-id pub-id-type="doi">10.1038/ncomms7260</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chakraborty</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Stivers</surname>
<given-names>D. N.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Budowle</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>The utility of short tandem repeat loci beyond human identification: implications for development of new DNA typing systems</article-title>. <source>Electrophoresis</source> <volume>20</volume>, <fpage>1682</fpage>&#x2013;<lpage>1696</lpage>. <pub-id pub-id-type="doi">10.1002/(SICI)1522-2683(19990101)20:8&#x3c;1682::AID-ELPS1682&#x3e;3.0.CO;2-Z</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Francioli</surname>
<given-names>L. C.</given-names>
</name>
<name>
<surname>Goodrich</surname>
<given-names>J. K.</given-names>
</name>
<name>
<surname>Collins</surname>
<given-names>R. L.</given-names>
</name>
<name>
<surname>Kanai</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>A genomic mutational constraint map using variation in 76,156 human genomes</article-title>. <source>Nature</source> <volume>625</volume> (<issue>7993</issue>), <fpage>92</fpage>&#x2013;<lpage>100</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-023-06045-0</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dowdeswell</surname>
<given-names>T. L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Forensic genetic genealogy project version December 2023</article-title>. <source>Mendeley Data</source>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://data.mendeley.com/datasets/cc5rh42mf9/1">https://data.mendeley.com/datasets/cc5rh42mf9/1</ext-link>.</comment>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dwarshuis</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kalra</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>McDaniel</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sanio</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Alvarez Jerez</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Jadhav</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>The GIAB genomic stratifications resource for human reference genomes</article-title>. <source>Nat. Comm.</source> <volume>15</volume> (<issue>1</issue>), <fpage>9029</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-024-53260-y</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Egeland</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Pinto</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Amorim</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Exact likelihood ratio calculations for pairwise cases</article-title>. <source>Forensic Sci. Int. Genet.</source> <volume>29</volume>, <fpage>218</fpage>&#x2013;<lpage>224</lpage>. <pub-id pub-id-type="doi">10.1016/j.fsigen.2017.04.018</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elston</surname>
<given-names>R. C.</given-names>
</name>
<name>
<surname>Stewart</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1971</year>). <article-title>A general model for the genetic analysis of pedigree data</article-title>. <source>Hum. Hered.</source> <volume>21</volume>, <fpage>523</fpage>&#x2013;<lpage>542</lpage>. <pub-id pub-id-type="doi">10.1159/000152448</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Epstein</surname>
<given-names>M. P.</given-names>
</name>
<name>
<surname>Duren</surname>
<given-names>W. L.</given-names>
</name>
<name>
<surname>Boehnke</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Improved inference of relationship for pairs of individuals</article-title>. <source>Am. J. Hum. Genet.</source> <volume>67</volume> (<issue>5</issue>), <fpage>1219</fpage>&#x2013;<lpage>1231</lpage>. <pub-id pub-id-type="doi">10.1016/S0002-9297(07)62952-8</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fedorova</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dutta</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Fedorov</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Atlas of cryptic genetic relatedness among 1000 human genomes</article-title>. <source>Genome Biol. Evol.</source> <volume>8</volume> (<issue>3</issue>), <fpage>777</fpage>&#x2013;<lpage>790</lpage>. <pub-id pub-id-type="doi">10.1093/gbe/evw034</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ge</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Budowle</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chakraborty</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>DNA identification by pedigree likelihood ratio accommodating population substructure and mutations</article-title>. <source>Investig. Genet.</source> <volume>1</volume>, <fpage>8</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1186/2041-2223-1-8</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ge</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chakraborty</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Eisenberg</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Budowle</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Comparisons of familial DNA database searching strategies</article-title>. <source>J. Forens. Sci.</source> <volume>56</volume> (<issue>6</issue>), <fpage>1448</fpage>&#x2013;<lpage>1456</lpage>. <pub-id pub-id-type="doi">10.1111/j.1556-4029.2011.01867.x</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Henn</surname>
<given-names>B. M.</given-names>
</name>
<name>
<surname>Hon</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Macpherson</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Eriksson</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Saxonov</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pe&#x27;er</surname>
<given-names>I.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Cryptic distant relatives are common in both isolated and cosmopolitan genetic samples</article-title>. <source>PLoS One</source> <volume>7</volume> (<issue>4</issue>), <fpage>e34267</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0034267</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huff</surname>
<given-names>C. D.</given-names>
</name>
<name>
<surname>Witherspoon</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Simonson</surname>
<given-names>T. S.</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Watkins</surname>
<given-names>W. S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Maximum-likelihood estimation of recent shared ancestry (ERSA)</article-title>. <source>Genome Res.</source> <volume>21</volume> (<issue>5</issue>), <fpage>768</fpage>&#x2013;<lpage>774</lpage>. <pub-id pub-id-type="doi">10.1101/gr.115972.110</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaplanis</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gordon</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shor</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Weissbrod</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Geiger</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wahl</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Quantitative analysis of population-scale family trees with millions of relatives</article-title>. <source>Science</source> <volume>360</volume> (<issue>6385</issue>), <fpage>171</fpage>&#x2013;<lpage>175</lpage>. <pub-id pub-id-type="doi">10.1126/science.aam9309</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kling</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>On the use of dense sets of SNP markers and their potential in relationship inference</article-title>. <source>Forensic Sci. Int. Genet.</source> <volume>39</volume>, <fpage>19</fpage>&#x2013;<lpage>31</lpage>. <pub-id pub-id-type="doi">10.1016/j.fsigen.2018.11.022</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kling</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Tillmar</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Forensic genealogy - a comparison of methods to infer distant relationships based on dense SNP data</article-title>. <source>Forensic Sci. Int. Genet.</source> <volume>42</volume>, <fpage>113</fpage>&#x2013;<lpage>124</lpage>. <pub-id pub-id-type="doi">10.1016/j.fsigen.2019.06.019</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Sacks</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>1954</year>). <article-title>The derivation of joint distribution and correlation between relatives by the use of stochastic matrices</article-title>. <source>Biometrics</source> <volume>10</volume>, <fpage>347</fpage>&#x2013;<lpage>360</lpage>. <pub-id pub-id-type="doi">10.2307/3001590</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Budowle</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ge</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Linkage and linkage disequilibrium among the markers in the forensic MPS panels</article-title>. <source>J. Forensic Sci.</source> <volume>66</volume> (<issue>5</issue>), <fpage>1637</fpage>&#x2013;<lpage>1646</lpage>. <pub-id pub-id-type="doi">10.1111/1556-4029.14724</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lipatov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sanjeevy</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Patroy</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Veeramah</surname>
<given-names>K. R.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Maximum likelihood estimation of biological relatedness from low coverage sequencing data</article-title>. <source>bioRxiv</source>. <pub-id pub-id-type="doi">10.1101/023374</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Cong</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>An approach to unified formulae for likelihood ratio calculation in pairwise kinship analysis</article-title>. <source>Front. Genet.</source> <volume>15</volume>, <fpage>1226228</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2024.1226228</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mandape</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Budowle</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Mittelman</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Mittelman</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Dense single nucleotide polymorphism testing revolutionizes scope and degree of certainty for source attribution in forensic investigations</article-title>. <source>Croat. Med. J.</source> <volume>65</volume> (<issue>3</issue>), <fpage>249</fpage>&#x2013;<lpage>260</lpage>. <pub-id pub-id-type="doi">10.3325/cmj.2024.65.249</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Morimoto</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Manabe</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Fujimoto</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hamano</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tamaki</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Discrimination of relationships with the same degree of kinship using chromosomal sharing patterns estimated from high-density SNPs</article-title>. <source>Forens. Sci. Int. Genet.</source> <volume>33</volume>, <fpage>10</fpage>&#x2013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1016/j.fsigen.2017.11.010</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mostad</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Tillmar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kling</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Improved computations for relationship inference using low-coverage sequencing data</article-title>. <source>BMC Bioinforma.</source> <volume>24</volume> (<issue>1</issue>), <fpage>90</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-023-05217-z</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Poplin</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>P. C.</given-names>
</name>
<name>
<surname>Alexander</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Schwartz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Colthurst</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ku</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>A universal SNP and small-indel variant caller using deep neural networks</article-title>. <source>Nat. Biotechnol.</source> <volume>36</volume> (<issue>10</issue>), <fpage>983</fpage>&#x2013;<lpage>987</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.4235</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ramstetter</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Shenoy</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Dyer</surname>
<given-names>T. D.</given-names>
</name>
<name>
<surname>Lehman</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Curran</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Duggirala</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Inferring identical-by-descent sharing of sample ancestors promotes high-resolution relative detection</article-title>. <source>Am. J. Hum. Genet.</source> <volume>103</volume> (<issue>1</issue>), <fpage>30</fpage>&#x2013;<lpage>44</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2018.05.008</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Seidman</surname>
<given-names>D. N.</given-names>
</name>
<name>
<surname>Shenoy</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Babu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Woods</surname>
<given-names>I. G.</given-names>
</name>
<name>
<surname>Dyer</surname>
<given-names>T. D.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Rapid, phase-free detection of long identity-by-descent segments enables effective relationship classification</article-title>. <source>Am. J. Hum. Genet.</source> <volume>106</volume> (<issue>4</issue>), <fpage>453</fpage>&#x2013;<lpage>466</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2020.02.012</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Staples</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Witherspoon</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Jorde</surname>
<given-names>L. B.</given-names>
</name>
<name>
<surname>Nickerson</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Below</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Huff</surname>
<given-names>C. D.</given-names>
</name>
<etal/>
</person-group>
<collab>University of Washington Center for Mendelian Genomics</collab> (<year>2016</year>). <article-title>PADRE: pedigree-aware distant-relationship estimation</article-title>. <source>Am. J. Hum. Genet.</source> <volume>99</volume> (<issue>1</issue>), <fpage>154</fpage>&#x2013;<lpage>162</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2016.05.020</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thompson</surname>
<given-names>E. A.</given-names>
</name>
</person-group> (<year>1975</year>). <article-title>The estimation of pairwise relationships</article-title>. <source>Ann. Hum. Genet.</source> <volume>39</volume> (<issue>2</issue>), <fpage>173</fpage>&#x2013;<lpage>188</lpage>. <pub-id pub-id-type="doi">10.1111/j.1469-1809.1975.tb00120.x</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yousefi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Abbassi-Daloii</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kraaijenbrink</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Vermaat</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mei</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>van &#x27;t Hof</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>A SNP panel for identification of DNA and RNA specimens</article-title>. <source>BMC Genomics</source> <volume>19</volume> (<issue>1</issue>), <fpage>90</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-018-4482-7</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>