<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1120312</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2023.1120312</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Evaluating the use of statistical and machine learning methods for estimating breed composition of purebred and crossbred animals in thirteen cattle breeds using genomic information</article-title>
<alt-title alt-title-type="left-running-head">Ryan et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2023.1120312">10.3389/fgene.2023.1120312</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Ryan</surname>
<given-names>C. A.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1492163/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Berry</surname>
<given-names>D. P.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/25252/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>O&#x2019;Brien</surname>
<given-names>A.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Pabiou</surname>
<given-names>T.</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1754401/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Purfield</surname>
<given-names>D. C.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/760881/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Teagasc</institution>, <addr-line>Co. Cork</addr-line>, <country>Ireland</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Munster Technological University</institution>, <addr-line>Cork</addr-line>, <country>Ireland</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Irish Cattle Breeding Federation</institution>, <addr-line>Cork</addr-line>, <country>Ireland</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/939157/overview">Francesco Tiezzi</ext-link>, University of Florence, Italy</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/799807/overview">Matthew L. Spangler</ext-link>, University of Nebraska-Lincoln, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/684942/overview">Luiz Brito</ext-link>, Purdue University, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/909009/overview">Hinayah Rojas De Oliveira</ext-link>, Purdue University, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: D. P. Berry, <email>Donagh.Berry@teagasc.ie</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>15</day>
<month>05</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1120312</elocation-id>
<history>
<date date-type="received">
<day>09</day>
<month>12</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>03</day>
<month>05</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Ryan, Berry, O&#x2019;Brien, Pabiou and Purfield.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Ryan, Berry, O&#x2019;Brien, Pabiou and Purfield</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>
<bold>Introduction:</bold> The ability to accurately predict breed composition using genomic information has many potential uses including increasing the accuracy of genetic evaluations, optimising mating plans and as a parameter for genotype quality control. The objective of the present study was to use a database of genotyped purebred and crossbred cattle to compare breed composition predictions using a freely available software, Admixture, with those from a single nucleotide polymorphism Best Linear Unbiased Prediction (SNP-BLUP) approach; a supplementary objective was to determine the accuracy and general robustness of low-density genotype panels for predicting breed composition.</p>
<p>
<bold>Methods:</bold> All animals had genotype information on 49,213 autosomal single nucleotide polymorphism (SNPs). Thirteen breeds were included in the analysis and 500 purebred animals per breed were used to establish the breed training populations. Accuracy of breed composition prediction was determined using a separate validation population of 3,146 verified purebred and 4,330 two and three-way crossbred cattle.</p>
<p>
<bold>Results:</bold> When all 49,213 autosomal SNPs were used for breed prediction, a minimal absolute mean difference of 0.04 between Admixture vs. SNP-BLUP breed predictions was evident. For crossbreds, the average absolute difference in breed prediction estimates generated using SNP-BLUP and Admixture was 0.068 with a root mean square error of 0.08. Breed predictions from low-density SNP panels were generated using both SNP-BLUP and Admixture and compared to breed prediction estimates using all 49,213 SNPs (representing the gold standard). Breed composition estimates of crossbreds required more SNPs than predicting the breed composition of purebreds. SNP-BLUP required &#x2265;3,000 SNPs to predict crossbred breed composition, but only 2,000 SNPs were required to predict purebred breed status. The absolute mean (standard deviation) difference across all panels &#x3c;2,000 SNPs was 0.091 (0.054) and 0.315 (0.316) when predicting the breed composition of all animals using Admixture and SNP-BLUP, respectively compared to the gold standard prediction.</p>
<p>
<bold>Discussion:</bold> Nevertheless, a negligible absolute mean (standard deviation) difference of 0.009 (0.123) in breed prediction existed between SNP-BLUP and Admixture once &#x2265;3,000 SNPs were considered, indicating that the prediction of breed composition could be readily integrated into SNP-BLUP pipelines used for genomic evaluations thereby avoiding the necessity for a stand-alone software.</p>
</abstract>
<kwd-group>
<kwd>genomic breed composition</kwd>
<kwd>cattle</kwd>
<kwd>crossbred</kwd>
<kwd>population assignment</kwd>
<kwd>low-density panels</kwd>
<kwd>best linear unbiased prediction</kwd>
<kwd>Admixture</kwd>
<kwd>genetic diversity</kwd>
</kwd-group>
<contract-sponsor id="cn001">Munster Technological University<named-content content-type="fundref-id">10.13039/501100022728</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Livestock Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>While genomic information in livestock breeding and management has predominately been used for parentage verification and discovery as well as genomic evaluations, it also has other potential applications such as the prediction of breed composition (<xref ref-type="bibr" rid="B21">Kuehn et al., 2011</xref>; <xref ref-type="bibr" rid="B29">Mcclure et al., 2017</xref>). In the absence of genomic information, the breed proportion of an animal is assumed to be simply the average breed composition of both parents (<xref ref-type="bibr" rid="B44">S&#xf6;lkner et al., 2010</xref>). However, breed composition of the offspring from a crossbred parent may deviate from expectation owing to parental recombination of chromosomes during gametogenesis. Genomic information should be more precise in predicting the breed composition of animals due to its capacity to determine the parental contribution (<xref ref-type="bibr" rid="B47">Strucken et al., 2017</xref>; <xref ref-type="bibr" rid="B23">Kumar et al., 2021</xref>) and therefore can help correct pedigree errors and estimate kinships when ancestry data are missing.</p>
<p>The ability to accurately predict the breed composition of an animal using genomic information has many potential uses. Firstly, genomic information can be used to verify that an animal is a purebred, thus preserving the integrity of the herd book. This may be of particular benefit for rare breeds where limited purebred breeding individuals exist or where no pedigree information is recorded and, therefore, it can help correct pedigree errors and estimate kinships when ancestry data are missing. Secondly, prediction of breed composition could assist in delivering consumer confidence in the authenticity of products from certain breeds which may command a higher market price (<xref ref-type="bibr" rid="B18">Judge et al., 2017</xref>; <xref ref-type="bibr" rid="B33">O&#x2019;Brien et al., 2020</xref>). Furthermore, where service providers genotype animals from multiple breeds, comparing breed composition estimates from genotype data against the expected breed composition for a given genotyping plate could curtail pedigree errors and act as a quality control measure by identifying mislabelled genotypes prior to their inclusion in downstream analyses (<xref ref-type="bibr" rid="B23">Kumar et al., 2021</xref>). While sex is a routine quality control step in the genotyping process, using breed composition prediction as an additional quality control measure may be particularly useful if a plate of exclusively male or female genotypes is mis-oriented. Additionally, the accurate determination of an animal&#x2019;s breed composition may improve the robustness of genetic evaluations where breed composition is frequently employed as an adjustment factor (<xref ref-type="bibr" rid="B50">Thomasen et al., 2013</xref>; <xref ref-type="bibr" rid="B30">McHugh et al., 2017</xref>), to correct for the differences in allele frequency and the relationship between SNPs and quantitative trait loci across breeds. Indeed, due to the mosaic nature of a crossbred animal&#x2019;s genome, <xref ref-type="bibr" rid="B43">Sevillano et al. (2017)</xref> confirmed that accounting for breed-specific SNP effects in admixed genomic evaluations outperformed genomic prediction models where the SNP effects were assumed to be the same across breeds. This suggests that the accurate determination of breed composition can enhance genomic predictions.</p>
<p>The SNP-BLUP method routinely used in genomic evaluations can also be used to predict breed composition, as proposed in sheep by <xref ref-type="bibr" rid="B33">O&#x2019;Brien et al. (2020)</xref>. By in large, the SNP-BLUP approach uses an infinitesimal model which assumes that the trait of interest is controlled by large number of SNPs, each of very small effect, fitted as random effects with a common variance structure. As SNP-BLUP is often used in genomic evaluations, the ability to exploit existing pipelines for predicting breed composition could be advantageous for quality control and be more computationally efficient than using a stand-alone software for breed prediction. Therefore, the objective of the present study was to use a large database of genotyped purebred and crossbred cattle to compare breed composition predictions using a freely available software, Admixture (<xref ref-type="bibr" rid="B1">Alexander et al., 2009</xref>), with those from SNP-BLUP. While statistical metrics and methods such as F<sub>st</sub> and PCA have been used to select informative SNPs to discriminate between cattle breeds (<xref ref-type="bibr" rid="B57">Wilkinson et al., 2011</xref>; <xref ref-type="bibr" rid="B17">Hulsegge et al., 2013</xref>), we wanted to determine the effectiveness of these methods in particular for identifying informative SNPs for predicting crossbred breed composition. Moreover, we aimed to compare the performance of these methods against other SNP selection approaches, including machine learning algorithms. Therefore, an additional objective was to determine the accuracy and general robustness of low-density genotype panels for predicting breed composition which was achieved by varying 1) the SNP density, and 2) the SNP selection strategy for alternative custom-derived low-density panels.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and methods</title>
<sec id="s2-1">
<title>Genotypic data</title>
<p>A total of 52,655 SNP were available from 703,078 dairy and beef cattle generated using a custom Illumina beadchip (IDBV3) which was developed to primarily increase the accuracy of genomic predictions whilest generating genotype information for mutations of interest (<xref ref-type="bibr" rid="B31">Mullen et al., 2013</xref>). All animals had a call rate &#x2265;90%. Only autosomal SNPs, SNPs with a known chromosome and position on the ARS UCD 1.2 genome build, and those with a call rate <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mo>&#x2265;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 90% were retained. SNPs were not filtered based on minor allele frequency to ensure that informative SNPs for distinguishing breeds with lower numbers were not omitted. Following all edits, 49,213 SNPs from 703,078 animals remained. Sporadically missing genotypes were imputed using FImpute V2.2 which uses an overlapping sliding window approach to efficiently exploit both family and population based information (<xref ref-type="bibr" rid="B41">Sargolzaei et al., 2014</xref>).</p>
</sec>
<sec id="s2-2">
<title>Establishment of purebred populations</title>
<p>Expected breed composition was available on all animals based on their recorded ancestry; 98,883 genotyped animals from 13 breeds were expected (based on ancestry) to be purebred. Breeds included were Angus, Aubrac, Blonde d&#x27;Aquitaine, Belgian Blue, Charolais, Friesian, Hereford, Holstein, Limousin, Parthenaise, Saler, Shorthorn, and Simmental. Using the available genotypes, a principal component analysis (PCA) based on a genomic relationship matrix was calculated using the approaches described by <xref ref-type="bibr" rid="B58">Yang et al. (2011)</xref> in the GCTA software package (<xref ref-type="bibr" rid="B58">Yang et al., 2011</xref>) to ensure animals were recorded correctly as being purebred. The 49,213 SNPs were pruned prior to PCA analysis by excluding one SNP from a pair of SNPs in strong linkage disequilibrium (pairwise squared correlation <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msup>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> &#x3e; 0.5) in a chromosomal window size of 50 SNPs, sliding the window 10 SNPs at a time as suggested by <xref ref-type="bibr" rid="B12">Dutheil (2020)</xref> to ensure that the resulting components were representative of the true underlying structure in the data and to reduce the risk of over-representation of certain regions of the genome; a total of 22,606 SNPs remained. Animals that deviated from their respective breed cluster in the PCA plot based on principal components 1, 2, and 3 were deemed to be incorrectly recorded as being purebred resulting in 11,210 animals being discarded.</p>
<p>Admixture V1.3 (<xref ref-type="bibr" rid="B1">Alexander et al., 2009</xref>) was also used to verify each animal&#x2019;s breed composition using the 22,606 pruned SNPs dataset as suggested by <xref ref-type="bibr" rid="B1">Alexander et al. (2009)</xref>. The pruned dataset was used solely to verify purebred status but the full SNP dataset was used for breed prediction analyses and SNP selection. An unsupervised analysis was initially performed to determine the most appropriate number of breed clusters (<italic>K</italic>) from 11 to 14. <italic>K</italic> &#x3d; 13 was the chosen number of breed clusters as it had the lowest cross-validation error; each of the 13 breeds separated into a distinct cluster. Individuals with a subsequent ancestry assignment of <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mo>&#x2265;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 90% attributed to one breed were retained as purebred-verified animals. The 44,802 purebred-verified animals were subsequently available to be stratified into three separate populations for analysis; 1) a training population, 2) a purebred validation population, and 3) a third purebred population which we will refer to as the SNP selection population; each population served a unique purpose described later. Given that some breeds had more purebred animals than other breeds, not all 44,802 purebred animals were used in the analysis; this was to ensure the number of animals selected per breed was relatively similar in order to minimise bias. A summary of the number of animals per breed within each of the three purebred populations is shown in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Number of animals per breed within the purebred training, validation and SNP selection populations.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Cattle breed</th>
<th align="center">Training</th>
<th align="center">Purebred validation</th>
<th align="center">SNP selection</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Angus</td>
<td align="center">500</td>
<td align="center">250</td>
<td align="center">1000</td>
</tr>
<tr>
<td align="left">Aubrac</td>
<td align="center">500</td>
<td align="center">250</td>
<td align="center">1000</td>
</tr>
<tr>
<td align="left">Blonde d&#x27;Aquitaine</td>
<td align="center">500</td>
<td align="center">250</td>
<td align="center">302</td>
</tr>
<tr>
<td align="left">Belgian Blue</td>
<td align="center">500</td>
<td align="center">129</td>
<td align="center">189</td>
</tr>
<tr>
<td align="left">Charolais</td>
<td align="center">500</td>
<td align="center">250</td>
<td align="center">1000</td>
</tr>
<tr>
<td align="left">Friesian</td>
<td align="center">500</td>
<td align="center">250</td>
<td align="center">249</td>
</tr>
<tr>
<td align="left">Hereford</td>
<td align="center">500</td>
<td align="center">250</td>
<td align="center">1000</td>
</tr>
<tr>
<td align="left">Holstein</td>
<td align="center">500</td>
<td align="center">250</td>
<td align="center">1000</td>
</tr>
<tr>
<td align="left">Limousin</td>
<td align="center">500</td>
<td align="center">250</td>
<td align="center">1000</td>
</tr>
<tr>
<td align="left">Parthenaise</td>
<td align="center">500</td>
<td align="center">73</td>
<td align="center">220</td>
</tr>
<tr>
<td align="left">Saler</td>
<td align="center">500</td>
<td align="center">250</td>
<td align="center">1000</td>
</tr>
<tr>
<td align="left">Shorthorn</td>
<td align="center">500</td>
<td align="center">250</td>
<td align="center">995</td>
</tr>
<tr>
<td align="left">Simmental</td>
<td align="center">500</td>
<td align="center">250</td>
<td align="center">1000</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s2-2-1">
<title>Purebred training population</title>
<p>Within breed identity-by-state (IBS) clustering was performed on all purebred animals in Plink V1.9 (<xref ref-type="bibr" rid="B38">Purcell et al., 2007</xref>), which investigates whether animals share zero, one, or two alleles at each locus across the genome. IBS clustering was used to identify the most genomically diverse animals within each breed to represent the training population. Within each breed, 500 clusters were created, and animals that had similar genomes were grouped together. One animal was randomly chosen from each cluster to represent the purebred training population for breed assignment. The purebred training population was established to calibrate models for predicting breed composition.</p>
</sec>
<sec id="s2-2-2">
<title>Purebred and admixed validation populations</title>
<p>To validate whether breed composition could be predicted using SNP data, a population of purebred and crossbred animals which had no direct relationship (i.e., parent-offspring and vice-versa) to the purebred training population was generated. Where possible, 250 purebred-verified animals from each of the 13 breeds were included in the validation population.</p>
<p>In order to identify a known admixed population for validating SNP-BLUP and Admixture breed composition predictions, a supervised Admixture analysis (<italic>K &#x3d; 13</italic>) was completed on all genotyped animals. The 500 purebred animals within each of the 13 breeds from the training population were fixed as purebred in a supervised Admixture analysis and the breed composition of all remaining admixed animals was predicted. Animals compromised of, at most 4 breeds were subsequently selected where each of the breeds represented had to belong to one of the 13 purebred populations included in the present study. In the two-way crosses, animals which had an Admixture breed composition prediction between 45%&#x2013;55%:45%&#x2013;55%, 20%&#x2013;30%:70%&#x2013;80% or 70%&#x2013;80%:20%&#x2013;30% were retained as a two-way validation population, consisting of 2,281 animals. Animals with an admixed breed composition comprised of <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mo>&#x2265;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 20% for each of three separate breeds and &#x3c;2.5% of a fourth breed were also included in a separate three-way cross validation population, consisting of 2,059 animals. A summary of the number of animals per breed in the crossbred validation population is in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Number of animals in the crossbred validation population.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Crosses</th>
<th align="center">Breed<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</th>
<th align="center">Number</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="5" align="center">2 Way Cross (<italic>n</italic> &#x3d; 2,281)</td>
<td align="center">AA &#xd7; CH</td>
<td align="center">998</td>
</tr>
<tr>
<td align="center">AA &#xd7; HE</td>
<td align="center">144</td>
</tr>
<tr>
<td align="center">AA &#xd7; SI</td>
<td align="center">311</td>
</tr>
<tr>
<td align="center">CH &#xd7; LM</td>
<td align="center">140</td>
</tr>
<tr>
<td align="center">HO &#xd7; FR</td>
<td align="center">233</td>
</tr>
<tr>
<td rowspan="5" align="center">3 Way Cross (<italic>n</italic> &#x3d; 2049)</td>
<td align="center">AA &#xd7; HO &#xd7; FR</td>
<td align="center">474</td>
</tr>
<tr>
<td align="center">AA &#xd7; BA &#xd7; LM</td>
<td align="center">180</td>
</tr>
<tr>
<td align="center">AU &#xd7; BA &#xd7; LM</td>
<td align="center">1280</td>
</tr>
<tr>
<td align="center">SI &#xd7; HO &#xd7; FR</td>
<td align="center">80</td>
</tr>
<tr>
<td align="center">SI &#xd7; SH &#xd7; CH</td>
<td align="center">55</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn1">
<label>
<sup>a</sup>
</label>
<p>AA, Angus; AU, Aubrac; BA, Blonde d&#x2019;Aquitaine; BB, Belgian Blue; CH, Charolais; FR, Friesian; HE, Hereford; HO, Holstein; LM, Limousin; SH, Shorthorn; SI, Simmental.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s2-2-3">
<title>Purebred SNP selection population</title>
<p>An additional purebred SNP selection population was established in order to quantify the information content of individual SNPs in predicting breed composition; this was necessary to rank the SNPs for the development of low-density panels. This SNP selection population consisted of 1,000 purebred animals per breed where possible that were not included in the purebred training or validation populations. The number of animals per breed included in the SNP selection population was capped at 1,000 where possible in order to keep a relatively similar number of animal per breed. This SNP selection population consisted of 9,955 purebred animals (<xref ref-type="table" rid="T1">Table 1</xref>).</p>
</sec>
</sec>
<sec id="s2-3">
<title>Divergence among breeds</title>
<p>The pairwise <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> statistic represents a measure of the genetic distance among breeds (<xref ref-type="bibr" rid="B54">Weir and Cockerham, 1984</xref>). The pairwise fixation indexes (<inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) were calculated for the SNP selection population in a supervised Admixture (<italic>K &#x3d; 13</italic>) analysis as:<disp-formula id="equ1">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msup>
<mml:mi>s</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>p</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>p</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:msup>
<mml:mi>s</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the standard deviation (SD) of the allele frequency among breeds and <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>p</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> is the mean allele frequency across breeds (<xref ref-type="bibr" rid="B54">Weir and Cockerham, 1984</xref>). A phylogenetic tree was computed using the breed pairwise <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> scores with the APE package in R software (<xref ref-type="bibr" rid="B35">Paradis et al., 2004</xref>) to visualise the genetic differentiation among all 13 breeds.</p>
</sec>
<sec id="s2-4">
<title>Breed composition estimated using single nucleotide polymorphisms best linear unbiased prediction</title>
<p>SNP-BLUP using MIX99 software (Mix99 Development <xref ref-type="bibr" rid="B49">Team, 2017</xref>) was used to estimate the breed composition of animals in the validation population, with the results compared to breed composition estimates from Admixture (<xref ref-type="bibr" rid="B1">Alexander et al., 2009</xref>). The SNP-BLUP approach followed the pipeline described by <xref ref-type="bibr" rid="B33">O&#x2019;Brien et al. (2020)</xref> for predicting breed composition in sheep using SNP genotypes. All SNPs were fitted as random effects which were assumed to be identically and independently distributed with mean zero and common variance structure N (0,<bold>I</bold> <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>):<disp-formula id="equ2">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>where the dependant variable <inline-formula id="inf11">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> was coded as either one if the animal was in the training population for the breed under investigation or zero if the animal was in the training population but not for the breed under investigation. The number of animals coded as purebred for each breed was equal to the number of animals coded as non-purebred for that breed. For example, 500 animals were classified as purebred Angus and coded as 1, while from the 12 remaining breeds, 500 animals were randomly selected such that each of the 12 breeds were equally represented. These 500 animals from the other 12 breeds were coded as 0, i.e., not Angus. All remaining animals were classified as missing. The intercept is denoted by &#xb5;, <inline-formula id="inf12">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the allele substitution effect of <inline-formula id="inf13">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>; <inline-formula id="inf14">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the random effect of the genotype of animal <inline-formula id="inf15">
<mml:math id="m17">
<mml:mrow>
<mml:mi>&#x2148;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> at locus j and <inline-formula id="inf16">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the random effect of residual term for animal <inline-formula id="inf17">
<mml:math id="m19">
<mml:mrow>
<mml:mi>&#x2148;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, with the common variance structure N (0,<bold>I</bold> <inline-formula id="inf18">
<mml:math id="m20">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>e</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>). The phenotypic SD for the dependent variable was estimated as <inline-formula id="inf19">
<mml:math id="m21">
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">q</mml:mi>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</inline-formula>, where p was the proportion of animals which were verified to be the breed under investigation (i.e., coded as 1); q was 1 minus this proportion. The genetic SD was estimated from the phenotypic SD assuming a heritability of 0.999 (<xref ref-type="bibr" rid="B33">O&#x2019;Brien et al., 2020</xref>). The SNP effects obtained were subsequently multiplied by the allele count of each animal to generate estimates of breed proportion.</p>
<p>All subsequent breed predictions &#x3c;0.05 were set to 0. The sum of all predicted breed compositions for each animal were rescaled as per <xref ref-type="bibr" rid="B33">O&#x2019;Brien et al. (2020)</xref>, where each animal&#x2019;s breed proportion estimated for the breed under investigation was divided by the sum of that animal&#x2019;s breed proportions estimated for all 13 breeds. Purebreds in the validation population were considered assigned if the prediction of breed composition was <inline-formula id="inf20">
<mml:math id="m22">
<mml:mrow>
<mml:mo>&#x2265;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.90 for any single breed. The SNP-BLUP approach was run for a series of different genotype panels constructed (described later) as well as the entire dataset (i.e., 49,213 SNPs).</p>
</sec>
<sec id="s2-5">
<title>Breed composition estimated using Admixture</title>
<p>Using the same training and validation populations and all 49,213 SNPs, a supervised analysis (<italic>K</italic> &#x3d; 13) was conducted in Admixture (<xref ref-type="bibr" rid="B1">Alexander et al., 2009</xref>). In the Admixture analysis, the same purebreds that were used in the SNP-BLUP analysis were set as purebreds for that breed, and the breed composition of the animals in the validation population was estimated. All breed proportion estimates <inline-formula id="inf21">
<mml:math id="m23">
<mml:mrow>
<mml:mo>&#x3c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.05 were fixed to 0 and the estimated breed proportions rescaled as with the SNP-BLUP method. Again, if the predicted breed proportion for any single breed in the purebred validation population was <inline-formula id="inf22">
<mml:math id="m24">
<mml:mrow>
<mml:mo>&#x2265;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.90, purebreds were regarded as being assigned to that breed.</p>
</sec>
<sec id="s2-6">
<title>Development of low-density genotype panels</title>
<p>Seven alternative low-density panels (i.e., 100, 500, 1,000, 2,000, 3,000, 5,000 and 7,500 SNPs) were generated using seven different SNP selection strategies. The SNP selection population (<xref ref-type="table" rid="T1">Table 1</xref>) was used to rank SNPs based on potential informativeness for the generation of these low-density panels. The number of SNP chosen per chromosome remained constant for each of the seven SNP selection methods evaluated and was proportional to the genome length of each chromosome (<xref ref-type="sec" rid="s11">Supplementary Table S1</xref>). The seven alternative methods used to generate the panels were as follows.</p>
<sec id="s2-6-1">
<title>Random selection</title>
<p>The number of predefined SNP required per chromosome was randomly selected until each of the respective panel densities was obtained.</p>
</sec>
<sec id="s2-6-2">
<title>Partitioning-around-medoids (PAM)</title>
<p>The partitioning-around-medoids (PAM) algorithm clusters SNPs on each chromosome together based on their proximity in genomic position, not taking LD into account. The algorithm was run for each chromosome separately with the number of clusters created per chromosome set to the number of predefined SNPs for that chromosome. The SNP located in the middle of each cluster was selected, as described by <xref ref-type="bibr" rid="B24">Lashmar et al. (2021)</xref> when developing low-density panels to assess imputation accuracy in cattle. The PAM algorithm was implemented in the R package <italic>&#x201c;cluster&#x201d;</italic> (V2.1.2 <xref ref-type="bibr" rid="B27">Maechler et al., 2021</xref>).</p>
</sec>
<sec id="s2-6-3">
<title>Fixation index (<inline-formula id="inf23">
<mml:math id="m25">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:math>
</inline-formula>
</title>
<p>The fixation index (<inline-formula id="inf24">
<mml:math id="m26">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) is used to evaluate the extent of genetic divergence between populations and identify genomic regions under selection pressure. The global <inline-formula id="inf25">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> was estimated using the method proposed by <xref ref-type="bibr" rid="B54">Weir and Cockerham (1984)</xref> across all 13 breeds in Plink V1.9 (<xref ref-type="bibr" rid="B38">Purcell et al., 2007</xref>) from the SNP selection population using all 49,213 SNPs. Three alternative strategies to picking SNPs based on the calculated <inline-formula id="inf26">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> statistic were investigated;<list list-type="simple">
<list-item>
<p>a) <inline-formula id="inf27">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and block method: Each chromosome was divided into blocks of SNPs with one SNP chosen per block. The number of blocks on each chromosome was equal to the number of predefined number of SNPs for that chromosome. The SNP with the highest <inline-formula id="inf28">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> statistic within each block was chosen.</p>
</list-item>
<list-item>
<p>b) <inline-formula id="inf29">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and PAM method: The SNP with the highest <inline-formula id="inf30">
<mml:math id="m32">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> within each PAM cluster already generated previously per chromosome was selected.</p>
</list-item>
<list-item>
<p>c) Highest ranking SNPs based on <inline-formula id="inf31">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> statistics: SNPs in the n<sup>th</sup> highest ranking for the <inline-formula id="inf32">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> statistic were chosen per chromosome, irrespective of location on that chromosome, where n was the number of predefined number of SNPs for that chromosome.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s2-6-4">
<title>PCA</title>
<p>SNP weightings were calculated using the &#x201c;smartpca&#x201d; algorithm in Eigensoft v7.2.1 (<xref ref-type="bibr" rid="B37">Patterson et al., 2006</xref>) applied to the SNP selection population. The greater the difference in allele frequency between populations, the greater the SNP weighting. Three alternative methods of picking SNPs based on PCA ranking were investigated similar to the <inline-formula id="inf33">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> approach already described;<list list-type="simple">
<list-item>
<p>a) PCA ranking and block method: The SNP with the highest SNP weighting within each block was chosen.</p>
</list-item>
<list-item>
<p>b) PCA ranking and PAM method: The SNP with the highest SNP weighting within each PAM cluster was selected.</p>
</list-item>
<list-item>
<p>c) Highest ranking SNPs based on PCA: SNPs in the n<sup>th</sup> highest ranking based on PCA SNP weightings were chosen per chromosome, irrespective of location on the chromosome, where n was the number of predefined number of SNPs for that chromosome.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s2-6-5">
<title>SNP-BLUP variance</title>
<p>SNP-BLUP was used to estimate the SNP effects within the SNP selection population of each breed individually using all 49,213 SNPs. From this, the standard deviation (SD) of the BLUP model solutions per SNP were estimated within the SNP selection population of all 13 breeds and SNPs were ranked based on the SD of the SNP effect across all 13 breeds; SNPs with a larger standard deviation were given a higher ranking. Three alternative methods of picking SNPs based on using the SNP-BLUP SD were investigated.<list list-type="simple">
<list-item>
<p>a) SNP-BLUP variance and block method: The SNP with the largest standard deviation of SNP effects within each block was chosen.</p>
</list-item>
<list-item>
<p>b) SNP-BLUP variance and PAM method: The SNP with the largest standard deviation of SNP effects within each PAM cluster was selected.</p>
</list-item>
<list-item>
<p>d) Highest ranking SNPs based on SNP-BLUP variance: SNPs in the n<sup>th</sup> highest ranking based on the standard deviation of SNP effects were chosen per chromosome, irrespective of location on the chromosome, where n was the number of predefined number of SNPs for that chromosome.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s2-6-6">
<title>Random Forest</title>
<p>Random Forest is a machine-learning method (<xref ref-type="bibr" rid="B5">Breiman, 2001</xref>) that employs decision trees, which are a set of rules for splitting data in a way that minimises variation. The Random Forest analysis was conducted in the R package random forest (<xref ref-type="bibr" rid="B26">Liaw and Wiener, 2001</xref>) using the genotypes of the SNP selection population to predict the dependant variable, which was breed, and was numbered 1 to 13. The built-in variable importance measures (VIM) ranked the SNPs according to their relevance for predicting breed. The highest ranking SNPs of a predefined number per chromosome were retained.</p>
</sec>
<sec id="s2-6-7">
<title>PLSDA</title>
<p>Partial least square discriminant analysis (PLSDA) is another machine learning method based on the PLS approach (<xref ref-type="bibr" rid="B2">Barker and Rayens, 2003</xref>). In the present study, a PLSDA regression model was constructed using the purebred SNP selection population and their corresponding genotypes in the R package Caret (<xref ref-type="bibr" rid="B22">Kuhn, 2020</xref>) for discriminative SNP selection. The dependant variable was breed, and was coded numerically as &#x2b;1 or &#x2212;1. If an animal was a member of the breed class under analysis, that animal was coded as &#x2b;1, which is referred to as the &#x2018;in-group&#x2019; and it it was a different breed group it was coded as &#x2212;1, representing the &#x2018;out-group&#x2019; (<xref ref-type="bibr" rid="B6">Brereton and Lloyd, 2014</xref>)<bold>.</bold> The regression model was run 13 times, once for each breed. Each SNP received a weighting, and SNPs which were the most informative for distinguishing between breed classes ranked highest. The highest ranking SNPs of a predefined number per chromosome were retained.</p>
</sec>
</sec>
<sec id="s2-7">
<title>Evaluating the difference in breed composition predictions using the low-density panels</title>
<p>Breed composition predictions from SNP-BLUP using all 49,213 SNPs were considered the gold standard and used for comparing the prediction performance from each of the low-density panels. Animals in the purebred validation population were considered to be accurately assigned when their estimated breed proportion of a specific breed was predicted to be &#x2265;0.90. The difference in the main breed proportion estimates for crossbred animals predicted using all the low-density panels and the gold standard 49,213 SNPs were compared. In addition, the three SNP selection methods with the smallest mean difference in breed composition predictions from the gold standard, were also used for breed composition prediction using Admixture (<xref ref-type="bibr" rid="B1">Alexander et al., 2009</xref>). The Admixture breed predictions using the low-density panels where then compared to those from the gold standard SNP-BLUP.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec id="s3-1">
<title>Population structure</title>
<p>The greatest genetic differentiation was observed between the Salers and both the Simmental and Shorthorns (<inline-formula id="inf34">
<mml:math id="m36">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.146) while the least genetic divergence existed between the Charolais and Blonde d&#x27;Aquitaine (<inline-formula id="inf35">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.039) (<xref ref-type="sec" rid="s11">Supplementary Table S2</xref>). The strong genetic relationship between Aubrac, Blonde d&#x27;Aquitaine, and Limousins was also demonstrated by their shared branch in the phylogenetic tree, with Simmentals situated on the neighbouring branch (<xref ref-type="fig" rid="F1">Figure 1</xref>). The PCA succesfully seperated out 13 breed clusters based on genomic data with the first, second and third principal components accounting for 22.1%, 15.7% and 13.6% of the variance, respectively. Within the PCA plot, Herefords were distinctly separated from other breeds, confirming their high <inline-formula id="inf36">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> value relative to other breeds (<xref ref-type="bibr" rid="B21">Kuehn et al., 2011</xref>; <xref ref-type="bibr" rid="B20">Kelleher et al., 2017</xref>). The close genetic relationship between Simmental, Blonde d&#x2019;Aquitaine, Aubrac and Limousin was again evident through the close proximity of their respective breed clusters (<xref ref-type="sec" rid="s11">Supplementary Figure S1</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>
<bold>(A)</bold> Phylogenetic tree showing the genetic distance between breeds based on pairwise fixation index (<inline-formula id="inf37">
<mml:math id="m39">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) estimates, <bold>(B)</bold> Population distribution of purebred animals across the first three principal components (PC1, PC2, PC3), <bold>(C)</bold> Admixture-estimated breed proportions for each purebred animal. Each animal is represented by a thin vertical line whose length represents its breed proportion, and each colour represents an inferred population. Breeds included Angus (AA), Aubrac (AU) Blonde d&#x27;Aquitaine (BA), Belgium Blue (BB), Charolais (CH), Friesian (FR), Hereford (HE), Holstein (HO), Limousin (LM), Parthenaise (PT), Saler (SA), Shorthorn (SH), and Simmental (SI).</p>
</caption>
<graphic xlink:href="fgene-14-1120312-g001.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>Breed composition prediction</title>
<p>The mean difference in predicted breed composition between SNP-BLUP and Admixture using all 49,213 SNPs was 0.04 across both purebred and crossbreds, which was not different (<italic>p</italic> &#x3e; 0.05) from zero, suggesting that there is no systematic difference between the methods that would lead to over or underestimation of breed composition. Both SNP-BLUP and Admixture accurately assigned <inline-formula id="inf38">
<mml:math id="m40">
<mml:mrow>
<mml:mo>&#x2265;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 98% of the purebred validation population to the correct breed. When comparing the prediction of breed composition of each breed individually, the largest difference observed in predictions between SNP-BLUP and Admixture for the purebreds was for the Belgian Blue (0.004) while no mean difference was detected for Angus, Aubrac, Charolais, Friesian, Hereford, Holstein, Limousin, Salers, Shorthorn and Simmental (<xref ref-type="table" rid="T3">Table 3</xref>). For both purebred and crossbreds in the validation population, the variability in predicted breed composition from SNP-BLUP and Admixture is shown in a Bland-Altman plot (<xref ref-type="fig" rid="F2">Figure 2</xref>). In comparison to purebred predictions, a larger absolute mean difference in predicted breed composition was observed in the crossbred validation population, with an average absolute mean difference of 0.08 and 0.05 for the two and three-way cross validation animals, respectively (<xref ref-type="table" rid="T4">Table 4</xref>). Ninety percent of the SNP-BLUP and Admixture breed composition predictions differed by less than 0.14. Of all the crossbred animals, the biggest discrepancy between SNP-BLUP and Admixture breed composition predictions was for Holstein-Friesian two-way cross animals.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Mean absolute difference and standard deviation of the difference of the absolute values between the SNP-BLUP and Admixture breed predictions for the purebred validation population in each breed.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Breed</th>
<th align="center">Mean difference</th>
<th align="center">Standard deviation</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Angus</td>
<td align="center">0.000</td>
<td align="center">0.000</td>
</tr>
<tr>
<td align="left">Aubrac</td>
<td align="center">0.000</td>
<td align="center">0.000</td>
</tr>
<tr>
<td align="left">Blonde d&#x27;Aquitaine</td>
<td align="center">0.003</td>
<td align="center">0.023</td>
</tr>
<tr>
<td align="left">Belgian Blue</td>
<td align="center">0.004</td>
<td align="center">0.037</td>
</tr>
<tr>
<td align="left">Charolais</td>
<td align="center">0.000</td>
<td align="center">0.000</td>
</tr>
<tr>
<td align="left">Friesian</td>
<td align="center">0.000</td>
<td align="center">0.007</td>
</tr>
<tr>
<td align="left">Hereford</td>
<td align="center">0.000</td>
<td align="center">0.000</td>
</tr>
<tr>
<td align="left">Holstein</td>
<td align="center">0.000</td>
<td align="center">0.000</td>
</tr>
<tr>
<td align="left">Limousin</td>
<td align="center">0.000</td>
<td align="center">0.000</td>
</tr>
<tr>
<td align="left">Parthenaise</td>
<td align="center">0.003</td>
<td align="center">0.021</td>
</tr>
<tr>
<td align="left">Saler</td>
<td align="center">0.000</td>
<td align="center">0.000</td>
</tr>
<tr>
<td align="left">Shorthorn</td>
<td align="center">0.000</td>
<td align="center">0.007</td>
</tr>
<tr>
<td align="left">Simmental</td>
<td align="center">0.000</td>
<td align="center">0.000</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Bland-Altman plot displaying the differences (y-axis) against the mean of the values (x-axis) of SNP-BLUP and Admixture breed proportions for two-way cross, three-way cross, and purebred animals. The horizontal red lines represent the mean <inline-formula id="inf39">
<mml:math id="m41">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 2 standard deviations and the horizontal blue line represents the mean difference between Admixture and SNP-BLUP prediction of breed composition.</p>
</caption>
<graphic xlink:href="fgene-14-1120312-g002.tif"/>
</fig>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Mean absolute difference and standard deviation of the difference of the absolute values between the SNP-BLUP and Admixture breed predictions for the crossbred validation population.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Breed</th>
<th align="center">Mean difference</th>
<th align="center">Standard deviation</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Angus</td>
<td align="center">0.062<xref ref-type="table-fn" rid="Tfn2">
<sup>a</sup>
</xref>
</td>
<td align="center">0.041</td>
</tr>
<tr>
<td align="left">Belgian Blue</td>
<td align="center">0.057</td>
<td align="center">0.067</td>
</tr>
<tr>
<td align="left">Charolais</td>
<td align="center">0.035<xref ref-type="table-fn" rid="Tfn2">
<sup>a</sup>
</xref>
</td>
<td align="center">0.031</td>
</tr>
<tr>
<td align="left">Friesian</td>
<td align="center">0.158<xref ref-type="table-fn" rid="Tfn2">
<sup>a</sup>
</xref>
</td>
<td align="center">0.111</td>
</tr>
<tr>
<td align="left">Hereford</td>
<td align="center">0.046<xref ref-type="table-fn" rid="Tfn2">
<sup>a</sup>
</xref>
</td>
<td align="center">0.032</td>
</tr>
<tr>
<td align="left">Holstein</td>
<td align="center">0.155<xref ref-type="table-fn" rid="Tfn2">
<sup>a</sup>
</xref>
</td>
<td align="center">0.068</td>
</tr>
<tr>
<td align="left">Limousin</td>
<td align="center">0.031<xref ref-type="table-fn" rid="Tfn2">
<sup>a</sup>
</xref>
</td>
<td align="center">0.042</td>
</tr>
<tr>
<td align="left">Shorthorn</td>
<td align="center">0.024</td>
<td align="center">0.023</td>
</tr>
<tr>
<td align="left">Simmental</td>
<td align="center">0.051<xref ref-type="table-fn" rid="Tfn2">
<sup>a</sup>
</xref>
</td>
<td align="center">0.039</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn2">
<label>
<sup>a</sup>
</label>
<p>Difference is significantly different from zero.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-3">
<title>Low-density panel predictions of breed composition with SNP-BLUP</title>
<sec id="s3-3-1">
<title>Purebred predictions</title>
<p>In general, the number of correctly assigned purebreds increased with increasing panel density across all SNP selection strategies (<xref ref-type="fig" rid="F3">Figure 3</xref>). All SNP selection strategies correctly assigned &#x3e;85% of the purebred validation population when the SNP density was &#x2265;2,000 SNPs, with the exception of the PLSDA and Random selection method, which both required a minimum of 3,000 SNPs to correctly assign &#x3e;85% of the purebred validation population to their respective breeds (<xref ref-type="fig" rid="F3">Figure 3</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>
<bold>(A)</bold> The percentage of animals in the purebred validation population correctly assigned to the respective breed (i.e., predicted to have a breed proportion &#x3e;0.9 for their respective breed). <bold>(B)</bold> The percentage difference between the gold standard (estimates using all 49,213 SNPs) and low-density breed proportion estimates for crossbreds. SNP selection methods for the creation of low-density panels included pairwise fixation index highest (Fst highest), partitioning-around-medoids (PAM), principal component analysis highest (PCA highest), partial least square discriminant analysis (PLSDA), random SNP selection (Random), Random Forest, and SNP-BLUP variance highest.</p>
</caption>
<graphic xlink:href="fgene-14-1120312-g003.tif"/>
</fig>
</sec>
<sec id="s3-3-2">
<title>Crossbred predictions</title>
<p>Similarly, as panel density increased the mean difference between the gold standard breed composition estimates and breed prediction estimates using the low-density panels reduced for crossbreds (<xref ref-type="fig" rid="F3">Figure 3</xref>). The estimation of crossbred breed composition was more challenging than that of purebreds, requiring a minimum of 3,000 SNPs for accurate crossbred breed composition predictions across the different SNP selection strategies, regardless of whether they were two or three-way crosses (<xref ref-type="fig" rid="F3">Figure 3</xref>). When panel density was <inline-formula id="inf40">
<mml:math id="m42">
<mml:mrow>
<mml:mo>&#x2265;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 3,000 SNPs, breed composition estimates deviated from the gold standard by an average of 0.055 and 0.079 for two and three-way crosses, respectively.</p>
</sec>
</sec>
<sec id="s3-4">
<title>Comparison of SNP selection strategies</title>
<p>There was little overlap in the actual SNPs selected by each SNP selection strategy (<xref ref-type="sec" rid="s11">Supplementary Figure S2</xref>). There was a minimal difference in performance between the SNPs selected using the various SNP selection methods for predicting breed composition at panel densities <inline-formula id="inf41">
<mml:math id="m43">
<mml:mrow>
<mml:mo>&#x2265;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 3,000 SNPs. At panel densities &#x3c;3,000 SNPs, SNPs selected using the <inline-formula id="inf42">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> method most accurately predicted breed composition, followed by the PCA selection strategy (<xref ref-type="fig" rid="F3">Figure 3</xref>). Interestingly, when the genomic position of the SNP was considered in the <inline-formula id="inf43">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and PCA SNP selection methods (i.e., the block and PAM method), breed composition estimates were considerably less accurate than the <inline-formula id="inf44">
<mml:math id="m46">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and PCA SNP selection method where position was not taken into account (<xref ref-type="fig" rid="F3">Figure 3</xref>). When comparing machine learning methods across densities, in general, SNPs selected using Random Forest were better at predicting the breed composition of both purebreds and crossbreds than SNPs selected using PLSDA (<xref ref-type="fig" rid="F3">Figure 3</xref>).</p>
</sec>
<sec id="s3-5">
<title>Comparison with Admixture predictions</title>
<p>When 49,213 SNPs were used, there was no systematic difference between the breed composition predictions by SNP-BLUP <italic>versus</italic> Admixture. SNPs selected using the three most accurate SNP selection methods (i.e., <inline-formula id="inf45">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> highest, PCA highest, and PAM) for the creation of low-density panels were also used for predicting breed composition in Admixture (<xref ref-type="bibr" rid="B1">Alexander et al., 2009</xref>). Admixture proved to be more accurate at predicting breed composition than SNP-BLUP when panel density was &#x3c;2,000 SNPs. Breed composition estimated from Admixture had an absolute mean difference of 0.091 from the gold standard SNP-BLUP breed composition predictions, whereas estimates from SNP-BLUP had an absolute mean difference of 0.315 from the gold standard estimates. Admixture required fewer SNPs than SNP-BLUP to accurately predict breed composition, with 500 and 1,000 SNPs sufficing to accurately predict the breed composition of purebred and crossbred cattle, respectively, whereas SNP-BLUP required 2,000 and 3,000 SNPs (<xref ref-type="fig" rid="F4">Figure 4</xref>). Across both SNP-BLUP and Admixture, SNPs selected using the <inline-formula id="inf46">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> highest SNP selection method generally preformed best at predicting breed composition.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>
<bold>(A)</bold> Purebred breed proportion estimates from Admixture and SNP-BLUP using low-density panels, <bold>(B)</bold> The percentage difference between the gold standard (estimates using all 49,213 SNPs) and low-density breed proportion estimates from Admixture and SNP-BLUP. SNP selection methods for the creation of low-density panels included pairwise fixation index highest (Fst), partitioning-around-medoids (PAM), and principal component analysis highest (PCA).</p>
</caption>
<graphic xlink:href="fgene-14-1120312-g004.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>The objective of the present study was to compare SNP-BLUP and Admixture as methods to predict the breed composition of purebred and crossbred cattle; of particular interest also was to investigate if the accuracy of predicting breed composition was eroded as SNP density reduced, and also if the approach to select these SNP impacted the conclusion. Marginal differences existed between both breed prediction methods once genotypes from &#x3e;2,000 informative SNPs were available on all animals. Moreover, once animals were genotyped for &#x3e;3,000 SNPs (which is generally the norm in cattle), how these SNPs were selected did not impact greatly the predictions.</p>
<sec id="s4-1">
<title>SNP-BLUP vs. Admixture</title>
<p>The only discrepancies observed between SNP-BLUP and Admixture breed composition predictions for purebreds using the full SNP dataset was for Blonde d&#x27;Aquitaine, Belgian Blue, and Parthenaise. Some animals in these breeds were misassigned to another closely related breed; this phenomenon may be due to the relatively smaller sample sizes of these breeds in the validation population (i.e., sampling variability) as well as their close genetic resemblance to other breeds. The largest discrepancy in prediction of breed composition detected between Admixture and SNP-BLUP predictions for crossbred animals was for the two-way Holstein-Friesian crosses (<xref ref-type="fig" rid="F2">Figure 2</xref>); because these two breeds are genetically similar, differentiating which portion of the genome is attributed to Holstein and which attributed to Friesian was challenging. Because the Holstein and Friesian compositions in a three-way cross animal represented a smaller proportion of the animal&#x2019;s overall breed composition, lesser differences between Admixture and SNP-BLUP predictions were evident in three-way crosses with a Holstein-Friesian component than in two-way crosses comprised of exclusively Holstein and Friesian. It should also be noted that only animals with a breed composition made up of at most four breeds were included in the crossbred validation population. The rationale behind this was that animals with more than four breeds in their genome might have experienced degraded breed haplotypes inherited from their ancestors over time, which would likely make predicting their breed composition particularly challenging.</p>
<p>In order to assess how mislabeled individuals in the training population affect breed composition predictions using SNP-BLUP and Admixture, 250 purebred Angus animals were substituted with 250 two-way cross Angus animals in the training population; they were all labelled as purebred Angus. All 250 purebred Angus animals in the validation population were predicted by Admixture to be &#x2265;0.90 Angus. In contrast, when predictions were based on SNP-BLUP, only six of the 250 purebred Angus animals in the validation population were predicted to be &#x2265;0.60 Angus. Therefore, SNP-BLUP appears to be more sensitive to mislabeled individuals in the training population, while Admixture was able to accurately estimate breed composition even with 250 (i.e., half the population) mislabeled individuals present.</p>
</sec>
<sec id="s4-2">
<title>Low-density panels</title>
<p>SNP-BLUP required a minimum of 2,000 and 3,000 SNPs to accurately predict purebred and crossbred breed composition, respectively; the respective values for Admixture was 500 and 1,000 SNPs, corroborating results of <xref ref-type="bibr" rid="B4">Bj&#xf8;rnstad and R&#xf8;ed (2002)</xref> who concluded that assigning crossbred (horses) to the correct breed using the frequency method outlined by <xref ref-type="bibr" rid="B34">Paetkau et al. (1995)</xref> is more challenging than that of purebreds. It should be noted that in the present study, a separate population was used to select SNPs for the creation of the low density panels so as to minimise SNP selection bias. <xref ref-type="bibr" rid="B47">Strucken et al. (2017)</xref> emphasised the importance of utilising a separate population for the selection of SNPs for predicting breed composition, reporting that when the prediction equations were not generated from a population independent of the test dataset, it resulted in a substantial increase in ascertainment bias.</p>
<p>Many factors impact the number of SNPs required for the accurate prediction of breed composition. These factors include, but are not limited to, the breeds included in the study (<xref ref-type="bibr" rid="B25">Lewis et al., 2011</xref>; <xref ref-type="bibr" rid="B17">Hulsegge et al., 2013</xref>), given that breeds which are closely related are likely to have similar allele frequencies and therefore be more difficult to differentiate than breeds which are not genetically similar (<xref ref-type="bibr" rid="B57">Wilkinson et al., 2011</xref>; <xref ref-type="bibr" rid="B19">Kavakiotis et al., 2015</xref>). In addition, the effective population size of the population also contributes to the number of SNPs needed to accurately determine breed composition as populations with larger effective population sizes are genetically more diverse. The effective population of the breeds in the present study were previously estimated by McParland et al. (2007) to range between 64 and 127 per breed. Similar estimates in cattle have been reported elsewhere (<xref ref-type="bibr" rid="B45">Stachowicz et al., 2011</xref>; <xref ref-type="bibr" rid="B40">Rodr&#xed;guez-Ramilo et al., 2015</xref>; <xref ref-type="bibr" rid="B11">Doekes et al., 2018</xref>) and are a reflection of the intense selection and genetic drift breeds have been subjected to. SNP selection methods therefore that choose the most informative SNPs for breed prediction require fewer SNPs for accurate breed composition predictions than using less or non-informative SNPs (<xref ref-type="bibr" rid="B10">Ding et al., 2011</xref>; <xref ref-type="bibr" rid="B7">Chhotaray et al., 2019</xref>) such as using the random SNP selection method, as demonstrated in the present study. The number of SNPs necessary for accurate breed composition predictions also depends on whether Admixture or a regression model such as SNP-BLUP is used for predictions, with Admixture requiring fewer SNPs than regression models (<xref ref-type="bibr" rid="B47">Strucken et al., 2017</xref>; <xref ref-type="bibr" rid="B15">He et al., 2018</xref>; <xref ref-type="bibr" rid="B39">Reverter et al., 2020</xref>).</p>
</sec>
<sec id="s4-3">
<title>SNP selection methods for low-density panels</title>
<p>The success of the F<sub>st</sub> SNP selection method for identifying informative SNPs which can be used for the prediction of breed composition in cattle has been extensively reported previously (<xref ref-type="bibr" rid="B25">Lewis et al., 2011</xref>; <xref ref-type="bibr" rid="B57">Wilkinson et al., 2011</xref>; <xref ref-type="bibr" rid="B17">Hulsegge et al., 2013</xref>; <xref ref-type="bibr" rid="B3">Bertolini et al., 2015</xref>), as has the PCA SNP selection method (<xref ref-type="bibr" rid="B36">Paschou et al., 2007</xref>; <xref ref-type="bibr" rid="B25">Lewis et al., 2011</xref>; <xref ref-type="bibr" rid="B3">Bertolini et al., 2015</xref>; <xref ref-type="bibr" rid="B7">Chhotaray et al., 2019</xref>). Unlike some previous studies (<xref ref-type="bibr" rid="B10">Ding et al., 2011</xref>; <xref ref-type="bibr" rid="B17">Hulsegge et al., 2013</xref>; <xref ref-type="bibr" rid="B47">Strucken et al., 2017</xref>), a linkage disequilibrium (LD) threshold or minimum distance between selected SNPs was not implemented when creating low-density panels in this study as no prior assumptions were made about which SNPs may or may not be informative for breed prediction. As a result, the F<sub>st</sub> and PCA highest methods both chose SNPs located in close proximity and consequently in strong LD on each autosome, particularly when panel density was <inline-formula id="inf47">
<mml:math id="m49">
<mml:mrow>
<mml:mo>&#x2264;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 1,000 SNPs (<xref ref-type="sec" rid="s11">Supplementary Figure S3</xref>). This was not surprising, as previous literature also reported the PCA (<xref ref-type="bibr" rid="B36">Paschou et al., 2007</xref>; <xref ref-type="bibr" rid="B25">Lewis et al., 2011</xref>; <xref ref-type="bibr" rid="B3">Bertolini et al., 2015</xref>) and F<sub>st</sub> method (<xref ref-type="bibr" rid="B56">Wilkinson et al., 2012</xref>) of ranking SNPs to be susceptible to choosing SNPs in strong LD with each other. Despite the strong LD observed between the SNPs chosen by the PCA and F<sub>st</sub> highest methods, these SNP selection strategies performed better at predicting breed composition than SNPs selected using the other SNP selection methods evaluated, all of which had weaker LD among SNPs. This suggests that informative SNPs for the prediction of breeds may be in LD and cluster together in close proximity along the genome, and the benefit of increasing SNP panel density was less with the PCA and F<sub>st</sub> highest methods in comparison to the other methods that had weaker LD among SNPs. <xref ref-type="bibr" rid="B56">Wilkinson et al. (2012)</xref> noticed a similar trend, and deduced that a strong level of LD when designing low-density panels could be a signature reflecting positive selection as result of modern breeding programmes, and that these SNPs may show strong breed differentiation due to positive selection for breed-specific characteristics. Consequently, despite recommendations to remove SNPs in LD prior to Admixture or PCA analysis (<xref ref-type="bibr" rid="B32">Novembre et al., 2008</xref>; <xref ref-type="bibr" rid="B1">Alexander et al., 2009</xref>; <xref ref-type="bibr" rid="B28">Mattucci et al., 2019</xref>; <xref ref-type="bibr" rid="B12">Dutheil, 2020</xref>), SNPs in LD could potentially be highly informative for breed composition prediction, particularly when SNP density was low (<xref ref-type="bibr" rid="B56">Wilkinson et al., 2012</xref>).</p>
<p>Although machine learning algorithms have been widely applied to cattle breeding for the prediction of a wide variety of traits such as lameness (<xref ref-type="bibr" rid="B53">Warner et al., 2020</xref>), longevity (<xref ref-type="bibr" rid="B51">Van Der Heide et al., 2019</xref>) and milk composition (<xref ref-type="bibr" rid="B14">Gianola et al., 2011</xref>; <xref ref-type="bibr" rid="B13">Frizzarin et al., 2021</xref>), these algorithms have not been utilized extensively in predicting breed composition in cattle. Prior research has reported that machine learning does not predict certain traits in cattle and sheep as effectively as other traditional methods such as regression models (<xref ref-type="bibr" rid="B8">Cortez et al., 2006</xref>; <xref ref-type="bibr" rid="B52">Van Hertem et al., 2014</xref>; <xref ref-type="bibr" rid="B16">Hempstalk et al., 2015</xref>), corroborating the findings of the present study. Previous literature has reported that the majority of PLSDA models suffer from overfitting (<xref ref-type="bibr" rid="B55">Westerhuis et al., 2008</xref>) and inconsistent performance (<xref ref-type="bibr" rid="B48">Szyma&#x144;ska et al., 2012</xref>). While <xref ref-type="bibr" rid="B3">Bertolini et al. (2015)</xref> successfully used Random Forest in conjunction with PCA for SNP selection and breed assignment in cattle, the accuracy of this method was based on the percentage of animals assigned to the correct breed, whereas accuracy in the present study was based on the more difficult task of assigning breed proportions and predicting the overall breed composition of individual cattle. Another key difference between the present study and that of <xref ref-type="bibr" rid="B3">Bertolini et al. (2015)</xref> is that the present study only used Random Forest for SNP selection and used SNP-BLUP and Admixture for breed proportion predictions, whereas <xref ref-type="bibr" rid="B3">Bertolini et al. (2015)</xref> used Random Forest to select informative SNPs, before fitting a new Random Forest algorithm to determine breed assignment.</p>
<p>The little overlap in SNPs selected across SNP selection approaches is likely due to the difference between the SNP selection methods used. Out of all SNP selection methods investigated, Random Forest was the only one that considered possible correlations among SNPs. F<sub>st</sub>-based selection focused on the standardized variance in allele frequency among populations, while PCA-based selection focused on patterns in the data, identifying SNPs that had high loadings on the first three principal components, which captured the most significant patterns in the data. On the other hand, the PAM method only considered the genomic position of the SNP. <xref ref-type="bibr" rid="B42">Schiavo et al. (2020)</xref> also reported little overlap in SNPs selected when comparing SNPs selected using the F<sub>st</sub>, PCA and Random Forest methods.</p>
</sec>
<sec id="s4-4">
<title>Training population</title>
<p>It should be noted that ensuring purebred animals are recorded correctly and a careful selection of the most genetically diverse animals within breed to be included in the training population is crucial. As suggested by others (<xref ref-type="bibr" rid="B4">Bj&#xf8;rnstad and R&#xf8;ed, 2002</xref>; <xref ref-type="bibr" rid="B9">Dalvit et al., 2008</xref>), when predicting breeds, some animals may never be correctly assigned regardless of the number of SNPs used because the breeds are too genetically similar or because the individuals are genetically atypical for their breeds. To avoid the latter from happening, and ensure maximum prediction accuracy, the training population in the present study consisted of very large numbers of animals in comparison to previous similar studies (<xref ref-type="bibr" rid="B25">Lewis et al., 2011</xref>; <xref ref-type="bibr" rid="B57">Wilkinson et al., 2011</xref>), increasing the within-breed variability. <xref ref-type="bibr" rid="B3">Bertolini et al. (2015)</xref> advocated that the more animals included in the training population the greater the within breed variability captured, possibly resulting in an enhanced performance for breed prediction. While previous studies randomly selected purebreds to represent their training population, <xref ref-type="bibr" rid="B17">Hulsegge et al. (2013)</xref> noted that for the optimum prediction of breed composition, when selecting the training population, it is crucial to choose the most genetically diverse animals within each breed. Bearing this in mind, a novel approach was implemented in the present study, utilising IBS clustering to aid with selecting the most genetically dissimilar animals to represent the training population for each breed. IBS clustering compares two individuals which share 0, 1 or 2 alleles at a given locus throughout the genome (<xref ref-type="bibr" rid="B46">Stevens et al., 2011</xref>), grouping genetically similar animals together. Randomly selecting one animal from each genetically different IBS cluster to represent the training population ensured that the training population captured the majority of the variation of genotypes in each breed.</p>
</sec>
<sec id="s4-5">
<title>Applications</title>
<p>The present study demonstrated that genomic information can be utilised in generating accurate predictions of breed composition which could potentially be useful for increasing the accuracy of genetic evaluations by being better able to fit breed covariates in an admixed population. <xref ref-type="bibr" rid="B43">Sevillano et al. (2017)</xref> confirmed the superior performance of genomic evaluation models that account for breed-specific SNP effects in admixed populations compared to those assuming uniform SNP effects across breeds. This suggests that the accurate determination of breed composition can enhance genomic predictions. Furthermore, accurate breed composition information could also potentially be utilised in quality control of genotypes entering the database and to further augment various breeding strategies for improvement of cattle breeds.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>Conclusion</title>
<p>There was a strong similarity in predicted breed composition per animal between the SNP-BLUP and Admixture approaches investigated when panel density was &#x2265;3,000 SNPs. This suggests that the prediction of breed composition could be readily integrated into the SNP-BLUP pipelines used for genomic evaluations thus replacing the use of a stand-alone software. Despite approximately 50,000 SNPs existing on most routinely-used genotyping panels, only small subsets of highly informative SNPs are required to accurately predict breed composition. This study provides a blueprint for the utilisation of the readily available next-generation sequencing technologies in the prediction of breed composition, by offering possible methods for how to identify the most informative SNPs and the optimum panel density. In general, SNPs selected using the F<sub>st</sub> highest approach performed the best in terms of predicting purebred and crossbred breed composition, but only a marginal difference was observed between the performance of SNPs selected across all SNP selection methods when <inline-formula id="inf48">
<mml:math id="m50">
<mml:mrow>
<mml:mo>&#x2265;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 3,000 SNPs were included in the analysis. This indicates that at this SNP density, all SNP selection methods could be a powerful computational time saving tool for the accurate prediction of purebred and crossbred breed composition.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The data analyzed in this study is subject to the following licenses/restrictions: The genotypes used are owned by the Irish Breeding Cattle Federation (ICBF). Requests to access these datasets should be directed to <ext-link ext-link-type="uri" xlink:href="https://www.icbf.com/">https://www.icbf.com/</ext-link>.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>CR, DB, AO&#x2019;B, TP, and DP participated in the design of the study and were involved in the interpretation of the results. CR performed the analyses and wrote the first draft of the manuscript. All authors read and approved the final manuscript. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>This publication has emanated from research supported in part by the Department of Agriculture, Food and the Marine (Dublin, Ireland) Research Stimulus Fund 2019R553 (Dairy4Beef) and by the European Commission in the frame of the Horizon 2020 INTAQT project (INnovative Tools for Assessment and Authentication of chicken meat, beef and dairy products&#x2019; QualiTies, Grant agreement ID: 101000250).</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>Author TP was employed by the company Irish Cattle Breeding Federation.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2023.1120312/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2023.1120312/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alexander</surname>
<given-names>D. H.</given-names>
</name>
<name>
<surname>Novembre</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lange</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Fast model-based estimation of ancestry in unrelated individuals</article-title>. <source>Genome Res.</source> <volume>19</volume>, <fpage>1655</fpage>&#x2013;<lpage>1664</lpage>. <pub-id pub-id-type="doi">10.1101/gr.094052.109</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barker</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rayens</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Partial least squares for discrimination</article-title>. <source>J. Chemom.</source> <volume>17</volume>, <fpage>166</fpage>&#x2013;<lpage>173</lpage>. <pub-id pub-id-type="doi">10.1002/cem.785</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bertolini</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Galimberti</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cal&#xf2;</surname>
<given-names>D. G.</given-names>
</name>
<name>
<surname>Schiavo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Matassino</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Fontanesi</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Combined use of principal component analysis and random forests identify population-informative single nucleotide polymorphisms: Application in cattle breeds</article-title>. <source>J. Animal Breed. Genet.</source> <volume>132</volume>, <fpage>346</fpage>&#x2013;<lpage>356</lpage>. <pub-id pub-id-type="doi">10.1111/jbg.12155</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bj&#xf8;rnstad</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>R&#xf8;ed</surname>
<given-names>K. H.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Evaluation of factors affecting individual assignment precision using microsatellite data from horse breeds and simulated breed crosses</article-title>. <source>Anim. Genet.</source> <volume>33</volume>, <fpage>264</fpage>&#x2013;<lpage>270</lpage>. <pub-id pub-id-type="doi">10.1046/j.1365-2052.2002.00868.x</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breiman</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2001</year>). <source>Mach. Learn.</source> <volume>45</volume>, <fpage>5</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1023/a:1010933404324</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brereton</surname>
<given-names>R. G.</given-names>
</name>
<name>
<surname>Lloyd</surname>
<given-names>G. R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Partial least squares discriminant analysis: Taking the magic away</article-title>. <source>J. Chemom.</source> <volume>28</volume>, <fpage>213</fpage>&#x2013;<lpage>225</lpage>. <pub-id pub-id-type="doi">10.1002/cem.2609</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chhotaray</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Panigrahi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pal</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ahmad</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bhushan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Gaur</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Ancestry informative markers derived from discriminant analysis of principal components provide important insights into the composition of crossbred cattle</article-title>. <source>Genomics</source> <volume>112</volume>, <fpage>1726</fpage>&#x2013;<lpage>1733</lpage>. <pub-id pub-id-type="doi">10.1016/j.ygeno.2019.10.008</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cortez</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Portelinha</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rodrigues</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cadavez</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Teixeira</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Lamb meat quality assessment by support vector machines</article-title>. <source>Neural Process. Lett.</source> <volume>24</volume>, <fpage>41</fpage>&#x2013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.1007/s11063-006-9009-6</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dalvit</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>De Marchi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zotto</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gervaso</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Meuwissen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Cassandro</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Breed assignment test in four Italian beef cattle breeds</article-title>. <source>Meat Sci.</source> <volume>80</volume>, <fpage>389</fpage>&#x2013;<lpage>395</lpage>. <pub-id pub-id-type="doi">10.1016/j.meatsci.2008.01.001</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wiener</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Abebe</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Altaye</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Go</surname>
<given-names>R. C.</given-names>
</name>
<name>
<surname>Kercsmar</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Comparison of measures of marker informativeness for ancestry and admixture mapping</article-title>. <source>BMC Genomics</source> <volume>12</volume>, <fpage>622</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-12-622</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Doekes</surname>
<given-names>H. P.</given-names>
</name>
<name>
<surname>Veerkamp</surname>
<given-names>R. F.</given-names>
</name>
<name>
<surname>Bijma</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Hiemstra</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Windig</surname>
<given-names>J. J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Trends in genome-wide and region-specific genetic diversity in the Dutch-Flemish Holstein&#x2013;Friesian breeding program from 1986 to 2015</article-title>. <source>Genet. Sel. Evol.</source> <volume>50</volume>, <fpage>15</fpage>&#x2013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1186/s12711-018-0385-y</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Dutheil</surname>
<given-names>J. Y.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Statistical population genomics</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Humana Press</publisher-name>.</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Frizzarin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gormley</surname>
<given-names>I. C.</given-names>
</name>
<name>
<surname>Berry</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Murphy</surname>
<given-names>T. B.</given-names>
</name>
<name>
<surname>Casa</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lynch</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Predicting cow milk quality traits from routinely available milk spectra using statistical machine learning methods</article-title>. <source>J. Dairy Sci.</source> <volume>104</volume>, <fpage>7438</fpage>&#x2013;<lpage>7447</lpage>. <pub-id pub-id-type="doi">10.3168/jds.2020-19576</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gianola</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Okut</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Weigel</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Rosa</surname>
<given-names>G. J.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Predicting complex quantitative traits with bayesian neural networks: A case study with Jersey cows and wheat</article-title>. <source>BMC Genet.</source> <volume>12</volume>, <fpage>87</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2156-12-87</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Fuller</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tait</surname>
<given-names>R. G.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Comparing SNP panels and statistical methods for estimating genomic breed composition of individual animals in ten cattle breeds</article-title>. <source>BMC Genet.</source> <volume>19</volume>, <fpage>56</fpage>. <pub-id pub-id-type="doi">10.1186/s12863-018-0654-3</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hempstalk</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Mcparland</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Berry</surname>
<given-names>D. P.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Machine learning algorithms for the prediction of conception success to a given insemination in lactating dairy cows</article-title>. <source>J. Dairy Sci.</source> <volume>98</volume>, <fpage>5262</fpage>&#x2013;<lpage>5273</lpage>. <pub-id pub-id-type="doi">10.3168/jds.2014-8984</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hulsegge</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Calus</surname>
<given-names>M. P. L.</given-names>
</name>
<name>
<surname>Windig</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Hoving-Bolink</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Maurice-Van Eijndhoven</surname>
<given-names>M. H. T.</given-names>
</name>
<name>
<surname>Hiemstra</surname>
<given-names>S. J.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Selection of SNP from 50K and 777K arrays to predict breed of origin in cattle</article-title>. <source>J. Animal Sci.</source> <volume>91</volume>, <fpage>5128</fpage>&#x2013;<lpage>5134</lpage>. <pub-id pub-id-type="doi">10.2527/jas.2013-6678</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Judge</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Kelleher</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Kearney</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Sleator</surname>
<given-names>R. D.</given-names>
</name>
<name>
<surname>Berry</surname>
<given-names>D. P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Ultra-low-density genotype panels for breed assignment of Angus and Hereford cattle</article-title>. <source>Animal</source> <volume>11</volume>, <fpage>938</fpage>&#x2013;<lpage>947</lpage>. <pub-id pub-id-type="doi">10.1017/S1751731116002457</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kavakiotis</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Triantafyllidis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ntelidou</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Alexandri</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Megens</surname>
<given-names>H.-J.</given-names>
</name>
<name>
<surname>Crooijmans</surname>
<given-names>R. P. M. A.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Tres: Identification of discriminatory and informative SNPs from population genomic data</article-title>. <source>J. Hered.</source> <volume>106</volume>, <fpage>672</fpage>&#x2013;<lpage>676</lpage>. <pub-id pub-id-type="doi">10.1093/jhered/esv044</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kelleher</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Berry</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Kearney</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Mcparland</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Buckley</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Purfield</surname>
<given-names>D. C.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Inference of population structure of purebred dairy and beef cattle using high-density genotype data</article-title>. <source>Animal</source> <volume>11</volume>, <fpage>15</fpage>&#x2013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1017/S1751731116001099</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kuehn</surname>
<given-names>L. A.</given-names>
</name>
<name>
<surname>Keele</surname>
<given-names>J. W.</given-names>
</name>
<name>
<surname>Bennett</surname>
<given-names>G. L.</given-names>
</name>
<name>
<surname>Mcdaneld</surname>
<given-names>T. G.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>T. P. L.</given-names>
</name>
<name>
<surname>Snelling</surname>
<given-names>W. M.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Predicting breed composition using breed frequencies of 50,000 markers from the US Meat Animal Research Center 2,000 bull project</article-title>. <source>J. Animal Sci.</source> <volume>89</volume>, <fpage>1742</fpage>&#x2013;<lpage>1750</lpage>. <pub-id pub-id-type="doi">10.2527/jas.2010-3530</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kuhn</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <source>caret: Classification and regression training</source>. <comment>R package version 6.0-85 Available at: <ext-link ext-link-type="uri" xlink:href="https://cran.r-projectorg/web/packages/caret/index.html">https://cran.r-projectorg/web/packages/caret/index.html</ext-link>
</comment>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Panigrahi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chhotaray</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Parida</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chauhan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bhushan</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Comparative analysis of five different methods to design a breed-specific SNP panel for cattle</article-title>. <source>Anim. Biotechnol.</source> <volume>32</volume>, <fpage>130</fpage>&#x2013;<lpage>136</lpage>. <pub-id pub-id-type="doi">10.1080/10495398.2019.1646266</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lashmar</surname>
<given-names>S. F.</given-names>
</name>
<name>
<surname>Berry</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Pierneef</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Muchadeyi</surname>
<given-names>F. C.</given-names>
</name>
<name>
<surname>Visser</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Assessing single nucleotide polymorphism selection methods for the development of a low-density panel optimized for imputation in South African Drakensberger beef cattle</article-title>. <source>J. Animal Sci.</source> <volume>99</volume>, <fpage>skab118</fpage>. <pub-id pub-id-type="doi">10.1093/jas/skab118</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lewis</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Abas</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Dadousis</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Lykidis</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Paschou</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Drineas</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Tracing cattle breeds with principal components analysis ancestry informative SNPs</article-title>. <source>PLoS ONE</source> <volume>6</volume>, <fpage>e18007</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0018007</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liaw</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wiener</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Classification and regression by RandomForest</article-title>. <source>Forest</source> <volume>23</volume>.</citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Maechler</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rousseeuw</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Struyf</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hubert</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hornik</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2021</year>). <source>cluster: Cluster analysis basics and extensions</source>.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mattucci</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Galaverni</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lyons</surname>
<given-names>L. A.</given-names>
</name>
<name>
<surname>Alves</surname>
<given-names>P. C.</given-names>
</name>
<name>
<surname>Randi</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Velli</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Genomic approaches to identify hybrids and estimate admixture times in European wildcat populations</article-title>. <source>Sci. Rep.</source> <volume>9</volume>, <fpage>11612</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-019-48002-w</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McClure</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>McCarthy</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Flynn</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>McClure</surname>
<given-names>J. C.</given-names>
</name>
<name>
<surname>Dair</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>O&#x2019;Connell</surname>
<given-names>D. K.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>SNP data quality control in a national beef and dairy cattle system and highly accurate SNP based parentage verification and identification</article-title>. <source>Front. Genet.</source> <volume>9</volume>, <fpage>84</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2018.00084</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McHugh</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Pabiou</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wall</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Mcdermott</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Berry</surname>
<given-names>D. P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Impact of alternative definitions of contemporary groups on genetic evaluations of traits recorded at lambing</article-title>. <source>J. Animal Sci.</source> <volume>95</volume>, <fpage>1926</fpage>&#x2013;<lpage>1938</lpage>. <pub-id pub-id-type="doi">10.2527/jas.2016.1344</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mullen</surname>
<given-names>M. P.</given-names>
</name>
<name>
<surname>Mcclure</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Kearney</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Waters</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Weld</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Flynn</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Development of a custom SNP chip for dairy and beef cattle breeding, parentage and research</article-title>. <source>Interbull Bull</source>.</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Novembre</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Bryc</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kutalik</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Boyko</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Auton</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Genes mirror geography within Europe</article-title>. <source>Nature</source> <volume>456</volume>, <fpage>98</fpage>&#x2013;<lpage>101</lpage>. <pub-id pub-id-type="doi">10.1038/nature07331</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>O&#x2019;Brien</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Purfield</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Judge</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Long</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Fair</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Berry</surname>
<given-names>D. P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Population structure and breed composition prediction in a multi-breed sheep population using genome-wide single nucleotide polymorphism genotypes</article-title>. <source>animal</source> <volume>14</volume>, <fpage>464</fpage>&#x2013;<lpage>474</lpage>. <pub-id pub-id-type="doi">10.1017/S1751731119002398</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paetkau</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Calvert</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Stirling</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Strobeck</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>1995</year>). <article-title>Microsatellite analysis of population structure in Canadian polar bears</article-title>. <source>Mol. Ecol.</source> <volume>4</volume>, <fpage>347</fpage>&#x2013;<lpage>354</lpage>. <pub-id pub-id-type="doi">10.1111/j.1365-294x.1995.tb00227.x</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paradis</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Claude</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Strimmer</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Ape: Analyses of phylogenetics and evolution in R language</article-title>. <source>Bioinformatics</source> <volume>20</volume>, <fpage>289</fpage>&#x2013;<lpage>290</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btg412</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paschou</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ziv</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Burchard</surname>
<given-names>E. G.</given-names>
</name>
<name>
<surname>Choudhry</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rodriguez-Cintron</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Mahoney</surname>
<given-names>M. W.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>PCA-correlated SNPs for structure identification in worldwide human populations</article-title>. <source>PLoS Genet.</source> <volume>3</volume>, <fpage>1672</fpage>&#x2013;<lpage>1686</lpage>. <pub-id pub-id-type="doi">10.1371/journal.pgen.0030160</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Patterson</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Price</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Reich</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Population structure and eigenanalysis</article-title>. <source>PLoS Genet.</source> <volume>2</volume>, <fpage>e190</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pgen.0020190</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Purcell</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Neale</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Todd-Brown</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Thomas</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>M. A. R.</given-names>
</name>
<name>
<surname>Bender</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>Plink: A tool set for whole-genome association and population-based linkage analyses</article-title>. <source>Am. J. Hum. Genet.</source> <volume>81</volume>, <fpage>559</fpage>&#x2013;<lpage>575</lpage>. <pub-id pub-id-type="doi">10.1086/519795</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reverter</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hudson</surname>
<given-names>N. J.</given-names>
</name>
<name>
<surname>Mcwilliam</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Alexandre</surname>
<given-names>P. A.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Barlow</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>A low-density SNP genotyping panel for the accurate prediction of cattle breeds</article-title>. <source>J. Animal Sci.</source> <volume>98</volume>, <fpage>skaa337</fpage>. <pub-id pub-id-type="doi">10.1093/jas/skaa337</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rodr&#xed;guez-Ramilo</surname>
<given-names>S. T.</given-names>
</name>
<name>
<surname>Fern&#xe1;ndez</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Toro</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Hern&#xe1;ndez</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Villanueva</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Genome-wide estimates of coancestry, inbreeding and effective population size in the Spanish Holstein population</article-title>. <source>PLoS One</source> <volume>10</volume>, <fpage>e0124157</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0124157</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sargolzaei</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chesnais</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Schenkel</surname>
<given-names>F. S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>A new approach for efficient genotype imputation using information from relatives</article-title>. <source>BMC Genomics</source> <volume>15</volume>, <fpage>478</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-15-478</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schiavo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bertolini</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Galimberti</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bovo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dall&#x27;Olio</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nanni Costa</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>A machine learning approach for the identification of population-informative markers from high-throughput genotyping data: Application to several pig breeds</article-title>. <source>Animal</source> <volume>14</volume>, <fpage>223</fpage>&#x2013;<lpage>232</lpage>. <pub-id pub-id-type="doi">10.1017/S1751731119002167</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sevillano</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Vandenplas</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bastiaansen</surname>
<given-names>J. W. M.</given-names>
</name>
<name>
<surname>Bergsma</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Calus</surname>
<given-names>M. P. L.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Genomic evaluation for a three-way crossbreeding system considering breed-of-origin of alleles</article-title>. <source>Genet. Sel. Evol.</source> <volume>49</volume>, <fpage>75</fpage>. <pub-id pub-id-type="doi">10.1186/s12711-017-0350-1</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>S&#xf6;lkner</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Hw</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>E</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>G</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>E</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> <year>2010</year>. <article-title>Estimation of individual levels of admixture in crossbred populations from SNP chip data: Examples with sheep and cattle populations</article-title>, <source>Interbull Bulletin</source>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stachowicz</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Sargolzaei</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Miglior</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Schenkel</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Rates of inbreeding and genetic diversity in Canadian Holstein and Jersey cattle</article-title>. <source>J. dairy Sci.</source> <volume>94</volume>, <fpage>5160</fpage>&#x2013;<lpage>5175</lpage>. <pub-id pub-id-type="doi">10.3168/jds.2010-3308</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stevens</surname>
<given-names>E. L.</given-names>
</name>
<name>
<surname>Heckenberg</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Roberson</surname>
<given-names>E. D. O.</given-names>
</name>
<name>
<surname>Baugher</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Downey</surname>
<given-names>T. J.</given-names>
</name>
<name>
<surname>Pevsner</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Inference of relationships in population data using identity-by-descent and identity-by-state</article-title>. <source>PLoS Genet.</source> <volume>7</volume>, <fpage>e1002287</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pgen.1002287</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Strucken</surname>
<given-names>E. M.</given-names>
</name>
<name>
<surname>Al-Mamun</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>Esquivelzeta-Rabell</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gondro</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mwai</surname>
<given-names>O. A.</given-names>
</name>
<name>
<surname>Gibson</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Genetic tests for estimating dairy breed proportion and parentage assignment in East African crossbred cattle</article-title>. <source>Genet. Sel. Evol.</source> <volume>49</volume>, <fpage>67</fpage>. <pub-id pub-id-type="doi">10.1186/s12711-017-0342-1</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Szyma&#x144;ska</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Saccenti</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Smilde</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Westerhuis</surname>
<given-names>J. A.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Double-check: Validation of diagnostic statistics for PLS-DA models in metabolomics studies</article-title>. <source>Metabolomics</source> <volume>8</volume>, <fpage>3</fpage>&#x2013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1007/s11306-011-0330-3</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Team</surname>
<given-names>M. D.</given-names>
</name>
</person-group> (<year>2017</year>). <source>MiX99: A software package for solving large mixed model equations</source>. <publisher-loc>Finland</publisher-loc>: <publisher-name>Natural Resources Institute Finland (Luke) Jokioinen</publisher-name>.</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thomasen</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>S&#xf8;rensen</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Madsen</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Lund</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Guldbrandtsen</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>The admixed population structure in Danish Jersey dairy cattle challenges accurate genomic predictions</article-title>. <source>J. Animal Sci.</source> <volume>91</volume>, <fpage>3105</fpage>&#x2013;<lpage>3112</lpage>. <pub-id pub-id-type="doi">10.2527/jas.2012-5490</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Der Heide</surname>
<given-names>E. M. M.</given-names>
</name>
<name>
<surname>Veerkamp</surname>
<given-names>R. F.</given-names>
</name>
<name>
<surname>Van Pelt</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Kamphuis</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Athanasiadis</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Ducro</surname>
<given-names>B. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Comparing regression, naive Bayes, and random forest methods in the prediction of individual survival to second lactation in Holstein cattle</article-title>. <source>J. Dairy Sci.</source> <volume>102</volume>, <fpage>9409</fpage>&#x2013;<lpage>9421</lpage>. <pub-id pub-id-type="doi">10.3168/jds.2019-16295</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Hertem</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Viazzi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Steensels</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Maltz</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Antler</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Alchanatis</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Automatic lameness detection based on consecutive 3D-video recordings</article-title>. <source>Biosyst. Eng.</source> <volume>119</volume>, <fpage>108</fpage>&#x2013;<lpage>116</lpage>. <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2014.01.009</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Warner</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Vasseur</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Lefebvre</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Lacroix</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A machine learning based decision aid for lameness in dairy herds using farm-based records</article-title>. <source>Comput. Electron. Agric.</source> <volume>169</volume>, <fpage>105193</fpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2019.105193</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weir</surname>
<given-names>B. S.</given-names>
</name>
<name>
<surname>Cockerham</surname>
<given-names>C. C.</given-names>
</name>
</person-group> (<year>1984</year>). <article-title>Estimating F-statistics for the analysis of population structure</article-title>. <source>Evolution</source> <volume>38</volume>, <fpage>1358</fpage>&#x2013;<lpage>1370</lpage>. <pub-id pub-id-type="doi">10.1111/j.1558-5646.1984.tb05657.x</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Westerhuis</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Hoefsloot</surname>
<given-names>H. C. J.</given-names>
</name>
<name>
<surname>Smit</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Vis</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Smilde</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Van Velzen</surname>
<given-names>E. J. J.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Assessment of PLSDA cross validation</article-title>. <source>Metabolomics</source> <volume>4</volume>, <fpage>81</fpage>&#x2013;<lpage>89</lpage>. <pub-id pub-id-type="doi">10.1007/s11306-007-0099-6</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wilkinson</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Archibald</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Haley</surname>
<given-names>C. S.</given-names>
</name>
<name>
<surname>Megens</surname>
<given-names>H.-J.</given-names>
</name>
<name>
<surname>Crooijmans</surname>
<given-names>R. P.</given-names>
</name>
<name>
<surname>Groenen</surname>
<given-names>M. A.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Development of a genetic tool for product regulation in the diverse British pig breed market</article-title>. <source>BMC Genomics</source> <volume>13</volume>, <fpage>580</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-13-580</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wilkinson</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wiener</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Archibald</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Law</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schnabel</surname>
<given-names>R. D.</given-names>
</name>
<name>
<surname>Mckay</surname>
<given-names>S. D.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Evaluation of approaches for identifying population informative markers from high density SNP Chips</article-title>. <source>BMC Genet.</source> <volume>12</volume>, <fpage>45</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2156-12-45</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S. H.</given-names>
</name>
<name>
<surname>Goddard</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Visscher</surname>
<given-names>P. M.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Gcta: A tool for genome-wide complex trait analysis</article-title>. <source>Am. J. Hum. Genet.</source> <volume>88</volume>, <fpage>76</fpage>&#x2013;<lpage>82</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2010.11.011</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>