<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">729867</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2021.729867</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A Comprehensive Comparison of Haplotype-Based Single-Step Genomic Predictions in Livestock Populations With Different Genetic Diversity Levels: A Simulation Study</article-title>
<alt-title alt-title-type="left-running-head">Araujo et&#x20;al.</alt-title>
<alt-title alt-title-type="right-running-head">Haplotype-Based Genomic Predictions</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Araujo</surname>
<given-names>Andre C.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1046316/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Carneiro</surname>
<given-names>Paulo L. S.</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1315614/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Oliveira</surname>
<given-names>Hinayah R.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/909009/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Schenkel</surname>
<given-names>Flavio S.</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/709239/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Veroneze</surname>
<given-names>Renata</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/887821/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lourenco</surname>
<given-names>Daniela A. L.</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/183636/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Brito</surname>
<given-names>Luiz F.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">
<sup>&#x2a;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/684942/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<label>
<sup>1</sup>
</label>Postgraduate Program in Animal Sciences, State University of Southwestern Bahia, <addr-line>Itapetinga</addr-line>, <country>Brazil</country>
</aff>
<aff id="aff2">
<label>
<sup>2</sup>
</label>Department of Animal Sciences, Purdue University, <addr-line>West Lafayette</addr-line>, <addr-line>IN</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff3">
<label>
<sup>3</sup>
</label>Department of Biology, State University of Southwestern Bahia, <addr-line>Jequi&#xe9;</addr-line>, <country>Brazil</country>
</aff>
<aff id="aff4">
<label>
<sup>4</sup>
</label>Centre for Genetic Improvement of Livestock, Department of Animal Biosciences, University of Guelph, <addr-line>Guelph</addr-line>, <addr-line>ON</addr-line>, <country>Canada</country>
</aff>
<aff id="aff5">
<label>
<sup>5</sup>
</label>Department of Animal Sciences, Federal University of Vi&#xe7;osa, <addr-line>Vi&#xe7;osa</addr-line>, <country>Brazil</country>
</aff>
<aff id="aff6">
<label>
<sup>6</sup>
</label>Department of Animal and Dairy Science, University of Georgia, <addr-line>Athens</addr-line>, <addr-line>GA</addr-line>, <country>United&#x20;States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/812302/overview">Guosheng Su</ext-link>, Aarhus University, Denmark</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1234201/overview">Beatriz Cuyabano</ext-link>, Institut National de recherche pour l&#x2019;agriculture, l&#x2019;alimentation et l&#x2019;environnement (INRAE), France</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1259488/overview">Lei Zhou</ext-link>, China Agricultural University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1247291/overview">Emre Karaman</ext-link>, Aarhus University, Denmark</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Luiz F. Brito, <email>britol@purdue.edu</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Livestock Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>14</day>
<month>10</month>
<year>2021</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>12</volume>
<elocation-id>729867</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>06</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>09</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2021 Araujo, Carneiro, Oliveira, Schenkel, Veroneze, Lourenco and Brito.</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Araujo, Carneiro, Oliveira, Schenkel, Veroneze, Lourenco and Brito</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these&#x20;terms.</p>
</license>
</permissions>
<abstract>
<p>The level of genetic diversity in a population is inversely proportional to the linkage disequilibrium (LD) between individual single nucleotide polymorphisms (SNPs) and quantitative trait loci (QTLs), leading to lower predictive ability of genomic breeding values (GEBVs) in high genetically diverse populations. Haplotype-based predictions could outperform individual SNP predictions by better capturing the LD between SNP and QTL. Therefore, we aimed to evaluate the accuracy and bias of individual-SNP- and haplotype-based genomic predictions under the single-step-genomic best linear unbiased prediction (ssGBLUP) approach in genetically diverse populations. We simulated purebred and composite sheep populations using literature parameters for moderate and low heritability traits. The haplotypes were created based on LD thresholds of 0.1, 0.3, and 0.6. Pseudo-SNPs from unique haplotype alleles were used to create the genomic relationship matrix (<inline-formula id="inf1">
<mml:math id="m1">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula>) in the ssGBLUP analyses. Alternative scenarios were compared in which the pseudo-SNPs were combined with non-LD clustered SNPs, only pseudo-SNPs, or haplotypes fitted in a second <inline-formula id="inf2">
<mml:math id="m2">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> (two relationship matrices). The GEBV accuracies for the moderate heritability-trait scenarios fitting individual SNPs ranged from 0.41 to 0.55 and with haplotypes from 0.17 to 0.54 in the most (Ne <inline-formula id="inf3">
<mml:math id="m3">
<mml:mo>&#x2245;</mml:mo>
</mml:math>
</inline-formula> 450) and less (Ne &#x3c; 200) genetically diverse populations, respectively, and the bias fitting individual SNPs or haplotypes ranged between &#x2212;0.14 and &#x2212;0.08 and from &#x2212;0.62 to &#x2212;0.08, respectively. For the low heritability-trait scenarios, the GEBV accuracies fitting individual SNPs ranged from 0.24 to 0.32, and for fitting haplotypes, it ranged from 0.11 to 0.32 in the more (Ne<inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#x2245;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 250) and less (Ne<inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#x2245;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 100) genetically diverse populations, respectively, and the bias ranged between &#x2212;0.36 and &#x2212;0.32 and from &#x2212;0.78 to &#x2212;0.33 fitting individual SNPs or haplotypes, respectively. The lowest accuracies and largest biases were observed fitting only pseudo-SNPs from blocks constructed with an LD threshold of 0.3 (<italic>p</italic>&#x20;&#x3c; 0.05), whereas the best results were obtained using only SNPs or the combination of independent SNPs and pseudo-SNPs in one or two <inline-formula id="inf6">
<mml:math id="m6">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> matrices, in both heritability levels and all populations regardless of the level of genetic diversity. In summary, haplotype-based models did not improve the performance of genomic predictions in genetically diverse populations.</p>
</abstract>
<kwd-group>
<kwd>effective population size</kwd>
<kwd>genomic estimated breeding value</kwd>
<kwd>haplotype blocks</kwd>
<kwd>linkage disequilibrium</kwd>
<kwd>pseudo-SNP</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Genomic selection (GS) (<xref ref-type="bibr" rid="B38">Meuwissen et&#x20;al., 2001</xref>) is now routinely used worldwide in livestock and plant breeding programs (<xref ref-type="bibr" rid="B34">Lourenco et&#x20;al., 2020</xref>; <xref ref-type="bibr" rid="B42">Moreira et&#x20;al., 2020</xref>). GS enables the prediction of more accurate genomic estimated breeding values (GEBVs) at earlier stages compared to the traditional pedigree-based evaluation (<xref ref-type="bibr" rid="B5">Brito et&#x20;al., 2017a</xref>; <xref ref-type="bibr" rid="B19">Guarini et&#x20;al., 2018</xref>, <xref ref-type="bibr" rid="B20">2019</xref>). The advantages of GS compared to the pedigree-based are even greater for lowly-heritable traits, traits measured late in life, and sex-limited or expensive-to-measure traits (<xref ref-type="bibr" rid="B11">Daetwyler et&#x20;al., 2012</xref>; <xref ref-type="bibr" rid="B34">Lourenco et&#x20;al., 2020</xref>).</p>
<p>Over the past 15&#x2013;20&#xa0;years, several statistical methods have been proposed aiming to obtain more accurate and less biased GEBVs. Among the available methods, the single-step genomic best linear unbiased prediction (ssGBLUP; <xref ref-type="bibr" rid="B30">Legarra et&#x20;al., 2009</xref>; <xref ref-type="bibr" rid="B1">Aguilar et&#x20;al., 2010</xref>) is widely used to perform genomic predictions in livestock. This method enables the simultaneous evaluation of both genotyped and non-genotyped individuals and has similar or better statistical properties and predictive ability compared to other approaches such as pedigree-based BLUP and multi-step GBLUP (<xref ref-type="bibr" rid="B1">Aguilar et&#x20;al., 2010</xref>; <xref ref-type="bibr" rid="B31">Legarra et&#x20;al., 2014</xref>; <xref ref-type="bibr" rid="B19">Guarini et&#x20;al., 2018</xref>; <xref ref-type="bibr" rid="B49">Piccoli et&#x20;al., 2020</xref>).</p>
<p>Although the pioneer GS study (i.e.,&#x20;<xref ref-type="bibr" rid="B38">Meuwissen et&#x20;al., 2001</xref>) fitted single nucleotide polymorphism (SNP) haplotypes as covariates in the models, subsequent studies were mainly performed based on individual SNPs. This is most likely due to the additional analytic steps and higher computational requirements when fitting haplotype-based models. In this sense, it is important to first define the haplotype blocks or haploblocks, which are sizable regions of the genome with little evidence of historical recombination (<xref ref-type="bibr" rid="B18">Gabriel et&#x20;al., 2002</xref>), i.e.,&#x20;a genomic region between two or more marker loci. More recently, the use of haplotypes as covariates in genomic evaluations rather than single SNPs has been further investigated due to many potential advantages. Haplotypes are more polymorphic than individual SNPs because they can be multi-allelic (<xref ref-type="bibr" rid="B39">Meuwissen et&#x20;al., 2014</xref>) and they can be in stronger linkage disequilibrium (LD) with Quantitative Trait Loci (QTLs) compared to individual SNPs with low minor allele frequency (MAF) (<xref ref-type="bibr" rid="B22">Hess et&#x20;al., 2017</xref>). In this context, the potential stronger LD between haplotypes and QTL in comparison to individual SNPs can yield more accurate GEBVs (<xref ref-type="bibr" rid="B8">Calus et&#x20;al., 2008</xref>; <xref ref-type="bibr" rid="B9">Cuyabano et&#x20;al., 2014</xref>; <xref ref-type="bibr" rid="B10">2015</xref>). Moreover, haplotype alleles have the potential to capture epistatic effects within blocks and the QTL can be flanked by SNPs that delimit the haploblock (<xref ref-type="bibr" rid="B22">Hess et&#x20;al., 2017</xref>; <xref ref-type="bibr" rid="B24">Jiang et&#x20;al., 2018</xref>; <xref ref-type="bibr" rid="B25">Karimi et&#x20;al., 2018</xref>).</p>
<p>Previous studies based on simulated data have shown that fitting haplotypes can substantially improve the performance of genomic predictions compared to individual SNP-based methods (<xref ref-type="bibr" rid="B8">Calus et&#x20;al., 2008</xref>; <xref ref-type="bibr" rid="B63">Villumsen et&#x20;al., 2009</xref>). However, none or only small increases in the predictive ability of GEBVs have been observed in practice (e.g., <xref ref-type="bibr" rid="B9">Cuyabano et&#x20;al., 2014</xref>, <xref ref-type="bibr" rid="B10">2015</xref>; <xref ref-type="bibr" rid="B22">Hess et&#x20;al., 2017</xref>; <xref ref-type="bibr" rid="B25">Karimi et&#x20;al., 2018</xref>; <xref ref-type="bibr" rid="B44">Mucha et&#x20;al., 2019</xref>; <xref ref-type="bibr" rid="B64">Won et&#x20;al., 2020</xref>). The large majority of the studies evaluating haplotype-based models were done in dairy cattle populations (real or simulated datasets), which usually have high LD levels between SNP markers and lower genetic diversity (Ne lower than 100; <xref ref-type="bibr" rid="B35">Makanjuola et&#x20;al., 2020</xref>). Haplotype-based genomic predictions in populations with increased genetic diversity, on the other hand, have not been widely explored yet, and the knowledge of their possible advantages is limited (<xref ref-type="bibr" rid="B16">Feitosa et&#x20;al., 2019</xref>; <xref ref-type="bibr" rid="B60">Teissier et&#x20;al., 2020</xref>).</p>
<p>Different from intensively selected populations and pure breeds, which present low genetic diversity (e.g., Holstein dairy cattle), genetically diverse populations (e.g., relatively recent breeding programs in small ruminants and crossbred or composite populations) may have more alleles segregating in the haplotype blocks and greater complexity in the interactions among haplotype allele effects within haploblocks. Thus, we hypothesize that haplotype-based methods could result in more accurate and less biased GEBV prediction when compared to SNP-based models in populations with high genetic diversity because of their development process (e.g., relatively lower selection pressures, crossbreeding) and more complex haplotype structure than observed in populations with low genetic diversity. Simulated data is an interesting approach to investigate this hypothesis because the true breeding values (TBVs) are known (<xref ref-type="bibr" rid="B43">Morris et&#x20;al., 2019</xref>; <xref ref-type="bibr" rid="B12">Oliveira et&#x20;al., 2019</xref>). Therefore, we simulated sheep populations with different genetic diversity levels to test our hypothesis. Sheep is a good model due to the large genetic diversity in commercial populations, with Ne ranging from less than 50 to over 1,000 (<xref ref-type="bibr" rid="B26">Kijas et&#x20;al., 2012</xref>; <xref ref-type="bibr" rid="B7">Brito et&#x20;al., 2017b</xref>; <xref ref-type="bibr" rid="B58">Stachowicz et&#x20;al., 2018</xref>). Hence, the main objective of this study was to evaluate the accuracy and bias of GEBVs in genetically diverse populations, using ssGBLUP when: 1) only individual SNPs are used to construct a single genomic relationship matrix (<inline-formula id="inf7">
<mml:math id="m7">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula>); 2) non-clustered (out of haploblocks) SNPs and haplotypes (fitted as pseudo-SNPs) are used to construct a single <inline-formula id="inf8">
<mml:math id="m8">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula>; 3) only haplotypes are used to construct a single <inline-formula id="inf9">
<mml:math id="m9">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula>; and 4) non-clustered SNPs and haplotypes are used to construct two <inline-formula id="inf10">
<mml:math id="m10">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> matrices. We also compared the impact of different SNP panel densities and haploblock-building methods on the performance of genomic prediction, as these factors could impact the accuracies and bias of genomic predictions.</p>
</sec>
<sec id="s2">
<title>2 Materials and Methods</title>
<p>The approval of Institutional Animal Care and Use Committee was not required because this study only used computationally simulated datasets.</p>
<sec id="s2-1">
<title>2.1 Data Simulation</title>
<sec id="s2-1-1">
<title>2.1.1 Population Structure</title>
<p>The simulation was performed to mimic datasets of purebred and composite sheep populations (<xref ref-type="bibr" rid="B26">Kijas et&#x20;al., 2012</xref>; <xref ref-type="bibr" rid="B51">Prieur et&#x20;al., 2017</xref>; <xref ref-type="bibr" rid="B5">Brito et&#x20;al., 2017a</xref>; <xref ref-type="bibr" rid="B47">Oliveira et&#x20;al., 2020</xref>). The QMSim software (<xref ref-type="bibr" rid="B56">Sargolzaei and Schenkel, 2009</xref>) was used to simulate a historical population initially with 80,000 individuals (40,000 males and 40,000 females). Then, a population bottleneck was simulated, reaching 50,000 individuals (25,000 males and 25,000 females) in the 1,000th generation. After that, there was an increase in the population to 60,000 individuals, with 20,000 males and 40,000 females in the 1,500th generation. There was random mating in the historical population, with gametes randomly sampled from the pool of males and females present in each generation. Mutation and genetic drift were considered in the historical population to create the initial LD. The complete simulation design is summarized in <xref ref-type="fig" rid="F1">Figure&#x20;1</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Simulation design to obtain pure and composite sheep populations.</p>
</caption>
<graphic xlink:href="fgene-12-729867-g001.tif"/>
</fig>
<p>Five random samples from the last historical population were selected to create five pure breeds, called A, B, C, D, and E (<xref ref-type="fig" rid="F1">Figure&#x20;1</xref>). The combination of different founder population sizes (2,480 animals for the breeds A and B, 12,480 for the breed C, and 41,600 for the breeds D and E) and generations of phenotypic selection (10 for the breeds A and B, and one generation for the breeds C, D, and E) were used to achieve different LD patterns and, consequently, different Ne in the most recent populations. There were random matings and exponential increase in the number of females in a rate of 0.10 for the breeds A and B and 0.15 for the breeds C, D, and E. During the generations of phenotypic selection, it can be considered that the breeds were separated geographically, restricting the mating within each population. Subsequently, the pure breeds were divergently selected based on estimated breeding values (EBVs) predicted using BLUP, with breeds A, C, and D selected for increasing and breeds B and E for decreasing the EBVs for the simulated trait. All breeds were selected based on the EBVs during 10 generations. The male/female ratio in the EBV-selected populations was 1/25, with a replacement rate of 40% for males and 20% for females. There were single, double, and triple births, with the odds of 30, 50, and 20%, respectively, to be similar with the ones observed in sheep flocks. The number of individuals in each generation of EBV-based selection were tested and at the end were greater than 7,000 to allow a reasonable number of selection candidates in each generation.</p>
<p>Crosses were made to obtain composite breeds, which had two or three pure breeds as the starting point (<xref ref-type="fig" rid="F1">Figure&#x20;1</xref>). Two composite populations were created based on either two breeds (Comp_2), which had 62.5% of breed D and 37.5% of breed E (<xref ref-type="fig" rid="F1">Figure&#x20;1</xref>), or three breeds (Comp_3), which had 37.5% of breed A, 37.5% of breed B, and 25.0% of breed C (<xref ref-type="fig" rid="F1">Figure&#x20;1</xref>). Random mating was restricted within each crossbreed population for five generations. According to <xref ref-type="bibr" rid="B54">Rasali et&#x20;al. (2006)</xref>, five-to-six generations are sufficient to stabilize the frequencies of linked genes in new populations. Thereafter, the composite breeds were divergently selected using EBVs for the next 10 generations, with Comp_2 and Comp_3 divergently selected for decreasing and increasing performance, respectively. Mating type, sire and dam replacements, and the number of births per dam in the composite breeds were the same as those previously described for the pure breeds. The number of individuals per generation in the composite breeds (during the selection based on EBVs) was more than 18,000, to keep a higher Ne on those populations compared to the pure breeds.</p>
</sec>
<sec id="s2-1-2">
<title>2.1.2 Effective Population Size in the Recent Populations</title>
<p>The number of generations in the pure breeds during the expansion of the recent populations were modified accordingly to achieve the LD patterns corresponding to Ne of &#x223c;100, &#x223c;250, and &#x223c;500. The Ne was calculated using the LD and the realized inbreeding in the recent populations for pure and composite breeds under EBV-based selection. With the LD approach, Ne was estimated using the formula: <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>4</mml:mn>
<mml:mi>c</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msup>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, which is a re-arrangement of the estimator <inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msup>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>4</mml:mn>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> proposed by <xref ref-type="bibr" rid="B59">Sved (1971)</xref>, where <inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msup>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the expected LD for a population with effective size Ne, <inline-formula id="inf14">
<mml:math id="m14">
<mml:mi>c</mml:mi>
</mml:math>
</inline-formula> is the genetic distance (chromosome segment size in Morgans&#x2014;M) within autosomal chromosomes. It was considered that 1&#xa0;Mb corresponds to a centimorgan (cM) when calculating the <inline-formula id="inf15">
<mml:math id="m15">
<mml:mi>c</mml:mi>
</mml:math>
</inline-formula> value, as this is an acceptable approximation in sheep (<xref ref-type="bibr" rid="B51">Prieur et&#x20;al., 2017</xref>). Lastly, populations were simulated to have an LD of approximately 0.024, 0.010, and 0.005 for SNPs spaced apart by 10&#xa0;Mb, which correspond to the values of <inline-formula id="inf16">
<mml:math id="m16">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msup>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> for Ne &#x3d; 100, 250, and 500, respectively. A 10&#xa0;Mb distance corresponds to an Ne that existed five generations ago (considered as current Ne), based on the relationship <inline-formula id="inf17">
<mml:math id="m17">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> proposed by <xref ref-type="bibr" rid="B21">Hayes et&#x20;al. (2003)</xref>, where <inline-formula id="inf18">
<mml:math id="m18">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula> is the number of generations ago and <inline-formula id="inf19">
<mml:math id="m19">
<mml:mi>c</mml:mi>
</mml:math>
</inline-formula> is as previously defined. Estimation of LD was performed considering only SNPs with MAF higher than 0.05 using the <inline-formula id="inf20">
<mml:math id="m20">
<mml:mrow>
<mml:msup>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>metric (<xref ref-type="bibr" rid="B23">Hill and Robertson, 1968</xref>). We also estimated the Ne based on the realized inbreeding five generations ago using the formula (<xref ref-type="bibr" rid="B14">Falconer and Mackay, 1996</xref>): <inline-formula id="inf21">
<mml:math id="m21">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>&#x394;</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf22">
<mml:math id="m22">
<mml:mrow>
<mml:mi>&#x394;</mml:mi>
<mml:mi>F</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf23">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the average inbreeding in the <italic>n</italic>th generation. The average inbreeding per generation was obtained from the QMSim software outputs (<xref ref-type="bibr" rid="B56">Sargolzaei and Schenkel, 2009</xref>).</p>
</sec>
<sec id="s2-1-3">
<title>2.1.3 Simulated Traits</title>
<p>We simulated two traits with initial heritability levels of 0.30 and 0.10 (global parameters for the QMSim software; <xref ref-type="bibr" rid="B56">Sargolzaei and Schenkel, 2009</xref>), to represent moderate (MH2) and low (LH2) additive genetic effects, respectively, affecting the total phenotypic variability of the trait. The phenotypic variance was set to 100 in both simulations. The heritability was estimated in the recent populations based on pedigree and phenotype information using the AIREMLf90 software (<xref ref-type="bibr" rid="B40">Misztal et&#x20;al., 2018</xref>) to verify if the desired values were achieved. All simulations were replicated five times using different seed values in order to simulate different populations. Only additive genetic effects were simulated due to the QMSim software (<xref ref-type="bibr" rid="B56">Sargolzaei and Schenkel, 2009</xref>) capabilities.</p>
</sec>
<sec id="s2-1-4">
<title>2.1.4 Genome and Data Editing</title>
<p>The genome was simulated with 26 autosomal chromosomes with size varying between 43 and 301&#xa0;cM (a total of 2,656&#xa0;cM), mimicking the sheep genome (<xref ref-type="sec" rid="s11">Supplementary Material S1</xref>). The number and size of chromosomes were defined based on information obtained from the most recent sheep reference genome (assembly OAR_v4.0) available in the NCBI platform (<ext-link ext-link-type="uri" xlink:href="https://wbww.ncbi.nlm.nih.gov/genome?term=ovis%20aries">www.ncbi.nlm.nih.gov/genome?term&#x3d;ovis%20aries</ext-link>). The genome simulation was also performed using the QMSim software (<xref ref-type="bibr" rid="B56">Sargolzaei and Schenkel, 2009</xref>).</p>
<p>A total of 3,057 QTLs were simulated, spanning the whole autosomal genome. The number of QTLs per chromosome varied between 51 and 391 (<xref ref-type="sec" rid="s11">Supplementary Material S1</xref>), which was chosen based on the information published in the AnimalQTLdb (<xref ref-type="bibr" rid="B2">AnimalQTLdb, 2019</xref>). QTLs with the number of alleles varying from two to six were simulated to evaluate the advantages of using haplotype-based approaches. All simulated markers were bi-allelic to mimic SNP markers, and the total number of SNPs was set to 576,595 (<xref ref-type="sec" rid="s11">Supplementary Material S1</xref>; similar number of autosomal SNPs included in the Ovine Infinium&#xae; HD SNP Beadchip 600K; <xref ref-type="bibr" rid="B15">FarmIQ, 2013</xref>; <xref ref-type="bibr" rid="B27">Kijas et&#x20;al., 2014</xref>) sampled from the segregating loci (MAF &#x2265;0.05) in the last historical generation. The information on the number of markers in each chromosome was obtained from the SNPchiMp v.3 platform (<xref ref-type="bibr" rid="B46">Nicolazzi et&#x20;al., 2015</xref>). Both QTL and markers were randomly distributed within chromosome and placed in different chromosomic positions, i.e.,&#x20;simulated QTLs were not among the SNPs, so that the genomic predictions rely only on the LD between&#x20;them.</p>
<p>The additive genetic effects of the QTL were sampled from a gamma distribution with the shape parameter equal to 0.4, whereas no effects were simulated for the SNP markers. The initial allele frequencies assumed for QTL and markers (generation 0 of the historical population) were 0.5. The QTL heritability on the MH2 and LH2 traits was equal to 50 and 10% of the trait heritability, i.e.,&#x20;0.15 and 0.01, respectively. The remaining genetic variance not explained by the QTLs was attributed to the polygenic effect. Recurrent mutation rates on the order of 1&#x20;&#xd7; 10<sup>&#x2212;4</sup> were simulated for the QTL and markers. Rates of 0.05 and 0.01 were used for the occurrence of missing genotypes and genotyping errors, respectively.</p>
<p>Quality control (QC) was performed in the genotype file of each simulated recent population for each replicate, using the PREGSf90 software from the BLUPf90 family programs (<xref ref-type="bibr" rid="B40">Misztal et&#x20;al., 2018</xref>). In this step, SNPs with no extreme departure from Hardy&#x2013;Weinberg equilibrium (difference between observed and expected frequency of heterozygous less than 0.15) and MAF &#x2265;0.01 were maintained. All SNPs passed this QC for all populations, indicating that there was enough variability on the simulated SNP chip&#x20;panel.</p>
</sec>
</sec>
<sec id="s2-2">
<title>2.2 Haplotype Blocks Construction</title>
<p>The FImpute v.3.0 software (<xref ref-type="bibr" rid="B55">Sargolzaei et&#x20;al., 2014</xref>) was used to phase the genotypes (i.e.,&#x20;to infer SNP allele inheritance). Subsequently, the haploblocks were constructed using different LD thresholds (variable haploblock sizes), as described below. The <inline-formula id="inf24">
<mml:math id="m24">
<mml:mrow>
<mml:msup>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> metric (<xref ref-type="bibr" rid="B23">Hill and Robertson, 1968</xref>) was used to calculate the LD between markers to construct the haploblocks, as this measure is less sensitive to allele frequency (<xref ref-type="bibr" rid="B4">Bohmanova et&#x20;al., 2010</xref>). The &#x201c;gpart&#x201d; package (<xref ref-type="bibr" rid="B28">Kim et&#x20;al., 2019</xref>) implemented in the R software (<xref ref-type="bibr" rid="B52">R Core Team, 2020</xref>) was used to build the haploblocks considering <inline-formula id="inf25">
<mml:math id="m25">
<mml:mrow>
<mml:msup>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> levels of 0.1 (low), 0.3 (moderate), and 0.6 (high) based on the Big-LD approach (<xref ref-type="bibr" rid="B29">Kim et&#x20;al., 2018</xref>). Following the previous definition of haploblocks (<xref ref-type="bibr" rid="B18">Gabriel et&#x20;al., 2002</xref>), a haploblock in this study was considered as a genomic region spanning at least two&#x20;SNPs.</p>
</sec>
<sec id="s2-3">
<title>2.3 Prediction of GEBV</title>
<p>All genomic predictions were performed using the ssGBLUP method implemented in the BLUPf90 family programs (<xref ref-type="bibr" rid="B40">Misztal et&#x20;al., 2018</xref>). Before using the BLUPf90 software, the AIREMLf90 software (<xref ref-type="bibr" rid="B40">Misztal et&#x20;al., 2018</xref>) was used to estimate the variance components for each simulation replicate for the models described in the next sections.</p>
<sec id="s2-3-1">
<title>2.3.1 ssGBLUP Using SNPs</title>
<p>The model used to predict the GEBVs under this approach was<disp-formula id="equ1">
<mml:math id="m26">
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="bold">Xb</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold">Zu</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold">e</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf26">
<mml:math id="m27">
<mml:mi mathvariant="normal">y</mml:mi>
</mml:math>
</inline-formula> is an N &#xd7; 1 vector of phenotypes for genotyped and non-genotyped animals, <inline-formula id="inf27">
<mml:math id="m28">
<mml:mi mathvariant="normal">b</mml:mi>
</mml:math>
</inline-formula> is the vector of fixed effects (i.e.,&#x20;generation), <inline-formula id="inf28">
<mml:math id="m29">
<mml:mi mathvariant="normal">u</mml:mi>
</mml:math>
</inline-formula> is a random vector of GEBVs for genotyped and non-genotyped animals with <inline-formula id="inf29">
<mml:math id="m30">
<mml:mrow>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf30">
<mml:math id="m31">
<mml:mi mathvariant="normal">e</mml:mi>
</mml:math>
</inline-formula> is the vector of random errors with <inline-formula id="inf31">
<mml:math id="m32">
<mml:mrow>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">I</mml:mi>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>e</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf32">
<mml:math id="m33">
<mml:mi mathvariant="normal">X</mml:mi>
</mml:math>
</inline-formula> is the incidence matrix of fixed effects, and <inline-formula id="inf33">
<mml:math id="m34">
<mml:mi mathvariant="normal">Z</mml:mi>
</mml:math>
</inline-formula> is the incidence matrix that relates the records to GEBVs. In the case of ssGBLUP fitting individual SNPs, the <inline-formula id="inf34">
<mml:math id="m35">
<mml:mi mathvariant="normal">H</mml:mi>
</mml:math>
</inline-formula> matrix is a hybrid relationship matrix that combines the genomic and pedigree relationships (<xref ref-type="bibr" rid="B30">Legarra et&#x20;al., 2009</xref>), and its inverse can be computed directly in the mixed model equations as follows (<xref ref-type="bibr" rid="B1">Aguilar et&#x20;al., 2010</xref>):<disp-formula id="equ2">
<mml:math id="m36">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold">H</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi mathvariant="bold">G</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:msub>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mrow>
<mml:mn>22</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">&#x2212;</mml:mi>
<mml:mi mathvariant="bold">1</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
<mml:msubsup>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold">22</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">&#x2212;</mml:mi>
<mml:mi mathvariant="bold">1</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf35">
<mml:math id="m37">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the inverse of pedigree relationship matrix, <inline-formula id="inf36">
<mml:math id="m38">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mrow>
<mml:mn>22</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the inverse of pedigree relationship matrix for the genotyped animals, and <inline-formula id="inf37">
<mml:math id="m39">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> is the genomic relationship matrix. The <inline-formula id="inf38">
<mml:math id="m40">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> matrix was constructed as in the first method proposed by <xref ref-type="bibr" rid="B62">Vanraden (2008)</xref>:<disp-formula id="equ3">
<mml:math id="m41">
<mml:mrow>
<mml:mi mathvariant="bold">G&#x3d;</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="bold">MM&#x2032;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">2</mml:mi>
<mml:mi>&#x3a3;</mml:mi>
<mml:msub>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf39">
<mml:math id="m42">
<mml:mi mathvariant="normal">M</mml:mi>
</mml:math>
</inline-formula> is the matrix of centered genotypes, with a dimension equal to the number of animals by the number of markers. The blending and weighting parameters for the genomic information were the default values in the PREGSf90 software (<inline-formula id="inf40">
<mml:math id="m43">
<mml:mi>&#x3b1;</mml:mi>
</mml:math>
</inline-formula> and <inline-formula id="inf41">
<mml:math id="m44">
<mml:mi>&#x3b2;</mml:mi>
</mml:math>
</inline-formula> equal to 0.95 and 0.05, respectively, and <inline-formula id="inf42">
<mml:math id="m45">
<mml:mi>&#x3c4;</mml:mi>
</mml:math>
</inline-formula> and <inline-formula id="inf43">
<mml:math id="m46">
<mml:mi>&#x3c9;</mml:mi>
</mml:math>
</inline-formula> equal to 1.0; <xref ref-type="bibr" rid="B40">Misztal et&#x20;al., 2018</xref>).</p>
</sec>
<sec id="s2-3-2">
<title>2.3.2 ssGBLUP Using SNPs and Haplotypes Combined in a Single Genomic Relationship Matrix</title>
<p>The model and assumptions in this approach are the same as described in <italic>ssGBLUP using SNPs</italic>. However, the <inline-formula id="inf44">
<mml:math id="m47">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> matrix used to construct the combined relationship in this model had both independent markers (i.e.,&#x20;non-blocked markers, which are SNPs out of the LD blocks) and haplotypes as pseudo-SNPs. To build the <inline-formula id="inf45">
<mml:math id="m48">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> matrix using haplotype information, the haplotype alleles were first converted to pseudo-SNPs, as in <xref ref-type="bibr" rid="B60">Teissier et&#x20;al. (2020)</xref>. Using this approach, if there were five unique haplotype alleles in a haploblock, five pseudo-SNPs were created for this haploblock. At the end, the number of copies of a specific pseudo-SNP allele were counted and coded as 0, 1, or 2 for each individual, similar to the codes used in <inline-formula id="inf46">
<mml:math id="m49">
<mml:mi mathvariant="normal">M</mml:mi>
</mml:math>
</inline-formula> (when creating the <inline-formula id="inf47">
<mml:math id="m50">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula>) as previously described based on individual SNPs. The pseudo-SNPs were subjected to the same QC steps as described above for individual&#x20;SNPs.</p>
</sec>
<sec id="s2-3-3">
<title>2.3.3 ssGBLUP Using Haplotypes</title>
<p>The model and assumptions in this approach were the same as described in <italic>ssGBLUP using SNPs</italic>. However, only haplotypes converted to pseudo-SNPs were used to create the <inline-formula id="inf48">
<mml:math id="m51">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> matrix used in the predictions, therefore, excluding non-blocked individual&#x20;SNPs.</p>
</sec>
<sec id="s2-3-4">
<title>2.3.4 ssGBLUP Using SNPs and Haplotypes Assigned to Two Different Genomic Relationship Matrices</title>
<p>The model used for these analyses was:<disp-formula id="equ4">
<mml:math id="m52">
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="bold">Xb</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold">Z</mml:mi>
<mml:msub>
<mml:mi mathvariant="bold">u</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi mathvariant="bold">Z</mml:mi>
<mml:msub>
<mml:mi mathvariant="bold">u</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold">e</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>where <inline-formula id="inf49">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf50">
<mml:math id="m54">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the random additive genetic effects of the first and second component of the overall GEBV, respectively, which, under this modeling, is equal to <inline-formula id="inf51">
<mml:math id="m55">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. All other vectors and matrices on this model are the same as described on the previous sections. The main assumption on this model is that the breeding value is divided into two uncorrelated components with their own covariance structure, being <inline-formula id="inf52">
<mml:math id="m56">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf53">
<mml:math id="m57">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, in which <inline-formula id="inf54">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf55">
<mml:math id="m59">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the hybrid relationship matrices with the same structure of the <inline-formula id="inf56">
<mml:math id="m60">
<mml:mi mathvariant="normal">H</mml:mi>
</mml:math>
</inline-formula> matrix described before. The only difference between <inline-formula id="inf57">
<mml:math id="m61">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf58">
<mml:math id="m62">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>is the <inline-formula id="inf59">
<mml:math id="m63">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> matrix that is combined with the pedigree relationship in each one of them, named as <inline-formula id="inf60">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf61">
<mml:math id="m65">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, respectively, containing the genomic relationships between the individuals based on single non-blocked SNPs and haplotypes, respectively. This parametrization was used to account for the fact that haplotypes and, therefore, the corresponding pseudo-SNPs, are more polymorphic than individual SNPs. Consequently, pseudo-SNPs could better capture the effect of large-sized QTL with lower allele frequency than individual SNPs and could have different distribution of their allele effects compared to individual&#x20;SNPs.</p>
</sec>
</sec>
<sec id="s2-4">
<title>2.4 Training and Validation Population Sets</title>
<p>The populations used in the genomic predictions were the pure breeds B, C, and E, defined as Breed_B, Breed_C, and Breed_E, respectively, and composite breeds Comp_2 and Comp_3. Only breeds Breed_B, Breed_C, and Breed_E were presented here because the genetic background simulated, i.e.,&#x20;the size of the founder population and generations of selection, was more divergent for these populations (<xref ref-type="fig" rid="F1">Figure&#x20;1</xref>). As breeds A and D had similar sizes of the founder populations and generations of selection when compared to breeds B and E, respectively, we observed similar results between breeds A and B and also D and E (data not shown).</p>
<p>The datasets (populations from the simulated EBV-based selection programs) were divided into training and validation sets to test the accuracy and bias of GEBVs. The training sets within each population were composed of 60,000 individuals with phenotypes randomly sampled from generations one to eight, and 8,000 of them also had genotypes for the simulated HD panel. The genotyped individuals in the training set were randomly sampled from generations four to seven. The validation populations were composed of 2,000 individuals randomly sampled from generations nine and ten and were also genotyped for the same panel. Generation eight was considered as a gap between training and validation populations in terms of genotypes. The whole pedigree (generations 1&#x2013;10) was used in all analyses. As we assume that validation individuals would not have phenotypes, their GEBVs were estimated based on the relationships of the validation cohort with the training set (with phenotypes and genotypes included in the analyses).</p>
</sec>
<sec id="s2-5">
<title>2.5 Evaluated Scenarios</title>
<p>Although the HD SNP panel datasets were first simulated, the main genomic predictions were performed using a medium density 50&#xa0;K SNP panel, which was designed based on randomly selected SNPs from the original HD panel. This step was performed because similar accuracies tend to be achieved when using a medium density SNP panel in sheep (<xref ref-type="bibr" rid="B41">Moghaddar et&#x20;al., 2017</xref>), as well as in other species (<xref ref-type="bibr" rid="B61">Binsbergen et&#x20;al., 2015</xref>; <xref ref-type="bibr" rid="B45">Ni et&#x20;al., 2017</xref>; <xref ref-type="bibr" rid="B17">Frischknecht et&#x20;al., 2018</xref>). The total number of SNPs selected for the 50&#xa0;K panel was 46,827, as currently available in the 50&#xa0;K SNP panel (for autosomal chromosomes) reported in the SNPchiMp v.3 platform (<xref ref-type="bibr" rid="B46">Nicolazzi et&#x20;al., 2015</xref>). The markers in the 50&#xa0;K SNP panel were randomly sampled within each autosome, and the number of SNPs per chromosome is reported in <xref ref-type="sec" rid="s11">Supplementary Material S1</xref>. In addition, previous analyses showed that both SNP and haplotype-based predictions based on the HD and 50&#xa0;K SNP panels were not statistically different (data not shown). Therefore, the haplotype blocks for all the prediction scenarios were created based on the 50&#xa0;K panel and the results for the HD SNP panel were presented as an additional scenario.</p>
<p>At the end, 11 scenarios were evaluated, which consisted of genomic predictions using: 1) SNPs from the 600&#xa0;K; 2) SNPs from the 50&#xa0;K; 3&#x2013;5) independent SNPs and pseudo-SNPs from haplotype blocks with LD equal to 0.1, 0.3, and 0.6 in a single relationship matrix (IPS_LD01, IPS_LD03, and IPS_LD06, respectively); 6&#x2013;8) only pseudo-SNPs from haplotype blocks with LD equal to 0.1, 0.3, and 0.6 (PS_LD01, PS_LD03, and PS_LD06, respectively); and 9&#x2013;11) independent SNPs and pseudo-SNPs from haplotype blocks with LD equal to 0.1, 0.3, and 0.6 in two different relationship matrices (IPS_2H_LD01, IPS_2H_LD03, and IPS_2H_LD06, respectively). All these scenarios were evaluated for two different heritability levels (moderate and low) and in each one of the five populations previously described (purebred and composite breeds with distinct Ne). Therefore, 110 different scenarios were evaluated in each one of the five replicates. A summary of the evaluated scenarios is shown in <xref ref-type="fig" rid="F2">Figure&#x20;2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Evaluated scenarios used in the genomic predictions with pseudo-single nucleotide polymorphisms (SNPs) from linkage disequilibrium (LD) blocks using independent and pseudo-SNPs in a single genomic relationship matrix (1H), and only pseudo-SNPs and independent and pseudo SNPs in two genomic relationship matrices (2H).</p>
</caption>
<graphic xlink:href="fgene-12-729867-g002.tif"/>
</fig>
</sec>
<sec id="s2-6">
<title>2.6 Scenario Comparisons</title>
<p>The statistics related to haplotype blocking strategies were compared between populations (pure and composite breeds) within each LD threshold to create the blocks (0.1, 0.3, and 0.6), and also, the LD thresholds were compared within each population to differentiate the haplotype block structures. These statistics are: average number of haploblocks, blocked SNPs, pseudo-SNPs before and after QC, non-blocked plus pseudo-SNPs after QC, and the additional computer time required by using pseudo-SNPs (e.g., SNPs phasing, haplotype blocking, and pseudo-SNP derivation). The GEBV accuracies and bias in each prediction scenario were compared within each population, to mimic population-specific (breed) genetic evaluation. Prediction accuracy was estimated as the Pearson correlation coefficient between the GEBVs and TBVs for the validation animals, for each replicate and scenario. Prediction bias was assessed as the deviation from one of the linear regression coefficients (<inline-formula id="inf62">
<mml:math id="m66">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) of the TBVs on the GEBVs (<inline-formula id="inf63">
<mml:math id="m67">
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mo>.</mml:mo>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>;</mml:mo>
<mml:mi mathvariant="normal">where</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>T</mml:mi>
<mml:mi>B</mml:mi>
<mml:mi>V</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi>B</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) in the validation population in each replicate and scenario.</p>
<p>A linear mixed model was used to test the effect of the population and LD level on the statistics from haplotype block strategies and the effect of marker information (SNP and haplotype prediction scenarios) on the accuracy and bias of GEBV prediction. The statistical model used was:<disp-formula id="equ5">
<mml:math id="m68">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf64">
<mml:math id="m69">
<mml:mrow>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the observation of the <italic>i</italic>th treatment on the <italic>j</italic>th repetition; <inline-formula id="inf65">
<mml:math id="m70">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the treatment effect, in which <italic>i</italic> is equal to Breed_B, Breed_C, Breed_E, Comp_2, and Comp_3 to compare the population effect over the statistics from haplotype block strategies within each LD threshold; equal to LD01, LD03, and LD06 to compare the effects of LD level over the statistics from haplotype block strategies within population; and equal to 600&#xa0;K, 50&#xa0;K, IPS_LD01, IPS_LD03, IPS_LD06, PS_LD01, PS_LD03, PS_LD06, IPS_2H_LD01, IPS_2H_LD03, and IPS_2H_LD06 to test the effect of marker information over the accuracy and bias of GEBV prediction within each population; <inline-formula id="inf66">
<mml:math id="m71">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the random effect of replicates which was assumed to follow <inline-formula id="inf67">
<mml:math id="m72">
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>b</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>; and <inline-formula id="inf68">
<mml:math id="m73">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the residual effect of the&#x20;model.</p>
<p>Replicate was used as a random effect in the model to account for the covariance between the scenarios, as the compared averages were obtained within the simulated populations in each replicate. This was done to reduce the occurrence of false negatives (Type-II error). Different covariance structures (<inline-formula id="inf69">
<mml:math id="m74">
<mml:mi mathvariant="normal">B</mml:mi>
</mml:math>
</inline-formula>) were evaluated (spherical, compound symmetry, simple autoregressive process, and unstructured covariance) to explain the covariances between replicates, and the structure that presented the lowest Akaike information criterion (AIC) and Bayesian information criterion (BIC) values was used in the final models for comparison purposes. After defining the appropriate covariance structure (which was not the same for all scenarios, with unstructured covariance being the best in the major part of the scenarios), the means of the <inline-formula id="inf70">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> levels were compared using the Tukey test at 5% of significance level. The &#x201c;nlme&#x201d; (<xref ref-type="bibr" rid="B50">Pinheiro et&#x20;al., 2021</xref>) and &#x201c;emmeans&#x201d; (<xref ref-type="bibr" rid="B32">Lenth, 2021</xref>) R packages were used to fit the models and compare the means, respectively, in the R environment (<xref ref-type="bibr" rid="B52">R Core Team, 2020</xref>).</p>
</sec>
</sec>
<sec id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Genetic Diversity and Genetic Parameters in the Simulated Populations</title>
<p>After the simulation process, several different Ne levels were observed in the recent populations studied (generations 1&#x2013;10 of pure and composite breeds under EBV-based selection). The total additive genetic effect variances estimated with the models that used two <inline-formula id="inf71">
<mml:math id="m76">
<mml:mi mathvariant="normal">H</mml:mi>
</mml:math>
</inline-formula> matrices (<italic>ssGBLUP using SNPs and Haplotypes Assigned to Two Different Genomic Relationship Matrices</italic>), taken as <inline-formula id="inf72">
<mml:math id="m77">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, and the residual variances were similar to the variances estimated with the models that fitted a single <inline-formula id="inf73">
<mml:math id="m78">
<mml:mi mathvariant="normal">H</mml:mi>
</mml:math>
</inline-formula> matrix (<italic>ssGBLUP using SNPs</italic>, <italic>ssGBLUP using SNPs and Haplotypes Combined in a Single Genomic Relationship Matrix</italic>, <italic>ssGBLUP Using Haplotypes</italic>) and similar to the variances estimated with the model that used only the pedigree relationship matrix (<italic>Simulated Traits</italic>; <xref ref-type="sec" rid="s11">Supplementary Materials S3</xref>, <xref ref-type="sec" rid="s11">S4</xref>). Therefore, for simplicity, only the genetic parameters estimated based on the pedigree relationship matrix are presented in <xref ref-type="table" rid="T1">Table&#x20;1</xref>. A population structure analysis based on principal components (PCs) of the <inline-formula id="inf74">
<mml:math id="m79">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> matrix using the SNPs from the 50&#xa0;K panel was also performed (<xref ref-type="sec" rid="s11">Supplementary Material S2</xref>). Individuals within the population were close to each other, and no clear clusters between populations existed at 95% confidence level based in the approximated unbiased test from a hierarchical clustering method using 10,000 bootstrap samples (<xref ref-type="bibr" rid="B57">Shimodaira, 2002</xref>; <xref ref-type="sec" rid="s11">Supplementary Material&#x20;S2</xref>).</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Average (SE) effective population size based on the linkage disequilibrium (Ne<sub>LD</sub>) and realized inbreeding (Ne<sub>Inb</sub>) methods, additive genetic variance (<inline-formula id="inf75">
<mml:math id="m80">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>a</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>), residual variance (<inline-formula id="inf76">
<mml:math id="m81">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>a</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>), and heritability (h<sup>2</sup>) estimates of the trait in simulated sheep populations.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Simulation</th>
<th align="center">Population<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</th>
<th align="center">Ne<sub>LD</sub>
<xref ref-type="table-fn" rid="Tfn2">
<sup>b</sup>
</xref>
</th>
<th align="center">Ne<sub>Inb</sub>
<xref ref-type="table-fn" rid="Tfn3">
<sup>c</sup>
</xref>
</th>
<th align="center">
<inline-formula id="inf77">
<mml:math id="m82">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="italic">&#x3c3;</mml:mi>
<mml:mi mathvariant="italic">a</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">
<inline-formula id="inf78">
<mml:math id="m83">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="italic">&#x3c3;</mml:mi>
<mml:mi mathvariant="italic">e</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">H<sup>2</sup>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Moderate h<sup>2</sup> (0.30)</td>
<td align="left">Breed_B</td>
<td align="char" char="(">110 (6)</td>
<td align="char" char="(">190 (17)</td>
<td align="char" char="(">27.12 (0.27)</td>
<td align="char" char="(">71.54 (0.10)</td>
<td align="char" char="(">0.27 (0.00)</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Breed_C</td>
<td align="char" char="(">379 (8)</td>
<td align="char" char="(">260 (15)</td>
<td align="char" char="(">28.09 (0.25)</td>
<td align="char" char="(">70.85 (0.26)</td>
<td align="char" char="(">0.29 (0.00)</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Breed_E</td>
<td align="char" char="(">359 (5)</td>
<td align="char" char="(">192 (6)</td>
<td align="char" char="(">27.45 (0.35)</td>
<td align="char" char="(">72.42 (0.34)</td>
<td align="char" char="(">0.28 (0.00)</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Comp_2</td>
<td align="char" char="(">644 (15)</td>
<td align="char" char="(">446 (7)</td>
<td align="char" char="(">25.82 (0.37)</td>
<td align="char" char="(">73.07 (0.25)</td>
<td align="char" char="(">0.26 (0.00)</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Comp_3</td>
<td align="char" char="(">466 (40)</td>
<td align="char" char="(">447 (53)</td>
<td align="char" char="(">26.80 (0.62)</td>
<td align="char" char="(">72.88 (0.50)</td>
<td align="char" char="(">0.27 (0.00)</td>
</tr>
<tr>
<td align="left">Moderate h<sup>2</sup> (0.10)</td>
<td align="left">Breed_B</td>
<td align="char" char="(">125 (8)</td>
<td align="char" char="(">94 (11)</td>
<td align="char" char="(">9.17 (0.26)</td>
<td align="char" char="(">90.30 (0.38)</td>
<td align="char" char="(">0.09 (0.00)</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Breed_C</td>
<td align="char" char="(">272 (11)</td>
<td align="char" char="(">120 (11)</td>
<td align="char" char="(">9.31 (0.28)</td>
<td align="char" char="(">89.91 (0.23)</td>
<td align="char" char="(">0.09 (0.00)</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Breed_E</td>
<td align="char" char="(">251 (22)</td>
<td align="char" char="(">119 (19)</td>
<td align="char" char="(">9.31 (0.23)</td>
<td align="char" char="(">90.38 (0.26)</td>
<td align="char" char="(">0.09 (0.00)</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Comp_2</td>
<td align="char" char="(">522 (32)</td>
<td align="char" char="(">259 (40)</td>
<td align="char" char="(">8.42 (0.27)</td>
<td align="char" char="(">91.13 (0.27)</td>
<td align="char" char="(">0.08 (0.00)</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Comp_3</td>
<td align="char" char="(">407 (32)</td>
<td align="char" char="(">235 (38)</td>
<td align="char" char="(">8.00 (0.29)</td>
<td align="char" char="(">91.90 (0.23)</td>
<td align="char" char="(">0.08 (0.00)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn1">
<label>a</label>
<p>Breed_B, Breed_C, and Breed_E: simulated pure breeds with different genetic backgrounds; Comp_2 and Comp_3: composite breeds based on two and three pure breeds, respectively.</p>
</fn>
<fn id="Tfn2">
<label>b</label>
<p>Estimated based on the re-arranged estimator present in <xref ref-type="bibr" rid="B59">Sved (1971)</xref>.</p>
</fn>
<fn id="Tfn3">
<label>c</label>
<p>Estimated based on the formula presented by <xref ref-type="bibr" rid="B14">Falconer and Mackay (1996)</xref>.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<sec id="s3-1-1">
<title>3.1.1 Ne and Genetic Parameters for the Simulation of a Trait With Moderate Heritability</title>
<p>The average Ne<sub>LD</sub> ranged between 110 and 644 (Breed_B and Comp_2, respectively), while the Ne<sub>Inb</sub> varied from 159 to 373 (Breed_B and composite breeds, respectively), being lower in pure breeds independently of the Ne measure (<xref ref-type="table" rid="T1">Table&#x20;1</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S3</xref>). The average additive genetic variance in the MH2 scenarios ranged from 25.82 (Comp_2) to 28.09 (Breed_C), while the residual variances ranged from 70.85 (Breed_C) to 73.07 (Comp_2). Average heritability estimates ranging from 0.26 (Comp_2) to 0.29 (Breed_C) were observed across populations, which are close to the global simulation parameters (heritability and phenotypic variance equal to 0.30 and 100, respectively).</p>
</sec>
<sec id="s3-1-2">
<title>3.1.2 Ne and Genetic Parameters for the Simulation of a Low Heritability Trait</title>
<p>The average Ne<sub>LD</sub> ranged from 125 (Breed_B) to 522 (Comp_2), while Ne<sub>Inb</sub> ranged between 94 and 259 for these same populations (<xref ref-type="table" rid="T1">Table&#x20;1</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S4</xref>). Average additive genetic variances ranging from 8.00 (Comp_3) to 9.31 (Breed_C and Breed_E) were observed. The average residual variances ranged from 90.30 (Breed_B) to 91.90 (Comp_3). In the LH2 scenarios, the average heritabilities were equal to 0.09 in the pure breeds and 0.08 in the composite breeds, which are close to the global simulation parameters (heritability and phenotypic variance equal to 0.10 and 100, respectively).</p>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 Statistics From Haplotype Blocks and Pseudo-SNPs: Moderate Heritability Trait</title>
<sec id="s3-2-1">
<title>3.2.1 Number of Blocks</title>
<p>The average number of blocks with two or more SNPs and the LD threshold equal to 0.1 ranged from 7,709.6 (Comp_2) to 8,607.6 (Comp_3), with Comp_2 and Breed_B showing similar and significantly lower number of blocks with this LD threshold level than the other populations (<xref ref-type="fig" rid="F3">Figure&#x20;3A</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S5</xref>). With the LD threshold equal to 0.3, the average number of blocks ranged from 145.0 (Comp_2) to 3,574.6 (Breed_B), and Breed_B showed significantly larger mean compared to the other populations (<xref ref-type="fig" rid="F3">Figure&#x20;3B</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S5</xref>). Only Breed_B had blocks with an LD threshold equal to 0.6, with an average equal to 23.8, which was statistically different from all the other populations (<xref ref-type="fig" rid="F3">Figure&#x20;3C</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S5</xref>). Within each population, the mean number of blocks from LD threshold levels of 0.1, 0.3, and 0.6 were statistically different for all populations, with the LD threshold equal to 0.1 being the largest, followed by the LD threshold equal to 0.3, and the 0.6 level yielding the lowest number of blocks.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Average number of blocks (Blocks) spanning two or more SNPs, markers within blocks (Blocked_SNPs), pseudo-SNPs (Pseudo_SNPs), pseudo-SNPs after quality control (PS_A_QC), non-blocked SNPs plus pseudo-SNPs after quality control (NB_PS_A_QC), and computing time to obtain the pseudo-SNPs (Duration_time) in the simulation for a trait with moderate heritability (h<sup>2</sup> &#x3d; 0.30). A, B, and C show the results for haplotype blocks with LD thresholds of 0.1, 0.3, and 0.6, respectively. Breed_B, Breed_C, and Breed_E: simulated pure breeds with different genetic backgrounds; Comp_2 and Comp_3: composite breeds from two and three pure breeds, respectively. The same lower- or upper-case letters mean no statistical difference comparing populations within LD thresholds and LD threshold across populations, respectively, at 5% significance level by the Tukey&#x20;test.</p>
</caption>
<graphic xlink:href="fgene-12-729867-g003.tif"/>
</fig>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Number of Blocked SNPs</title>
<p>The average number of blocked SNPs for the LD threshold equal to 0.1 varied between 17,122.2 (Comp_2) and 19,199.8 (Comp_3) (<xref ref-type="fig" rid="F3">Figure&#x20;3A</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S5</xref>), and for Comp_2, it was significantly lower than all the other populations. The average number of SNPs within blocks with an LD threshold equal to 0.3 ranged from 340.4 (Comp_2) to 8,195.4 (Breed_B) (<xref ref-type="fig" rid="F3">Figure&#x20;3B</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S5</xref>). The number of blocked SNPs for Breed_B was significantly higher than for the other populations (which did not differ among them). The average number of blocked SNPs with LD threshold equal to 0.6 in Breed_B was 56.8 (<xref ref-type="fig" rid="F3">Figure&#x20;3C</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S5</xref>) and was significantly greater, as no blocks were created for all the other populations.</p>
</sec>
<sec id="s3-2-3">
<title>3.2.3 Number of Pseudo-SNPs After Quality Control</title>
<p>After QC, the average number of pseudo-SNPs from blocks with an LD threshold equal to 0.1 was reduced, ranging from 35,524.6 (Comp_2) to 39,713 (Breed_E) (<xref ref-type="fig" rid="F3">Figure&#x20;3A</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S5</xref>). In general, Breed_B and Comp_2 were statistically similar and had lower averages compared to all other populations. The average number of pseudo-SNPs after QC with haploblocks constructed with the LD threshold of 0.3 was between 718.6 (Comp_2) and 16,259.4 (Breed_B), in which only Breed_B was statistically different from all other populations (<xref ref-type="fig" rid="F3">Figure&#x20;3B</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S5</xref>). With an LD threshold equal to 0.6, the average number of pseudo-SNPs for Breed_B was 91 and no pseudo-SNPs were generated with this LD threshold for all the other populations (<xref ref-type="fig" rid="F3">Figure&#x20;3C</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S5</xref>). The average number of pseudo-SNPs before QC is also shown in <xref ref-type="fig" rid="F3">Figure&#x20;3A</xref> and <xref ref-type="sec" rid="s11">Supplementary Material&#x20;S5</xref>.</p>
</sec>
<sec id="s3-2-4">
<title>3.2.4 Number of Non-blocked SNPs Plus Pseudo-SNPs After Quality Control</title>
<p>The average number of non-blocked plus pseudo-SNPs after QC varied from 64,987.0 (Breed_B) to 67,367.2 (Breed_E) when using blocks with an LD threshold of 0.1 (<xref ref-type="fig" rid="F3">Figure&#x20;3A</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S5</xref>). Breed_B and Comp_2 showed lower averages compared to all the other populations. Regarding the LD threshold of 0.3, the number of non-blocked plus pseudo-SNPs after QC ranged from 47,205.2 (Comp_2) to 54,891.0 (Breed_B) (<xref ref-type="fig" rid="F3">Figure&#x20;3B</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S5</xref>). For this LD threshold, the Breed_B average was statistically greater than all the other populations. The average number of non-blocked plus pseudo-SNPs after QC was equal to 46,867.8 for Breed_B and 46,827 for all the other populations when using an LD threshold of 0.6 to create the haploblocks (<xref ref-type="fig" rid="F3">Figure&#x20;3C</xref> and <xref ref-type="sec" rid="s11">Supplementary Material&#x20;S5</xref>).</p>
</sec>
<sec id="s3-2-5">
<title>3.2.5 Additional Time to Create Pseudo-SNPs</title>
<p>The average computing time to create the pseudo-SNPs (also considering the haplotype phasing and blocking) was between 8,800.6&#xa0;s (2&#xa0;h and 26&#xa0;min; Comp_2) and 22,650.0&#xa0;s (6&#xa0;h and 18&#xa0;min; Breed_B) with the LD threshold of 0.1 (<xref ref-type="fig" rid="F3">Figure&#x20;3A</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S5</xref>). For this LD threshold, the computing time for Breed_B was statistically similar to that in Breed_C, but significantly different from all the other populations. When using an LD threshold of 0.3 to create the blocks, the average computing time ranged from 675.4&#xa0;s (11&#xa0;min; Comp_2) to 2,935.0&#xa0;s (49&#xa0;min; Breed_B) (<xref ref-type="fig" rid="F3">Figure&#x20;3B</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S5</xref>). The computing time for Breed_B was statistically higher than all the other populations, which were not statistically different among them. The average computing time for pseudo-SNPs from blocks with an LD threshold equal to 0.6 ranged from 591.4 (10&#xa0;min) to 666.8&#xa0;s (11&#xa0;min) (Breed_C and Breed_B, respectively; <xref ref-type="fig" rid="F3">Figure&#x20;3C</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S5</xref>), and no statistical differences were observed across populations. The computing time compared across LD thresholds within the population showed that LD thresholds of 0.3 and 0.6 were statistically similar and lower than with the LD threshold of&#x20;0.1.</p>
</sec>
</sec>
<sec id="s3-3">
<title>3.3 Statistics From Haplotype Blocks and Pseudo SNPs: Low Heritability Trait</title>
<p>We have also checked the statistics from haplotype blocks and pseudo-SNPs in the low heritability trait scenarios because the simulation was done for each heritability level at a time. In general, the number of blocks, blocked SNPs, pseudo-SNPs before and after the QC, the number of non-blocked plus pseudo-SNPs after QC, and computing time to generate the pseudo-SNPs for a trait with a low heritability were similar to those for a trait with moderate heritability and are shown in <xref ref-type="fig" rid="F4">Figure&#x20;4</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S6</xref>. The results for the statistical comparisons in each one of these metrics for both populations, within each LD threshold, and for LD thresholds across populations were also similar between the LH2 and MH2 scenarios. The exceptions for the statistical comparisons under LH2 scenario was that the number of blocks in Breed_C and Breed_E would show a similar or lower average number of blocks, blocked SNPs, pseudo-SNPs after QC, and number of non-blocked plus pseudo-SNPs after QC than Breed_B, whereas the opposite would occur under the MH2 scenario. However, as pointed out before, the values were similar across the LH2 and MH2 scenarios. Therefore, the interpretation of the statistical comparisons for haplotype blocks in the MH2 scenario are also extended to&#x20;LH2.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Average number of blocks (Blocks) spanning two or more SNPs, markers within blocks (Blocked_SNPs), pseudo-SNPs (Pseudo_SNPs), pseudo-SNPs after quality control (PS_A_QC), non-blocked SNPs plus pseudo-SNPs after quality control (NB_PS_A_QC), and computing time to obtain the pseudo-SNPs (Comp_time) in the simulation for a trait with low heritability (h<sup>2</sup> &#x3d; 0.10). A, B, and C show the results for the haplotype blocks with LD thresholds of 0.1, 0.3, and 0.6, respectively. Breed_B, Breed_C, and Breed_E: simulated pure breeds with different genetic backgrounds; Comp_2 and Comp_3: composite breeds from two and three pure breeds, respectively. The same lower- or upper-case letters mean no statistical difference comparing populations within LD thresholds and LD threshold across populations, respectively, at 5% significance level based on the Tukey&#x20;test.</p>
</caption>
<graphic xlink:href="fgene-12-729867-g004.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>3.4 Accuracy and Bias of Genomic Predictions: Moderate Heritability Trait</title>
<sec id="s3-4-1">
<title>3.4.1 Pure Breed With Lower Genetic Diversity (Breed_B)</title>
<p>The average accuracy for GEBVs based on individual SNPs in the Breed_B was 0.54 and 0.55 for the 50 and 600&#xa0;K panels, respectively, whereas it varied from 0.48 (pseudo-SNPs from blocks with an LD threshold of 0.3, PS_LD03) to 0.54 (independent SNPs and pseudo-SNPs from blocks with an LD threshold of 0.6, IPS_LD06) using haplotypes (<xref ref-type="fig" rid="F5">Figure&#x20;5A</xref>, <xref ref-type="sec" rid="s11">Supplementary Material S7</xref>). In general, genomic predictions that used pseudo-SNPs and independent SNPs in one or two relationship matrices did not statistically differ from those with SNPs in the 50 and 600&#xa0;K panels. Using only pseudo-SNPs in the genomic predictions showed significantly lower accuracy than all other methods, when considering an LD threshold equal to 0.1 and 0.3 to create the blocks (PS_LD01 and PS_LD03, respectively). No predictions with PS_LD06 and IPS_2H_LD06 (independent SNPs and pseudo-SNPs from blocks with an LD threshold of 0.6 in two relationship matrices) were performed due to the low correlations observed between off-diagonal elements in <inline-formula id="inf79">
<mml:math id="m84">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mrow>
<mml:mn>22</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf80">
<mml:math id="m85">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> constructed with only pseudo-SNPs from haploblocks with an LD threshold of 0.6 (<xref ref-type="sec" rid="s11">Supplementary Material S8</xref>). The average GEBV bias was equal to &#x2212;0.09 and &#x2212;0.08 for the 50 and 600&#xa0;K SNP panels, respectively, whereas it ranged between &#x2212;0.20 (PS_LD03) and &#x2212;0.08 (IPS_2H_LD01) with haplotypes. No statistical differences were observed in the average bias when the two SNP panel densities or the independent and pseudo-SNP in one or two relationship matrices were used. PS_LD01 and PS_LD03 generated statistically more biased GEBVs than all the other scenarios.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Accuracies and bias of genomic predictions based on individual SNPs and haplotypes for the simulations of traits with moderate <bold>(A)</bold> and low <bold>(B)</bold> heritability (0.30 and 0.10, respectively). Breed_B, Breed_C, and Breed_E: simulated pure breeds with different genetic backgrounds; Comp_2 and Comp_3: composite breeds from two and three pure breeds, respectively. 600&#xa0;K: high-density panel; 50&#xa0;K: medium-density panel; IPS_LD01, IPS_LD03, and IPS_LD06: independent and pseudo-SNPs from blocks with LD thresholds of 0.1, 0.3, and 0.6, respectively, in a single genomic relationship matrix; PS_LD01, PS_LD03, and PS_LD06: only pseudo-SNPs from blocks with LD threshold of 0.1, 0.3, and 0.6, respectively; and IPS_2H_LD01, IPS_2H_LD03, and IPS_2H_LD06: independent and pseudo-SNPs from blocks with LD thresholds of 0.1, 0.3, and 0.6, respectively, in two genomic relationship matrices. Zero values for both accuracies and bias mean no results were obtained, due to poor quality of genomic information or no convergence of the genomic prediction models. The same lower-case letters mean no statistical difference comparing genomic prediction methods within population at 5% significance level based on the Tukey&#x20;test.</p>
</caption>
<graphic xlink:href="fgene-12-729867-g005.tif"/>
</fig>
</sec>
<sec id="s3-4-2">
<title>3.4.2 Pure Breed With Medium-Size Founder Population and Moderate Genetic Diversity (Breed_C)</title>
<p>The average accuracy observed in the Breed_C was equal to 0.53 and 0.54 with the 50 and 600&#xa0;K, respectively, while with haplotypes, it ranged from 0.25 (PS_LD03) to 0.52 (IPS_LD03) (<xref ref-type="fig" rid="F5">Figure&#x20;5A</xref>, <xref ref-type="sec" rid="s11">Supplementary Material S7</xref>). Similar to Breed_B, the PS_LD01 and PS_LD03 models yielded statistically less accurate GEBVs than all the other models, with PS_LD03 being the worst one. Fitting pseudo-SNPs and independent SNPs in one or two relationship matrices did not have statistical differences when compared with individual-SNP predictions. The IPS_2H_LD03 scenario did not converge during the genetic parameter estimation, and no pseudo-SNPs were generated for any haplotype method that used an LD threshold of 0.6 (IPS_LD06, PS_LD06, and IPS_2H_LD06). Consequently, no results were obtained for these scenarios. Average GEBV bias equal to &#x2212;0.05 and &#x2212;0.02 were observed for the 50 and 600&#xa0;K SNP panels, whereas in the haplotype-based predictions, it ranged from &#x2212;0.49 (PS_LD03) to &#x2212;0.03 (IPS_2H_LD01). PS_LD01 and PS_LD03 were statistically more biased than all the other scenarios (statistically similar among them).</p>
</sec>
<sec id="s3-4-3">
<title>3.4.3 Pure Breed With Larger Founder Population and Moderate Genetic Diversity (Breed_E)</title>
<p>The average accuracy was equal to 0.52 and 0.53 for the 50 and 600&#xa0;K SNP panel, respectively, while the haplotype-based approach yielded accuracy varying between 0.28 (PS_LD03) and 0.51 (IPS_LD03) in Breed_E (<xref ref-type="fig" rid="F5">Figure&#x20;5A</xref>, <xref ref-type="sec" rid="s11">Supplementary Material S7</xref>). Using only pseudo-SNPs from haplotype blocks with an LD threshold of 0.3 (PSLD03) yielded the less accurate genomic predictions, being statistically lower than all the other models (with similar accuracy among them). No blocks with an LD threshold equal to 0.6 were created in this population, and therefore, no predictions were obtained with the models that would use pseudo-SNPs from these blocks. For the GEBV bias, averages of &#x2212;0.09 and &#x2212;0.06 were observed for the 50 and 600&#xa0;K panels, respectively, ranging from &#x2212;0.53 (PS_LD03) to &#x2212;0.09 (IPS_2H_LD01) when haplotypes were fitted. Similar to the accuracy findings, the PSLD03 showed statistically lower average GEBV bias of prediction compared to all other models, showing the more biased predictions.</p>
</sec>
<sec id="s3-4-4">
<title>3.4.4 Composite Breed From Two Populations With High Genetic Diversity (Comp_2)</title>
<p>The average accuracy for the 50 and 600&#xa0;K SNP panels in Comp_2 were 0.41 and 0.42, respectively, with haplotype-based predictions ranging from 0.17 (PSLD03) to 0.41 (IPS_LD03) (<xref ref-type="fig" rid="F5">Figure&#x20;5A</xref>, <xref ref-type="sec" rid="s11">Supplementary Material S7</xref>). As observed in the pure breeds, there were no statistical differences between the predictions with SNPs based on both SNP density panels and the scenarios that fitted pseudo-SNPs and independent SNPs in one or two relationship matrices. Using only pseudo-SNPs to create the <inline-formula id="inf81">
<mml:math id="m86">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> matrix also provided statistically lower accuracy, with PS_LD03 yielding the worst results. No predictions were made with IPS_2H_LD03 in this population because of convergence problems during the genetic parameter estimation process. No pseudo-SNPs were obtained with the LD threshold of 0.6 and, consequently, no subsequent genomic prediction results. Average GEBV bias of &#x2212;0.14 and &#x2212;0.10 was observed for the 50 and 600&#xa0;K SNP panels, respectively, while the average GEBV bias ranged from &#x2212;0.62 (PS_LD03) to &#x2212;0.15 (IPS_2H_LD01) when fitting haplotypes. Statistically, more biased predictions were obtained only when pseudo-SNPs from haplotype blocks with an LD threshold of 0.3 were used (PS_LD03).</p>
</sec>
<sec id="s3-4-5">
<title>3.4.5 Composite Breed From Three Populations With High Genetic Diversity (Comp_3)</title>
<p>The average accuracy for the 50 and 600&#xa0;K SNP panels were 0.41 and 0,42, respectively, and with haplotype-based predictions, they ranged from 0.22 (PS_LD03) to 0.41 (IPS_LD03) (<xref ref-type="fig" rid="F5">Figure&#x20;5A</xref>, <xref ref-type="sec" rid="s11">Supplementary Material S7</xref>). The PS_LD01 and PS_LD03 scenarios yielded statistically lower accuracy than all the other methods (statistically similar among them). Similarly to Comp_2, no genomic predictions were performed for the IPS_2H_LD03 and models fitting pseudo-SNPs from blocks with an LD threshold of 0.6. The average GEBV bias was &#x2212;0.19 and &#x2212;0.14 for the 50 and 600&#xa0;K SNP panels, respectively, and ranged from &#x2212;0.60 (PS_LD03) to &#x2212;0.18 (IPS_LD01) for the haplotype-based predictions. Using only pseudo-SNPs from LD blocks constructed based on an LD threshold of 0.3 resulted in more biased GEBV predictions for the Comp_3 population.</p>
</sec>
</sec>
<sec id="s3-5">
<title>3.5 Accuracy and Bias of Genomic Predictions: Low Heritability Trait</title>
<p>The effects of fitting haplotypes in the genomic predictions under the LH2 scenarios were similar to those observed in the MH2 scenarios for all populations, with also similar average results (<xref ref-type="fig" rid="F5">Figure&#x20;5B</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S9</xref>). Therefore, the interpretations of the results for MH2 can be extended to the LH2 scenario, in which the worst results were observed for the PS_LD03 and similar accuracy and bias using SNPs or haplotypes (with independent SNPs) were observed. The GEBVs from the LH2 scenarios were less accurate and more biased than those from the MH2 scenarios within populations (e.g., lower accuracy and greater bias in LH2 within Breed_B), as would be expected due to the lower heritability of the trait. No GEBV predictions were made for the PS_LD06 and IPS_2H_LD06 for Breed_B due to the low correlation between the off-diagonal elements of the <inline-formula id="inf82">
<mml:math id="m87">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mrow>
<mml:mn>22</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf83">
<mml:math id="m88">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> created with pseudo-SNPs from blocks with an LD threshold of 0.6 (<xref ref-type="sec" rid="s11">Supplementary Material S10</xref>). No results for all scenarios fitting pseudo-SNPs from blocks with an LD threshold of 0.6 were obtained for Breed_C, Breed_E, Comp_2, and Comp_3 because no blocks were created based on this threshold.</p>
</sec>
</sec>
<sec id="s4">
<title>4. Discussion</title>
<p>We hypothesized that the predicted GEBV in populations with higher genetic diversity, such as composite sheep breeds (e.g., <xref ref-type="bibr" rid="B26">Kijas et&#x20;al., 2012</xref>; <xref ref-type="bibr" rid="B7">Brito et&#x20;al., 2017b</xref>; <xref ref-type="bibr" rid="B47">Oliveira et&#x20;al., 2020</xref>), could benefit from the use of haplotype-based rather than SNP-based genomic predictions, by obtaining GEBVs with higher accuracy and lower bias of prediction. Therefore, we investigated the impact of including haplotype information in ssGBLUP for populations with high genetic diversity, assessed based on the Ne metric, and different genetic background. Furthermore, we evaluated the performance of haplotype-based models by fitting the haplotypes as pseudo-SNPs in different ways under the ssGBLUP framework. For that, we considered only pseudo-SNPs to construct the genomic relationships and also two different relationship matrices (i.e.,&#x20;derived from individual SNPs and pseudo-SNPs from haplotype blocks), assuming no correlation between them. To evaluate our hypothesis, simulated data was used to calculate the true accuracy and bias of genomic predictions for simulated traits with moderate and low heritability level. These two sets of heritability levels comprise the major part of traits of interest in livestock breeding programs (e.g., growth, carcass, feed efficiency, reproductive performance, disease resistance, overall resilience).</p>
<sec id="s4-1">
<title>4.1 Genetic Diversity and Genetic Parameters</title>
<p>The genetic diversity and variance components were assessed in the subsets of the data used for the predictions to verify the consistency of the initial simulation parameters. In addition to the first three recent Ne idealized at the beginning of this study (100, 250, and 500), several other genetic diversity measures were obtained after the simulation process was finalized, which are measures of recent Ne (until five generations ago) based on LD (Ne<sub>LD</sub>) and on realized inbreeding (Ne<sub>Inb</sub>) (<xref ref-type="table" rid="T1">Table&#x20;1</xref> and <xref ref-type="sec" rid="s11">Supplementary Materials S3</xref>, <xref ref-type="sec" rid="s11">S4</xref>). Ne<sub>LD</sub> would be more useful in the absence of accurate pedigree information, as it relies on the <inline-formula id="inf84">
<mml:math id="m89">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msup>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> estimation in a pre-defined chromosomic segment size and was proposed for simpler population structures (e.g., random mating and no selection; <xref ref-type="bibr" rid="B59">Sved, 1971</xref>). However, we also calculated Ne<sub>Inb</sub> as an alternative indicator of Ne, because this estimate is based on the realized inbreeding and relies on the actual increase in population autozygosity (<xref ref-type="bibr" rid="B14">Falconer and Mackay, 1996</xref>).</p>
<p>One thousand and six hundred individuals from each one of the five populations (8,000 in total) were used to obtain the principal components (PCs) shown in <xref ref-type="sec" rid="s11">Supplementary Material S2</xref>, which actually explained a small proportion of the overall variance (1.71 and 2.13% for the first two and first three PCs, respectively). <xref ref-type="bibr" rid="B36">McVean (2009)</xref> highlighted several situations that can affect the structure and spatial distribution of the PCA using SNPs (e.g., current and recurrent bottlenecks, admixture, waves of expansion, sample size) and potentially cause bias in the scatter with the first PCs, especially if they explain a little proportion of the overall variance. <xref ref-type="bibr" rid="B53">Rao (1964)</xref> also indicated that inferences about structural relationships using the first PCs are only recommended when they explain a substantial amount of variation, which was not our case. Also, <xref ref-type="bibr" rid="B13">Deniskova et&#x20;al. (2016)</xref> found a sheep population with a lower Ne (176) more scattered in the first two PCs than populations with higher Ne (&#x3e;500), indicating the need for a third PC to observe differences within the high genetically diverse, similar to what we observed in this current study. The authors mentioned that a small founder population could be the reason for the lower Ne in the more scattered population along the first two PCs, and the Breed_B in our study (lower Ne) also had the smallest founder population. Another important point to highlight is that when using commercially available SNP chips, there tends to be ascertainment bias in the design of the SNP panels, which then contributes to a greater differentiation among populations (depending if they contributed or not to the SNP panel design) and crossbred/composite animals tend to have greater SNP diversity and be more scattered in the plots. This does not tend to happen when using simulated datasets. In summary, as it is not recommended to make inferences with PCs that are not significant (<xref ref-type="bibr" rid="B53">Rao, 1964</xref>; <xref ref-type="bibr" rid="B36">McVean, 2009</xref>), the Ne should be used to make conclusions about the genetic diversity of the simulated populations, with the PCs used only for the illustration of the population structure.</p>
<p>Both Ne measures showed values close to those observed for some terminal and composite sheep breeds (125&#x2013;974) as reported by <xref ref-type="bibr" rid="B7">Brito et&#x20;al. (2017b)</xref>, indicating that the simulation analyses resulted in datasets mimicking the genetic structure of commercial sheep populations. In addition to sheep, other species also present similar genetic diversity levels to some of the simulated populations used in this research, such as goats (Ne from 38 to 149; <xref ref-type="bibr" rid="B6">Brito et&#x20;al., 2015</xref>), beef cattle (Ne from 153 to 220; <xref ref-type="bibr" rid="B3">Biegelmeyer et&#x20;al., 2016</xref>), and dairy cattle (Ne from 58 to 120; <xref ref-type="bibr" rid="B35">Makanjuola et&#x20;al., 2020</xref>). The genetic parameters estimated after the simulation process were similar and consistent among replicates across all recent populations used for the subsequent analyses in both scenarios (MH2 and LH2; <xref ref-type="table" rid="T1">Table&#x20;1</xref> and <xref ref-type="sec" rid="s11">Supplementary Materials S3</xref>,&#x20;<xref ref-type="sec" rid="s11">S4</xref>).</p>
</sec>
<sec id="s4-2">
<title>4.2 Statistics From Haplotype Blocks and Pseudo-SNPs</title>
<p>The differences observed on the haplotype block statistics across the simulated populations within LD thresholds and also across LD thresholds within populations are a consequence of the genetic events experienced by them. The number and size of the LD blocks can vary according to recombination hotspots and evolutionary events such as mutation, selection, migration, and random drift (<xref ref-type="bibr" rid="B37">McVean et&#x20;al., 2004</xref>). In this context, a lower number of blocks with high LD thresholds would be expected in more genetically diverse populations, simply because in these populations, a large number of SNPs are expected to be excluded from all haploblocks, left to be considered as individual SNP effects. This was observed in Breed_B (less diverse, Ne ranging from 94 to 159) having a larger number of blocks not only when 0.6 was used as the LD threshold but also when the LD threshold was set to 0.3 in both MH2 and LH2 scenarios (<xref ref-type="fig" rid="F3">Figures 3</xref>, <xref ref-type="fig" rid="F4">4</xref> and <xref ref-type="sec" rid="s11">Supplementary Materials S5</xref>,&#x20;<xref ref-type="sec" rid="s11">S6</xref>).</p>
<p>The average number of blocks was similar (LH2, <xref ref-type="fig" rid="F4">Figure&#x20;4</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S7</xref>) or even lower (MH2, <xref ref-type="fig" rid="F3">Figure&#x20;3</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S6</xref>) in Breed_B compared to the other populations when the LD threshold was set to 0.1. The Big-LD method used in this study defines the LD blocks by using weights estimated based on the number of SNPs from all possible overlapping intervals (<xref ref-type="bibr" rid="B29">Kim et&#x20;al., 2018</xref>). Therefore, low LD thresholds could imply in similar intervals to derive the independent blocks regardless of the level of genetic diversity in populations derived from the same historical population (i.e.,&#x20;same species). When setting low LD thresholds to construct the LD-blocks, more intervals of linked SNPs are obtained as the number of blocks increase with less SNPs excluded (and vice versa). Therefore, this might explain the distribution of the number of blocks across populations with an LD threshold of 0.1. Consequently, a greater number of blocks are expected, as observed when comparing the number of blocks across LD thresholds (the number of blocks with an LD threshold of 0.1 &#x3e; 0.3 &#x3e; 0.6, <xref ref-type="fig" rid="F3">Figures 3</xref>, <xref ref-type="fig" rid="F4">4</xref> and <xref ref-type="sec" rid="s11">Supplementary Materials S5</xref>,&#x20;<xref ref-type="sec" rid="s11">S6</xref>).</p>
<p>The number of blocked SNPs and pseudo-SNPs before and after QC in both MH2 and LH2 (<xref ref-type="fig" rid="F3">Figures 3</xref>, <xref ref-type="fig" rid="F4">4</xref> and <xref ref-type="sec" rid="s11">Supplementary Materials S5</xref>, <xref ref-type="sec" rid="s11">S6</xref>) is a function of the genetic diversity level of the populations. Longer blocks with many SNPs are expected in less genetically diverse populations (<xref ref-type="bibr" rid="B21">Hayes et&#x20;al., 2003</xref>; <xref ref-type="bibr" rid="B63">Villumsen et&#x20;al., 2009</xref>; <xref ref-type="bibr" rid="B22">Hess et&#x20;al., 2017</xref>) likely due to selection and inbreeding, whereas more pseudo-SNPs (unique haplotypes) are expected in more genetically diverse populations (<xref ref-type="bibr" rid="B60">Teissier et&#x20;al., 2020</xref>), when the single SNPs out of the LD-clusters are not considered as a block, following the standard definition of haplotype block (<xref ref-type="bibr" rid="B18">Gabriel et&#x20;al., 2002</xref>). However, this also depends on the LD threshold used to create the haplotype blocks, as this pattern was clear only when LD was greater than&#x20;0.1.</p>
<p>Independently of the LD level used to create the blocks, the relative reduction in the number of pseudo-SNPs after QC was greater on the less genetically diverse population, with approximately 40% in Breed_B when the LD threshold was set to 0.6. The greatest reduction of pseudo-SNPs in populations with less genetic diversity was due to the low frequency of the haplotypes in this research, which agrees with the literature [e.g., based on simulated data (<xref ref-type="bibr" rid="B63">Villumsen et&#x20;al., 2009</xref>); in dairy cattle populations (<xref ref-type="bibr" rid="B22">Hess et&#x20;al., 2017</xref>; <xref ref-type="bibr" rid="B25">Karimi et&#x20;al., 2018</xref>); and in dairy goats (<xref ref-type="bibr" rid="B60">Teissier et&#x20;al., 2020</xref>)].</p>
<p>The additional computing time needed for genotype phasing, creating the haplotype blocks and the covariates for the models (<xref ref-type="bibr" rid="B16">Feitosa et&#x20;al., 2019</xref>; <xref ref-type="bibr" rid="B60">Teissier et&#x20;al., 2020</xref>), and running the genomic predictions (<xref ref-type="bibr" rid="B10">Cuyabano et&#x20;al., 2015</xref>; <xref ref-type="bibr" rid="B22">Hess et&#x20;al., 2017</xref>) have been indicated as the main drawbacks for the use of haplotypes in routine genomic predictions. In this study, the maximum additional computing time observed was approximately 7&#xa0;h (23,663.6&#xa0;s, Breed_B with LD equal to 0.1 under the LH2 scenario&#x2014;<xref ref-type="fig" rid="F4">Figure&#x20;4A</xref> and <xref ref-type="sec" rid="s11">Supplementary Material S6</xref>). <xref ref-type="bibr" rid="B22">Hess et&#x20;al. (2017)</xref> used marker effect models under Bayesian approaches and observed additional time of up to 27.2&#xa0;h for predictions with haplotypes derived from 37&#xa0;K SNPs with training and validation populations of about 30,000 dairy cattle individuals. <xref ref-type="bibr" rid="B10">Cuyabano et&#x20;al. (2015)</xref> reported that genomic predictions using Bayesian approaches and haplotypes took approximately from 1 to 46&#xa0;h, depending on the number of previously associated SNPs included in the GEBV predictions (1&#x2013;50&#xa0;K, respectively), with approximately 4,000 individuals in the training and validation populations. Differently from these studies, we used the ssGBLUP method, which showed consistent time for the predictions in the 50&#xa0;K SNP panel or when fitting haplotypes (as pseudo-SNPs) in the same <inline-formula id="inf85">
<mml:math id="m90">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> matrix. This was likely observed because the GEBVs are estimated directly based on the <inline-formula id="inf86">
<mml:math id="m91">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> matrix and the number of pseudo-SNPs added to the non-blocked SNPs (<xref ref-type="fig" rid="F3">Figures 3</xref>, <xref ref-type="fig" rid="F4">4</xref> and <xref ref-type="sec" rid="s11">Supplementary Materials S5</xref>, <xref ref-type="sec" rid="s11">S6</xref>) was not large enough to require longer time to create the genomic relationship matrices. As we calculated GEBVs for more than 62,000 individuals (genotyped and non-genotyped) using haplotype information with a relatively low increase of time, ssGBLUP is a feasible alternative for that purpose.</p>
<p>Interestingly, our results suggest that the computing time to obtain pseudo-SNPs in less genetically diverse populations is higher than in more diverse populations. This could be because more diverse populations have a smaller number of intervals with a determined LD level than populations with low genetic diversity, implying in less iterations for the algorithm to create the haplotype blocks. The smaller number of candidate intervals to create the blocks, leading to a lower computing time, might also explain the differences observed when comparing the LD levels within populations, with the computing time being significantly greater with an LD threshold of 0.1, followed by 0.3 and 0.6 LD thresholds.</p>
</sec>
<sec id="s4-3">
<title>4.3 Accuracy and Bias of Genomic Predictions</title>
<p>Genomic predictions based on whole genome sequence (WGS) data could be more advantageous because all the causal mutations are expected to be included in the data. However, practical results have shown no increase in GEBV accuracy when using WGS over HD (<xref ref-type="bibr" rid="B61">Binsbergen et&#x20;al., 2015</xref>; <xref ref-type="bibr" rid="B45">Ni et&#x20;al., 2017</xref>) or even medium density (&#x223c;50&#xa0;K) SNP panels (<xref ref-type="bibr" rid="B17">Frischknecht et&#x20;al., 2018</xref>). HD SNP panels were developed to better capture the LD between SNPs and QTLs and thus improve the ability to detect QTLs and obtain more accurate GEBVs (<xref ref-type="bibr" rid="B27">Kijas et&#x20;al., 2014</xref>), especially in more genetically diverse populations or even across-breed genomic predictions. However, the 50&#xa0;K SNP panel has shown a similar predictive ability to the HD even in highly diverse populations as in sheep (<xref ref-type="bibr" rid="B41">Moghaddar et&#x20;al., 2017</xref>). These findings corroborate with our results using the 50&#xa0;K SNP panel, regardless of the trait heritability. This suggests that both SNP panels (i.e.,&#x20;50 and 600&#xa0;K) are sufficient to capture the genetic relationships of the individuals, which is the base of the genomic predictions based on the ssGBLUP method (<xref ref-type="bibr" rid="B30">Legarra et&#x20;al., 2009</xref>; <xref ref-type="bibr" rid="B1">Aguilar et&#x20;al., 2010</xref>; <xref ref-type="bibr" rid="B34">Lourenco et&#x20;al., 2020</xref>). Therefore, we used the 50&#xa0;K SNP panel for haplotype-based genomic predictions.</p>
<p>Genomic predictions are expected to be more accurate with haplotypes instead of individual SNPs mainly because they are expected to be in greater LD with the QTL than are individual markers (<xref ref-type="bibr" rid="B8">Calus et&#x20;al., 2008</xref>; <xref ref-type="bibr" rid="B63">Villumsen et&#x20;al., 2009</xref>; <xref ref-type="bibr" rid="B9">Cuyabano et&#x20;al., 2014</xref>, <xref ref-type="bibr" rid="B10">2015</xref>; <xref ref-type="bibr" rid="B22">Hess et&#x20;al., 2017</xref>). In this context, <xref ref-type="bibr" rid="B8">Calus et&#x20;al. (2008)</xref> and <xref ref-type="bibr" rid="B63">Villumsen et&#x20;al. (2009)</xref> reported better results for the haplotype-based predictions of GEBVs than individual SNPs in simulated data, highlighting the possibility of improving both the accuracy and bias of genomic predictions. The Ne of the populations used by <xref ref-type="bibr" rid="B8">Calus et&#x20;al. (2008)</xref> and <xref ref-type="bibr" rid="B63">Villumsen et&#x20;al. (2009)</xref> is similar to the one in Breed_B (&#x223c;100). However, in this current study, haplotype-based models provided similar or lower accuracy and they were also similar or more biased than individual SNP-based models under both MH2 or LH2 scenarios (<xref ref-type="fig" rid="F5">Figure&#x20;5</xref> and <xref ref-type="sec" rid="s11">Supplementary Materials S7</xref>, <xref ref-type="sec" rid="s11">S9</xref>). This might be related to the LD level between SNP-QTL and haplotype-QTL and also the amount of information used to estimate the SNP and haplotype effects. <xref ref-type="bibr" rid="B8">Calus et&#x20;al. (2008)</xref> and <xref ref-type="bibr" rid="B63">Villumsen et&#x20;al. (2009)</xref> had fewer individuals (&#x223c;1,000), and their simulations were done with more general parameters compared to our study. The training set in this research for all populations was composed by 60,000 individuals with phenotypes, in which 8,000 of them were also genotyped. This amount of data is likely enough to estimate SNP effects and also the SNP-QTL LD properly. Thus, predictions with SNPs and haplotypes did not differ in some cases due to both of them capturing well the genetic relationships to achieve similar prediction results.</p>
<p>The correlations between off-diagonal, diagonal, and all elements in <inline-formula id="inf87">
<mml:math id="m92">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mrow>
<mml:mn>22</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf88">
<mml:math id="m93">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> created with pseudo-SNPs and independent SNPs together were similar to fit only individual SNPs in both SNP panel densities for all LD thresholds and in all populations, regardless of the heritability (<xref ref-type="sec" rid="s11">Supplementary Materials S8</xref>, <xref ref-type="sec" rid="s11">S10</xref>). Furthermore, the average, maximum, and minimum values of the diagonal elements in <inline-formula id="inf89">
<mml:math id="m94">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> created when combining pseudo-SNPs and independent SNPs were also similar to using only individual SNPs for both SNP panel densities in all scenarios investigated. Therefore, combining haplotypes and SNPs in a single <inline-formula id="inf90">
<mml:math id="m95">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> matrix captured the same information as fitting only individual SNPs, and, consequently, resulting in similar GEBV predictions.</p>
<p>Another reason for the similar genomic predictions when fitting individual SNPs and haplotypes might be the absence of or negligible epistatic interaction effects between SNP loci within haplotype blocks. In humans, a species with high Ne (<xref ref-type="bibr" rid="B48">Park, 2011</xref>), <xref ref-type="bibr" rid="B33">Liang et&#x20;al. (2020)</xref> showed that epistasis was the reason for increased accuracy with haplotypes over individual SNPs for health traits. In other words, a similar accuracy between SNPs and haplotypes was observed when there was negligible epistasis effect. The same authors also pointed out that predictions using haplotypes might only be worse than fitting individual SNPs because of a possible &#x201c;haplotype loss,&#x201d; which can happen when SNP effects are not accurately estimated by the haplotypes. As no epistatic effects are currently simulated by QMSim (<xref ref-type="bibr" rid="B56">Sargolzaei and Schenkel, 2009</xref>) and, therefore, were not simulated in the current study, different from our assumption that haplotypes could improve the predictions in more genetically diverse populations (Breed_C, Breed_E, Comp_2, and Comp_3), the accuracy and bias estimated based on haplotypes were similar or worse compared to fitting individual&#x20;SNPs.</p>
<p>Many studies based on real datasets have shown small improvements in the performance of haplotype-based genomic predictions. For instance, <xref ref-type="bibr" rid="B9">Cuyabano et&#x20;al. (2014)</xref> showed up to a 3.1% increase in the accuracy for milk protein when using LD-based haplotypes. <xref ref-type="bibr" rid="B10">Cuyabano et&#x20;al. (2015)</xref> also obtained gains in accuracy of up to 1.3% using pre-selected SNPs associated with the trait combined with the haplotypes as covariates in the models for production, fertility, and health traits. <xref ref-type="bibr" rid="B44">Mucha et&#x20;al. (2019)</xref> showed no differences in predictions with high-frequency haplotypes compared to SNPs when evaluating reproductive performance traits and somatic cell score in Polish dairy cattle. Additionally, <xref ref-type="bibr" rid="B16">Feitosa et&#x20;al. (2019)</xref> obtained nearly the same accuracy and bias for meat fatty acid (MFA) traits in Nellore cattle when fitting individual SNPs or haplotypes. These findings indicate that, even in instances where haplotypes are better than SNPs, the improvements are negligible or small. However, considerable improvements in haplotype-based predictions have also been reported in the literature for relatively less polygenic traits with known major genes or when using biological information to construct the haplotype blocks. <xref ref-type="bibr" rid="B64">Won et&#x20;al. (2020)</xref> reported a significant increase of 4.6% in GEBV accuracy with LD-clustering-based haplotypes for eye muscle area in Korean cattle. In Simmental cattle, <xref ref-type="bibr" rid="B65">Xu et&#x20;al. (2020)</xref> reported increases of 9.8% in carcass weight when incorporating haplotype information based on SNPs from functionally related genomic regions. <xref ref-type="bibr" rid="B60">Teissier et&#x20;al. (2020)</xref> reported an increase in accuracy of up to 22% when using haplotypes from fixed length or LD blocking strategies under an ssGBLUP setting. Based on these literature reports in livestock, it seems that haplotype predictions could provide better results when traits are oligogenic or affected by major genes, which are less common in livestock breeding goals. In addition, the presence of epistatic interactions in a real situation can also provide better results (<xref ref-type="bibr" rid="B33">Liang et&#x20;al., 2020</xref>). In this sense, using biological information to create the blocks of linked markers to make haplotype predictions can be an alternative to improve the genomic predictions in genetically diverse livestock populations. Unfortunately, there are limited real datasets of enough size with both phenotypes and genotypes for populations with large Ne that could be used for validating our findings.</p>
<p>It is worth mentioning that haplotype-based models without including the independent SNPs (markers not assigned to any block) to create the genomic relationships always provided the worst results, regardless of the LD threshold to create the haploblocks (0.1, 0.3, and 0.6). These models were also less accurate and more biased in all the populations, regardless of the genetic diversity level and heritability (<xref ref-type="fig" rid="F5">Figure&#x20;5</xref> and <xref ref-type="sec" rid="s11">Supplementary Materials S7</xref>, <xref ref-type="sec" rid="s11">S9</xref>). The worst results were obtained when fitting only pseudo-SNPs from blocks with an LD threshold of 0.3 (PSLD03) and in more genetically diverse populations (Breed_C, Breed_E, Comp_2, and Comp_3). This might have occurred because fitting only pseudo-SNPs from the haploblocks with two or more SNPs is not enough to consider all the important chromosomic regions influencing the trait of interest. The number of blocks, blocked SNPs, and pseudo-SNPs that were used to make the predictions were significantly lower with the LD level of 0.3 compared to 0.1 in both simulations (<xref ref-type="fig" rid="F3">Figures 3</xref>, <xref ref-type="fig" rid="F4">4</xref> and <xref ref-type="sec" rid="s11">Supplementary Materials S5</xref>, <xref ref-type="sec" rid="s11">S6</xref>), with this being likely the reason for the lowest accuracy and largest bias observed for PS_LD03. In this context, increasing the LD threshold to create the haploblocks have hampered the prediction with only haplotypes because a larger number of genomic markers were not considered to make the predictions. However, increasing the LD threshold to create the blocks and using the non-clustered SNPs together with the pseudo-SNPs did not affect the prediction results, presenting similar GEBV accuracies and bias compared to SNP-based predictions. In addition, the main differences in the properties of the <inline-formula id="inf91">
<mml:math id="m96">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> matrix were observed when only pseudo-SNPs from haploblocks with bigger LD thresholds were used, with lower correlations between off-diagonal and all elements in the <inline-formula id="inf92">
<mml:math id="m97">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mrow>
<mml:mn>22</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf93">
<mml:math id="m98">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> matrices and differences in the maximum and minimum values of the diagonal elements of the <inline-formula id="inf94">
<mml:math id="m99">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> (<xref ref-type="sec" rid="s11">Supplementary Materials S8</xref>, <xref ref-type="sec" rid="s11">S10</xref>). Therefore, independently of the LD threshold used to create the haploblocks, we recommend using the non-clustered SNPs with pseudo-SNPs from multi-marker haploblocks to make haplotype-based predictions, as well as in genome-wide association studies (GWAS) using haplotypes, because these variants may play an important&#x20;role.</p>
<p>Separating the independent and pseudo-SNPs in two different random effects, with no shared covariances structures, did not significantly impact the genomic predictions, but had a computational cost. The genetic parameter estimation and GEBV prediction required more computing time using these two genetic components in the model, with more iterations and greater time in each iteration than the other models (data not shown), sometimes leading to no convergence of the solutions (IPS_2H_LD03 in the Breed_C, Comp_2, and Comp_3 under MH2). The model with pseudo-SNPs and independent SNPs in two genetic components is more complex, and the convergence difficulty might suggest poor model parametrization, potentially because the random effects were assumed to be uncorrelated. This fact can be confirmed by high correlations (above than 0.90) between the inverted <inline-formula id="inf95">
<mml:math id="m100">
<mml:mi mathvariant="normal">H</mml:mi>
</mml:math>
</inline-formula> matrices with non-clustered SNPs and pseudo-SNPs (data not shown). Although increased computational time was a common problem in both heritability levels, convergence was achieved in all analyses with low heritability. Our findings suggest that a single <inline-formula id="inf96">
<mml:math id="m101">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> matrix with individual SNPs is enough to capture the QTL variation, regardless of the genetic diversity and heritability. Nonetheless, using two uncorrelated genetic components can be useful in other situations such as fitting SNPs and structural variants (e.g., copy number variation&#x2014;CNVs) in the same&#x20;model.</p>
</sec>
</sec>
<sec id="s5">
<title>5 Conclusion</title>
<p>Haplotype-based models did not improve the performance of genomic prediction of breeding values in genetically diverse populations (assumed as Ne &#x3e; 150) under ssGBLUP settings. A medium-density 50&#xa0;K SNP panel provided similar results to the high-density panel for the genomic predictions using individual SNPs or haplotypes, regardless of the heritability and genetic diversity levels. ssGBLUP can be used to predict breeding values for both genotyped and non-genotyped individuals using haplotype information in large datasets with no increase in computing time when fitting a single genomic relationship matrix.</p>
</sec>
</body>
<back>
<sec id="s6">
<title>Data Availability Statement</title>
<p>The simulated datasets used and the pipelines developed to carry out this research are available upon request.</p>
</sec>
<sec id="s7">
<title>Author Contributions</title>
<p>AA, PC, HO, and LB: conception of the work. AA: data simulation and data analyses. AA, PC, HO, and LB: interpretation of the results. AA, HO, and LB: drafted the manuscript. AA, PC, HO, RV, FS, DL, and LB: critical revision of the manuscript. AA, PC, HO, RV, FS, DL, and LB: final approval of the version to be published. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>This study was funded by Purdue University (West Lafayette, IN, United&#x20;States), State University of Southwestern Bahia (Itapetininga, BA, Brazil), and the <italic>Coordena&#xe7;&#xe3;o de Aperfei&#xe7;oamento de Pessoal de N&#xed;vel Superior&#x2014;Brazil</italic> (CAPES) Award Number&#x20;001.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>We acknowledge the Dr. Brito&#x2019;s Lab at Purdue University for&#x20;providing the scientific support to develop this research and researchers from Purdue University and State University of Southwestern Bahia for providing training to the first author and the infrastructure and resources needed for the research. We also acknowledge the National Development Council Scientific Technological (Conselho Nacional de Desenvolvimeno Cientifico e Tecnologico&#x2013;CNPq) for the fellowship.</p>
</ack>
<sec id="s11">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2021.729867/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2021.729867/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table4.DOC" id="SM1" mimetype="application/DOC" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table6.DOC" id="SM2" mimetype="application/DOC" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table7.DOC" id="SM3" mimetype="application/DOC" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table8.DOCX" id="SM4" mimetype="application/DOCX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table3.DOC" id="SM5" mimetype="application/DOC" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table5.DOC" id="SM6" mimetype="application/DOC" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table2.DOCX" id="SM7" mimetype="application/DOCX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table10.DOCX" id="SM8" mimetype="application/DOCX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table9.DOC" id="SM9" mimetype="application/DOC" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aguilar</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Misztal</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>D. L.</given-names>
</name>
<name>
<surname>Legarra</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tsuruta</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lawlor</surname>
<given-names>T. J.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Hot Topic: A Unified Approach to Utilize Phenotypic, Full Pedigree, and Genomic Information for Genetic Evaluation of Holstein Final Scorefied Approach to Utilize Phenotypic, Full Pedigree, and Genomic Information for Genetic Evaluation of Holstein FInal Score</article-title>. <source>J.&#x20;Dairy Sci.</source> <volume>93</volume>, <fpage>743</fpage>&#x2013;<lpage>752</lpage>. <pub-id pub-id-type="doi">10.3168/jds.2009-2730</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>AnimalQTLdb</surname>
</name>
</person-group> (<year>2019</year>). <article-title>QTL Data Base for Sheep by Number of Chromosome</article-title>. <comment>Availableat: <ext-link ext-link-type="uri" xlink:href="https://www.animalgenome.org/cgi-bin/QTLdb/OA/summary?summ=chro&amp;qtl=2,325&amp;pub=158&amp;trait=251">https://www.animalgenome.org/cgi-bin/QTLdb/OA/summary?summ&#x3d;chro&#x26;qtl&#x3d;2,325&#x26;pub&#x3d;158&#x26;trait&#x3d;251</ext-link> (Accessed April 15, 2020)</comment>. </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Biegelmeyer</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Gulias-Gomes</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Caetano</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Steibel</surname>
<given-names>J.&#x20;P.</given-names>
</name>
<name>
<surname>Cardoso</surname>
<given-names>F. F.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Linkage Disequilibrium, Persistence of Phase and Effective Population Size Estimates in Hereford and Braford Cattle</article-title>. <source>BMC Genet.</source> <volume>17</volume>, <fpage>32</fpage>. <pub-id pub-id-type="doi">10.1186/s12863-016-0339-8</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bohmanova</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sargolzaei</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schenkel</surname>
<given-names>F. S.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Characteristics of Linkage Disequilibrium in North American Holsteins</article-title>. <source>BMC Genomics</source> <volume>11</volume>, <fpage>421</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-11-421</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brito</surname>
<given-names>L. F.</given-names>
</name>
<name>
<surname>Clarke</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Mcewan</surname>
<given-names>J.&#x20;C.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>S. P.</given-names>
</name>
<name>
<surname>Pickering</surname>
<given-names>N. K.</given-names>
</name>
<name>
<surname>Bain</surname>
<given-names>W. E.</given-names>
</name>
<etal/>
</person-group> (<year>2017a</year>). <article-title>Prediction of Genomic Breeding Values for Growth, Carcass and Meat Quality Traits in a Multi-Breed Sheep Population Using a HD SNP Chip</article-title>. <source>BMC Genet.</source> <volume>18</volume>, <fpage>7</fpage>&#x2013;<lpage>24</lpage>. <pub-id pub-id-type="doi">10.1186/s12863-017-0476-8</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brito</surname>
<given-names>L. F.</given-names>
</name>
<name>
<surname>Jafarikia</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Grossi</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Kijas</surname>
<given-names>J.&#x20;W.</given-names>
</name>
<name>
<surname>Porto-Neto</surname>
<given-names>L. R.</given-names>
</name>
<name>
<surname>Ventura</surname>
<given-names>R. V.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Characterization of Linkage Disequilibrium, Consistency of Gametic Phase and Admixture in Australian and Canadian Goats</article-title>. <source>BMC Genet.</source> <volume>16</volume>, <fpage>67</fpage>. <pub-id pub-id-type="doi">10.1186/s12863-015-0220-1</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brito</surname>
<given-names>L. F.</given-names>
</name>
<name>
<surname>Mcewan</surname>
<given-names>J.&#x20;C.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>S. P.</given-names>
</name>
<name>
<surname>Pickering</surname>
<given-names>N. K.</given-names>
</name>
<name>
<surname>Bain</surname>
<given-names>W. E.</given-names>
</name>
<name>
<surname>Dodds</surname>
<given-names>K. G.</given-names>
</name>
<etal/>
</person-group> (<year>2017b</year>). <article-title>Genetic Diversity of a New&#x20;Zealand Multi-Breed Sheep Population and Composite Breed&#x27;s History Revealed by a High-Density SNP Chip</article-title>. <source>BMC Genet.</source> <volume>18</volume>, <fpage>25</fpage>. <pub-id pub-id-type="doi">10.1186/s12863-017-0492-8</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Calus</surname>
<given-names>M. P. L.</given-names>
</name>
<name>
<surname>Meuwissen</surname>
<given-names>T. H. E.</given-names>
</name>
<name>
<surname>de Roos</surname>
<given-names>A. P. W.</given-names>
</name>
<name>
<surname>Veerkamp</surname>
<given-names>R. F.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Accuracy of Genomic Selection Using Different Methods to Define Haplotypesfine Haplotypes</article-title>. <source>Genetics</source> <volume>178</volume>, <fpage>553</fpage>&#x2013;<lpage>561</lpage>. <pub-id pub-id-type="doi">10.1534/genetics.107.080838</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cuyabano</surname>
<given-names>B. C.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Lund</surname>
<given-names>M. S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Genomic Prediction of Genetic merit Using LD-Based Haplotypes in the Nordic Holstein Population</article-title>. <source>BMC Genomics</source> <volume>15</volume>, <fpage>1171</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-15-1171</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cuyabano</surname>
<given-names>B. C.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Lund</surname>
<given-names>M. S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Selection of Haplotype Variables from a High-Density Marker Map for Genomic Prediction</article-title>. <source>Genet. Sel. Evol.</source> <volume>47</volume>, <fpage>61</fpage>. <pub-id pub-id-type="doi">10.1186/s12711-015-0143-3</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Daetwyler</surname>
<given-names>H. D.</given-names>
</name>
<name>
<surname>Kemper</surname>
<given-names>K. E.</given-names>
</name>
<name>
<surname>Van Der Werf</surname>
<given-names>J.&#x20;H. J.</given-names>
</name>
<name>
<surname>Hayes</surname>
<given-names>B. J.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Components of the Accuracy of Genomic Prediction in a Multi-Breed Sheep Population1</article-title>. <source>J.&#x20;Anim. Sci.</source> <volume>90</volume>, <fpage>3375</fpage>&#x2013;<lpage>3384</lpage>. <pub-id pub-id-type="doi">10.2527/jas.2011-4557</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>de Oliveira</surname>
<given-names>H. R.</given-names>
</name>
<name>
<surname>Brito</surname>
<given-names>L. F.</given-names>
</name>
<name>
<surname>Sargolzaei</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>e Silva</surname>
<given-names>F. F.</given-names>
</name>
<name>
<surname>Jamrozik</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lourenco</surname>
<given-names>D. A. L.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Impact of Including Information from Bulls and Their Daughters in the Training Population of Multiple&#x2010;step Genomic Evaluations in Dairy Cattle: A Simulation Study</article-title>. <source>J.&#x20;Anim. Breed. Genet.</source> <volume>136</volume>, <fpage>441</fpage>&#x2013;<lpage>452</lpage>. <pub-id pub-id-type="doi">10.1111/jbg.12407</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deniskova</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Dotsev</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lushihina</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Shakhin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kunz</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Medugorac</surname>
<given-names>I.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Population Structure and Genetic Diversity of Sheep Breeds in the Kyrgyzstan</article-title>. <source>Front. Genet.</source> <volume>10</volume>, <fpage>1</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2019.01311</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Falconer</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Mackay</surname>
<given-names>T. F. C.</given-names>
</name>
</person-group> (<year>1996</year>). <source>Introduction to Quantitative Genetics</source>. <publisher-loc>Essex, UK</publisher-loc>: <publisher-name>Longman</publisher-name>, <fpage>4</fpage>. </citation>
</ref>
<ref id="B15">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>FarmIQ</surname>
</name>
</person-group> (<year>2013</year>). <article-title>Release of a High-Density SNP Genotyping Chip for the Sheep Genome</article-title>. <comment>Availableat: <ext-link ext-link-type="uri" xlink:href="http://www.farmiq.co.nz/whatsnew/news/release-high-densitysnp-genotyping-chip-sheep-genome">http://www.farmiq.co.nz/whatsnew/news/release-high-densitysnp-genotyping-chip-sheep-genome</ext-link> (Access June 6, 2020)</comment>. </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feitosa</surname>
<given-names>F. L. B.</given-names>
</name>
<name>
<surname>Pereira</surname>
<given-names>A. S. C.</given-names>
</name>
<name>
<surname>Amorim</surname>
<given-names>S. T.</given-names>
</name>
<name>
<surname>Peripolli</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Silva</surname>
<given-names>R. M. d. O.</given-names>
</name>
<name>
<surname>Braz</surname>
<given-names>C. U.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Comparison between Haplotype&#x2010;based and Individual Snp&#x2010;based Genomic Predictions for Beef Fatty Acid Profile in Nelore Cattle</article-title>. <source>J.&#x20;Anim. Breed. Genet.</source> <volume>137</volume>, <fpage>468</fpage>&#x2013;<lpage>476</lpage>. <pub-id pub-id-type="doi">10.1111/jbg.12463</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Frischknecht</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Meuwissen</surname>
<given-names>T. H. E.</given-names>
</name>
<name>
<surname>Bapst</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Seefried</surname>
<given-names>F. R.</given-names>
</name>
<name>
<surname>Flury</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Garrick</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Short Communication: Genomic Prediction Using Imputed Whole-Genome Sequence Variants in Brown Swiss Cattle</article-title>. <source>J.&#x20;Dairy Sci.</source> <volume>101</volume>, <fpage>1292</fpage>&#x2013;<lpage>1296</lpage>. <pub-id pub-id-type="doi">10.3168/jds.2017-12890</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gabriel</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Schaffner</surname>
<given-names>S. F.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Moore</surname>
<given-names>J.&#x20;M.</given-names>
</name>
<name>
<surname>Roy</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Blumenstiel</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2002</year>). <article-title>The Structure of Haplotype Blocks in the Human Genome</article-title>. <source>Sci</source> <volume>296</volume>, <fpage>2225</fpage>&#x2013;<lpage>2229</lpage>. <pub-id pub-id-type="doi">10.1126/science.1069424</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guarini</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Lourenco</surname>
<given-names>D. A. L.</given-names>
</name>
<name>
<surname>Brito</surname>
<given-names>L. F.</given-names>
</name>
<name>
<surname>Sargolzaei</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Baes</surname>
<given-names>C. F.</given-names>
</name>
<name>
<surname>Miglior</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Comparison of Genomic Predictions for Lowly Heritable Traits Using Multi-step and Single-step Genomic Best Linear Unbiased Predictor in Holstein Cattle</article-title>. <source>J.&#x20;Dairy Sci.</source> <volume>101</volume>, <fpage>8076</fpage>&#x2013;<lpage>8086</lpage>. <pub-id pub-id-type="doi">10.3168/jds.2017-14193</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guarini</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Lourenco</surname>
<given-names>D. A. L.</given-names>
</name>
<name>
<surname>Brito</surname>
<given-names>L. F.</given-names>
</name>
<name>
<surname>Sargolzaei</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Baes</surname>
<given-names>C. F.</given-names>
</name>
<name>
<surname>Miglior</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Genetics and Genomics of Reproductive Disorders in Canadian Holstein Cattle</article-title>. <source>J.&#x20;Dairy Sci.</source> <volume>102</volume>, <fpage>1341</fpage>&#x2013;<lpage>1353</lpage>. <pub-id pub-id-type="doi">10.3168/jds.2018-15038</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hayes</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Visscher</surname>
<given-names>P. M.</given-names>
</name>
<name>
<surname>McPartlan</surname>
<given-names>H. C.</given-names>
</name>
<name>
<surname>Goddard</surname>
<given-names>M. E.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Novel Multilocus Measure of Linkage Disequilibrium to Estimate Past Effective Population Size</article-title>. <source>Genome Res.</source> <volume>13</volume>, <fpage>635</fpage>&#x2013;<lpage>643</lpage>. <pub-id pub-id-type="doi">10.1101/gr.387103</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hess</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Druet</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hess</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Garrick</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Fixed-length Haplotypes Can Improve Genomic Prediction Accuracy in an Admixed Dairy Cattle Population</article-title>. <source>Genet. Sel. Evol.</source> <volume>49</volume>, <fpage>54</fpage>. <pub-id pub-id-type="doi">10.1186/s12711-017-0329-y</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hill</surname>
<given-names>W. G.</given-names>
</name>
<name>
<surname>Robertson</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>1968</year>). <article-title>Linkage Disequilibrium in Finite Populations</article-title>. <source>Theoret. Appl. Genet.</source> <volume>38</volume>, <fpage>226</fpage>&#x2013;<lpage>231</lpage>. <pub-id pub-id-type="doi">10.1007/BF01245622</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Schmidt</surname>
<given-names>R. H.</given-names>
</name>
<name>
<surname>Reif</surname>
<given-names>J.&#x20;C.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Haplotype-based Genome-wide Prediction Models Exploit Local Epistatic Interactions Among Markers</article-title>. <source>G</source> <volume>8</volume>, <fpage>1687</fpage>&#x2013;<lpage>1699</lpage>. <pub-id pub-id-type="doi">10.1534/g3.117.300548</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karimi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Sargolzaei</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Robinson</surname>
<given-names>J.&#x20;A. B.</given-names>
</name>
<name>
<surname>Schenkel</surname>
<given-names>F. S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Assessing Haplotype-Based Models for Genomic Evaluation in Holstein Cattle</article-title>. <source>Can. J.&#x20;Anim. Sci.</source> <volume>98</volume>, <fpage>750</fpage>&#x2013;<lpage>759</lpage>. <pub-id pub-id-type="doi">10.1139/cjas-2018-0009</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kijas</surname>
<given-names>J.&#x20;W.</given-names>
</name>
<name>
<surname>Lenstra</surname>
<given-names>J.&#x20;A.</given-names>
</name>
<name>
<surname>Hayes</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Boitard</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Porto Neto</surname>
<given-names>L. R.</given-names>
</name>
<name>
<surname>San Cristobal</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Genome-Wide Analysis of the World&#x27;s Sheep Breeds Reveals High Levels of Historic Mixture and Strong Recent Selection</article-title>. <source>Plos Biol.</source> <volume>10</volume>, <fpage>e1001258</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pbio.1001258</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kijas</surname>
<given-names>J.&#x20;W.</given-names>
</name>
<name>
<surname>Porto-Neto</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Dominik</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Reverter</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bunch</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>McCulloch</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Linkage Disequilibrium over Short Physical Distances Measured in Sheep Using a High-Density SNP Chip</article-title>. <source>Anim. Genet.</source> <volume>45</volume>, <fpage>754</fpage>&#x2013;<lpage>757</lpage>. <pub-id pub-id-type="doi">10.1111/age.12197</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Brossard</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Roshandel</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Paterson</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Bull</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Yoo</surname>
<given-names>Y. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Gpart: Human Genome Partitioning and Visualization of High-Density SNP Data by Identifying Haplotype Blocks</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>4419</fpage>&#x2013;<lpage>4421</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz308</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Cho</surname>
<given-names>C.-S.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S.-R.</given-names>
</name>
<name>
<surname>Bull</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Yoo</surname>
<given-names>Y. J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A New Haplotype Block Detection Method for Dense Genome Sequencing Data Based on Interval Graph Modeling of Clusters of Highly Correlated SNPs</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>388</fpage>&#x2013;<lpage>397</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx609</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Legarra</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Aguilar</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Misztal</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>A Relationship Matrix Including Full Pedigree and Genomic Information</article-title>. <source>J.&#x20;Dairy Sci.</source> <volume>92</volume>, <fpage>4656</fpage>&#x2013;<lpage>4663</lpage>. <pub-id pub-id-type="doi">10.3168/jds.2009-2061</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Legarra</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Christensen</surname>
<given-names>O. F.</given-names>
</name>
<name>
<surname>Aguilar</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Misztal</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Single Step, a General Approach for Genomic Selection</article-title>. <source>Livestock Sci.</source> <volume>166</volume>, <fpage>54</fpage>&#x2013;<lpage>65</lpage>. <pub-id pub-id-type="doi">10.1016/j.livsci.2014.04.029</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Lenth</surname>
<given-names>R. V.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Emmeans: Estimated Marginal Means, Aka Least-Squares Means</article-title>. <comment>R package version 1.5.4</comment>. <comment>Availableat: <ext-link ext-link-type="uri" xlink:href="https://CRAN.R-project.org/package=emmeans">https://CRAN.R-project.org/package&#x3d;emmeans</ext-link>
</comment>. </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Prakapenka</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Da</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Haplotype Analysis of Genomic Prediction Using Structural and Functional Genomic Information for Seven Human Phenotypes</article-title>. <source>Front. Genet.</source> <volume>11</volume>, <fpage>1</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2020.588907</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lourenco</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Legarra</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tsuruta</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Masuda</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Aguilar</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Misztal</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Single-step Genomic Evaluations from Theory to Practice: Using SNP Chips and Sequence Data in BLUPF90</article-title>. <source>Genes</source> <volume>11</volume>, <fpage>790</fpage>. <pub-id pub-id-type="doi">10.3390/genes11070790</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Makanjuola</surname>
<given-names>B. O.</given-names>
</name>
<name>
<surname>Miglior</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Abdalla</surname>
<given-names>E. A.</given-names>
</name>
<name>
<surname>Maltecca</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Schenkel</surname>
<given-names>F. S.</given-names>
</name>
<name>
<surname>Baes</surname>
<given-names>C. F.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Effect of Genomic Selection on Rate of Inbreeding and Coancestry and Effective Population Size of Holstein and Jersey Cattle Populations</article-title>. <source>J.&#x20;Dairy Sci.</source> <volume>103</volume>, <fpage>5183</fpage>&#x2013;<lpage>5199</lpage>. <pub-id pub-id-type="doi">10.3168/jds.2019-18013</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McVean</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>A Genealogical Interpretation of Principal Components Analysis</article-title>. <source>Plos Genet.</source> <volume>5</volume>, <fpage>e1000686</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pgen.1000686</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McVean</surname>
<given-names>G. A. T.</given-names>
</name>
<name>
<surname>Myers</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Hunt</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Deloukas</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bentley</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Donnelly</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>The fine-scale Structure of Recombination Rate Variation in the Human Genome</article-title>. <source>Science</source> <volume>304</volume>, <fpage>581</fpage>&#x2013;<lpage>584</lpage>. <pub-id pub-id-type="doi">10.1126/science.1092500</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meuwissen</surname>
<given-names>T. H. E.</given-names>
</name>
<name>
<surname>Hayes</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Goddard</surname>
<given-names>M. E.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Prediction of Total Genetic Value Using Genome-wide Dense Marker Maps</article-title>. <source>Genetics</source> <volume>157</volume>, <fpage>1819</fpage>&#x2013;<lpage>1829</lpage>. <pub-id pub-id-type="doi">10.1093/genetics/157.4.1819</pub-id> </citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meuwissen</surname>
<given-names>T. H.</given-names>
</name>
<name>
<surname>&#xd8;deg&#xe5;rd</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Andersen-Ranberg</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Grindflek</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>On the Distance of Genetic Relationships and the Accuracy of Genomic Prediction in Pig Breeding</article-title>. <source>Genet. Selection Evol.</source> <volume>46</volume>, <fpage>49</fpage>. <pub-id pub-id-type="doi">10.1186/1297-9686-46-49</pub-id> </citation>
</ref>
<ref id="B40">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Misztal</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Tsuruta</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lourenco</surname>
<given-names>D. A. L.</given-names>
</name>
<name>
<surname>Masuda</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Aguilar</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Legarra</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <source>Manual for BLUPF90 Family Programs</source>. <publisher-name>University of Georgia</publisher-name>. <comment>Availableat: <ext-link ext-link-type="uri" xlink:href="http://nce.ads.uga.edu/%20wiki/doku.php?id=documentation">http://nce.ads.uga.edu/wiki/doku.php?id&#x3d;documentation</ext-link>
</comment>. </citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moghaddar</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Swan</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Van der Werf</surname>
<given-names>J.&#x20;H. J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Genomic Prediction from Observed and Imputed High-Density Ovine Genotypes</article-title>. <source>Genet. Sel. Evol.</source> <volume>49</volume>, <fpage>40</fpage>. <pub-id pub-id-type="doi">10.1186/s12711-017-0315-4</pub-id> </citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moreira</surname>
<given-names>F. F.</given-names>
</name>
<name>
<surname>Oliveira</surname>
<given-names>H. R.</given-names>
</name>
<name>
<surname>Volenec</surname>
<given-names>J.&#x20;J.</given-names>
</name>
<name>
<surname>Rainey</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Brito</surname>
<given-names>L. F.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Integrating High-Throughput Phenotyping and Statistical Genomic Methods to Genetically Improve Longitudinal Traits in Crops</article-title>. <source>Front. Plant Sci.</source> <volume>11</volume>, <fpage>681</fpage>. <pub-id pub-id-type="doi">10.3389/fpls.2020.00681</pub-id> </citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Morris</surname>
<given-names>T. P.</given-names>
</name>
<name>
<surname>White</surname>
<given-names>I. R.</given-names>
</name>
<name>
<surname>Crowther</surname>
<given-names>M. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Using Simulation Studies to Evaluate Statistical Methods</article-title>. <source>Stat. Med.</source> <volume>38</volume>, <fpage>2074</fpage>&#x2013;<lpage>2102</lpage>. <pub-id pub-id-type="doi">10.1002/sim.8086</pub-id> </citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mucha</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wierzbicki</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kami&#x144;ski</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ole&#x144;ski</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hering</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>High-frequency Marker Haplotypes in the Genomic Selection of Dairy Cattle</article-title>. <source>J.&#x20;Appl. Genet.</source> <volume>60</volume>, <fpage>179</fpage>&#x2013;<lpage>186</lpage>. <pub-id pub-id-type="doi">10.1007/s13353-019-00489-9</pub-id> </citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ni</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cavero</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Fangmann</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Erbe</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Simianer</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Whole-genome Sequence-Based Genomic Prediction in Laying Chickens with Different Genomic Relationship Matrices to Account for Genetic Architecture</article-title>. <source>Genet. Sel. Evol.</source> <volume>49</volume>, <fpage>8</fpage>. <pub-id pub-id-type="doi">10.1186/s12711-016-0277-y</pub-id> </citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nicolazzi</surname>
<given-names>E. L.</given-names>
</name>
<name>
<surname>Caprera</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nazzicari</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Cozzi</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Strozzi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Lawley</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>SNPchiMp v.3: Integrating and Standardizing Single Nucleotide Polymorphism Data for Livestock Species</article-title>. <source>BMC Genomics</source> <volume>16</volume>, <fpage>283</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-015-1497-1</pub-id> </citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oliveira</surname>
<given-names>H. R.</given-names>
</name>
<name>
<surname>McEwan</surname>
<given-names>J.&#x20;C.</given-names>
</name>
<name>
<surname>Jakobsen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Blichfeldt</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Meuwissen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Pickering</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Genetic Connectedness between Norwegian White Sheep and New&#x20;Zealand Composite Sheep Populations with Similar Development History</article-title>. <source>Front. Genet.</source> <volume>11</volume>, <fpage>371</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2020.00371</pub-id> </citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Effective Population Size of Current Human Population</article-title>. <source>Genet. Res.</source> <volume>93</volume>, <fpage>105</fpage>&#x2013;<lpage>114</lpage>. <pub-id pub-id-type="doi">10.1017/S0016672310000558</pub-id> </citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Piccoli</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Brito</surname>
<given-names>L. F.</given-names>
</name>
<name>
<surname>Braccini</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Oliveira</surname>
<given-names>H. R.</given-names>
</name>
<name>
<surname>Cardoso</surname>
<given-names>F. F.</given-names>
</name>
<name>
<surname>Roso</surname>
<given-names>V. M.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Comparison of Genomic Prediction Methods for Evaluation of Adaptation and Productive Efficiency Traits in Braford and Hereford Cattle</article-title>. <source>Livestock Sci.</source> <volume>231</volume>, <fpage>103864</fpage>. <pub-id pub-id-type="doi">10.1016/j.livsci.2019.103864</pub-id> </citation>
</ref>
<ref id="B50">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Pinheiro</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bates</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>DebRoy</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sarkar</surname>
<given-names>D. R.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Nlme: Linear and Nonlinear Mixed Effects Models. R Package Version 3.1-152</article-title>. <comment>Availableat: <ext-link ext-link-type="uri" xlink:href="https://CRAN.R-project.org/package=nlme">https://CRAN.R-project.org/package&#x3d;nlme</ext-link>
</comment>. </citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prieur</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Clarke</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Brito</surname>
<given-names>L. F.</given-names>
</name>
<name>
<surname>McEwan</surname>
<given-names>J.&#x20;C.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Brauning</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Estimation of Linkage Disequilibrium and Effective Population Size in New&#x20;Zealand Sheep Using Three Different Methods to Create Genetic Maps</article-title>. <source>BMC Genet.</source> <volume>18</volume>, <fpage>68</fpage>. <pub-id pub-id-type="doi">10.1186/s12863-017-0534-2</pub-id> </citation>
</ref>
<ref id="B52">
<citation citation-type="book">
<collab>R Core Team</collab> (<year>2020</year>). <source>R: A Language and Environment for Statistical Computing</source>. <publisher-loc>Vienna, Austria</publisher-loc>: <publisher-name>R Foundation forStatistical Computing</publisher-name>. <comment>Availableat: <ext-link ext-link-type="uri" xlink:href="http://www.R-project.org/">http://www.R-project.org/</ext-link>
</comment>. </citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rao</surname>
<given-names>C. R.</given-names>
</name>
</person-group> (<year>1964</year>). <article-title>The Use and Interpretation of Principal Component Analysis in Applied Research</article-title>. <source>Sankhya: Indian J.&#x20;Stat.</source> <volume>9</volume>, <fpage>1</fpage>. <comment>Availableat: <ext-link ext-link-type="uri" xlink:href="https://www.jstor.org/stable/25049339">https://www.jstor.org/stable/25049339</ext-link>
</comment>. </citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rasali</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Shrestha</surname>
<given-names>J.&#x20;N. B.</given-names>
</name>
<name>
<surname>Crow</surname>
<given-names>G. H.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Development of Composite Sheep Breeds in the World: A Review</article-title>. <source>Can. J.&#x20;Anim. Sci.</source> <volume>86</volume>, <fpage>1</fpage>&#x2013;<lpage>24</lpage>. <pub-id pub-id-type="doi">10.4141/a06-ai</pub-id> </citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sargolzaei</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chesnais</surname>
<given-names>J.&#x20;P.</given-names>
</name>
<name>
<surname>Schenkel</surname>
<given-names>F. S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>A New Approach for Efficient Genotype Imputation Using Information from Relatives</article-title>. <source>BMC Genomics</source> <volume>15</volume>, <fpage>478</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-15-478</pub-id> </citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sargolzaei</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schenkel</surname>
<given-names>F. S.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>QMSim: a Large-Scale Genome Simulator for Livestock</article-title>. <source>Bioinformatics</source> <volume>25</volume>, <fpage>680</fpage>&#x2013;<lpage>681</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp045</pub-id> </citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shimodaira</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>An Approximately Unbiased Test of Phylogenetic Tree Selection</article-title>. <source>Syst. Biol.</source> <volume>51</volume>, <fpage>492</fpage>&#x2013;<lpage>508</lpage>. <pub-id pub-id-type="doi">10.1080/10635150290069913</pub-id> </citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stachowicz</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Brito</surname>
<given-names>L. F.</given-names>
</name>
<name>
<surname>Oliveira</surname>
<given-names>H. R.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>S. P.</given-names>
</name>
<name>
<surname>Schenkel</surname>
<given-names>F. S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Assessing Genetic Diversity of Various Canadian Sheep Breeds through Pedigree Analyses</article-title>. <source>Can. J.&#x20;Anim. Sci.</source> <volume>98</volume>, <fpage>741</fpage>&#x2013;<lpage>749</lpage>. <pub-id pub-id-type="doi">10.1139/cjas-2017-0187</pub-id> </citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sved</surname>
<given-names>J.&#x20;A.</given-names>
</name>
</person-group> (<year>1971</year>). <article-title>Linkage Disequilibrium and Homozygosity of Chromosome Segments in Finite Populations</article-title>. <source>Theor. Popul. Biol.</source> <volume>2</volume>, <fpage>125</fpage>&#x2013;<lpage>141</lpage>. <pub-id pub-id-type="doi">10.1016/0040-5809(71)90011-6</pub-id> </citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Teissier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Larroque</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Brito</surname>
<given-names>L. F.</given-names>
</name>
<name>
<surname>Rupp</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Schenkel</surname>
<given-names>F. S.</given-names>
</name>
<name>
<surname>Robert-Grani&#xe9;</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Genomic Predictions Based on Haplotypes Fitted as Pseudo-SNP for Milk Production and Udder Type Traits and SCS in French Dairy Goats</article-title>. <source>J.&#x20;Dairy Sci.</source> <volume>103</volume>, <fpage>11559</fpage>&#x2013;<lpage>11573</lpage>. <pub-id pub-id-type="doi">10.3168/jds.2020-18662</pub-id> </citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>van Binsbergen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Calus</surname>
<given-names>M. P. L.</given-names>
</name>
<name>
<surname>Bink</surname>
<given-names>M. C. A. M.</given-names>
</name>
<name>
<surname>van Eeuwijk</surname>
<given-names>F. A.</given-names>
</name>
<name>
<surname>Schrooten</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Veerkamp</surname>
<given-names>R. F.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Genomic Prediction Using Imputed Whole-Genome Sequence Data in Holstein Friesian Cattle</article-title>. <source>Genet. Sel. Evol.</source> <volume>47</volume>, <fpage>71</fpage>. <pub-id pub-id-type="doi">10.1186/s12711-015-0149-x</pub-id> </citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vanraden</surname>
<given-names>P. M.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Efficient Methods to Compute Genomic Predictions</article-title>. <source>J.&#x20;Dairy Sci.</source> <volume>91</volume>, <fpage>4414</fpage>&#x2013;<lpage>4423</lpage>. <pub-id pub-id-type="doi">10.3168/jds.2007-0980</pub-id> </citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Villumsen</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Janss</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lund</surname>
<given-names>M. S.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>The Importance of Haplotype Length and Heritability Using Genomic Selection in Dairy Cattle</article-title>. <source>J.&#x20;Anim. Breed. Genet.</source> <volume>126</volume>, <fpage>3</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1111/j.1439-0388.2008.00747.x</pub-id> </citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Won</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>J.-E.</given-names>
</name>
<name>
<surname>Son</surname>
<given-names>J.-H.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S.-H.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>B. H.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Genomic Prediction Accuracy Using Haplotypes Defined by Size and Hierarchical Clustering Based on Linkage Disequilibrium</article-title>. <source>Front. Genet.</source> <volume>11</volume>, <fpage>134</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2020.00134</pub-id> </citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Incorporating Genome Annotation into Genomic Prediction for Carcass Traits in Chinese Simmental Beef Cattle</article-title>. <source>Front. Genet.</source> <volume>11</volume>, <fpage>481</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2020.00481</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>