<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2021.756877</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Novel Design of Imputation-Enabled SNP Arrays for Breeding and Research Applications Supporting Multi-Species Hybridization</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Keeble-Gagn&#x00E8;re</surname> <given-names>Gabriel</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1437108/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Pasam</surname> <given-names>Raj</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/606566/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Forrest</surname> <given-names>Kerrie L.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/886988/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wong</surname> <given-names>Debbie</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Robinson</surname> <given-names>Hannah</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Godoy</surname> <given-names>Jayfred</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Rattey</surname> <given-names>Allan</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Moody</surname> <given-names>David</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1487310/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Mullan</surname> <given-names>Daniel</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/476865/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Walmsley</surname> <given-names>Tresslyn</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Daetwyler</surname> <given-names>Hans D.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/338573/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Tibbits</surname> <given-names>Josquin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1363544/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Hayden</surname> <given-names>Matthew J.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Agriculture Victoria, AgriBio, Centre for AgriBioscience</institution>, <addr-line>Bundoora, VIC</addr-line>, <country>Australia</country></aff>
<aff id="aff2"><sup>2</sup><institution>InterGrain</institution>, <addr-line>Bibra Lake, WA</addr-line>, <country>Australia</country></aff>
<aff id="aff3"><sup>3</sup><institution>School of Applied Systems Biology, La Trobe University</institution>, <addr-line>Bundoora, VIC</addr-line>, <country>Australia</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Sanghyeob Lee, Sejong University, South Korea</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Ahmed Sallam, Assiut University, Egypt; Wuyun Yang, Sichuan Academy of Agricultural Sciences, China</p></fn>
<corresp id="c001">&#x002A;Correspondence: Matthew J. Hayden, <email>matthew.hayden@agriculture.vic.gov.au</email></corresp>
<fn fn-type="other" id="fn004"><p>This article was submitted to Plant Breeding, a section of the journal Frontiers in Plant Science</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>22</day>
<month>12</month>
<year>2021</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>12</volume>
<elocation-id>756877</elocation-id>
<history>
<date date-type="received">
<day>11</day>
<month>08</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>27</day>
<month>10</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2021 Keeble-Gagn&#x00E8;re, Pasam, Forrest, Wong, Robinson, Godoy, Rattey, Moody, Mullan, Walmsley, Daetwyler, Tibbits and Hayden.</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Keeble-Gagn&#x00E8;re, Pasam, Forrest, Wong, Robinson, Godoy, Rattey, Moody, Mullan, Walmsley, Daetwyler, Tibbits and Hayden</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Array-based single nucleotide polymorphism (SNP) genotyping platforms have low genotype error and missing data rates compared to genotyping-by-sequencing technologies. However, design decisions used to create array-based SNP genotyping assays for both research and breeding applications are critical to their success. We describe a novel approach applicable to any animal or plant species for the design of cost-effective imputation-enabled SNP genotyping arrays with broad utility and demonstrate its application through the development of the Illumina Infinium Wheat Barley 40K SNP array Version 1.0. We show that the approach delivers high quality and high resolution data for wheat and barley, including when samples are jointly hybridised. The new array aims to maximally capture haplotypic diversity in globally diverse wheat and barley germplasm while minimizing ascertainment bias. Comprising mostly biallelic markers that were designed to be species-specific and single-copy, the array permits highly accurate imputation in diverse germplasm to improve the statistical power of genome-wide association studies (GWAS) and genomic selection. The SNP content captures tetraploid wheat (A- and B-genome) and <italic>Aegilops tauschii</italic> Coss. (D-genome) diversity and delineates synthetic and tetraploid wheat from other wheat, as well as tetraploid species and subgroups. The content includes SNP tagging key trait loci in wheat and barley, as well as direct connections to other genotyping platforms and legacy datasets. The utility of the array is enhanced through the web-based tool, <italic>Pretzel</italic> (<ext-link ext-link-type="uri" xlink:href="https://plantinformatics.io/">https://plantinformatics.io/</ext-link>) which enables the content of the array to be visualized and interrogated interactively in the context of numerous genetic and genomic resources to be connected more seamlessly to research and breeding. The array is available for use by the international wheat and barley community.</p>
</abstract>
<kwd-group>
<kwd>SNP genotyping</kwd>
<kwd>imputation</kwd>
<kwd>GWAS</kwd>
<kwd>genomic selection</kwd>
<kwd>molecular breeding</kwd>
<kwd>dual sample hybridization</kwd>
<kwd>wheat</kwd>
<kwd>barley</kwd>
</kwd-group>
<counts>
<fig-count count="7"/>
<table-count count="2"/>
<equation-count count="7"/>
<ref-count count="55"/>
<page-count count="16"/>
<word-count count="11810"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="intro">
<title>Introduction</title>
<p>High-density genotyping arrays that simultaneously interrogate thousands of single nucleotide polymorphisms (SNPs) have proven to be a powerful tool in genetic studies. The first generation of these have been widely used in wheat (<italic>Triticum aestivum</italic> L.) and barley (<italic>Hordeum vulgare</italic> L.) for various applications including genome-wide association studies (GWAS), characterization of genetic resources, marker-assisted breeding, and genomic selection (<xref ref-type="bibr" rid="B20">Joukhadar et al., 2017</xref>; <xref ref-type="bibr" rid="B36">Pasam et al., 2017</xref>; <xref ref-type="bibr" rid="B3">Balfourier et al., 2019</xref>). Continued advances in genome assembly and genotyping technologies present powerful new opportunities to continue the integration of genomics information into operational plant breeding systems and extend the potential of more academic research applications; e.g., studying genomic patterns of diversity, inferring ancestral relationships between individuals in populations and studying marker-trait associations in mapping experiments.</p>
<p>Chromosome-scale genome assemblies are becoming available for more and more species and this availability is expected to accelerate with international projects such as the Earth BioGenome Project<sup><xref ref-type="fn" rid="footnote1">1</xref></sup> which aims to sequence, catalog, and characterize the genomes of all of the eukaryotic biodiversity of the earth over the next 10 years. High quality assemblies are already available in cereal crop species, such as barley (<xref ref-type="bibr" rid="B27">Mascher et al., 2017</xref>; <xref ref-type="bibr" rid="B30">Monat et al., 2019</xref>), emmer wheat (<xref ref-type="bibr" rid="B2">Avni et al., 2017</xref>), durum wheat (<xref ref-type="bibr" rid="B26">Maccaferri et al., 2019</xref>), and bread wheat (<xref ref-type="bibr" rid="B46">The International Wheat Genome Sequencing Consortium [IWGSC], 2018</xref>), as well as for the diploid ancestors of wheat (<xref ref-type="bibr" rid="B25">Luo et al., 2017</xref>; <xref ref-type="bibr" rid="B23">Ling et al., 2018</xref>). These assemblies have accelerated SNP discovery and our understanding of the breeding history of wheat and patterns of genome-wide linkage disequilibrium (LD) in different germplasm pools. For example, <xref ref-type="bibr" rid="B15">He et al. (2019)</xref> used an exome capture array in 890 globally diverse hexaploid and tetraploid wheat accessions to discover 7.3M varietal SNPs and investigate the role of wild relative introgressions in shaping wheat improvement and environmental adaptation. <xref ref-type="bibr" rid="B37">Pont et al. (2019)</xref> exome-sequenced a worldwide panel of 487 accessions selected from across the geographical range of complex wheat species to explore how 10,000 years of hybridization, selection, adaptation, and plant breeding have shaped the genetic makeup of modern bread wheat. Similarly, <xref ref-type="bibr" rid="B28">Mascher et al. (2019)</xref> discovered almost 15M varietal SNPs from exome sequence generated for 96 two-row spring and winter barley accessions, a subset of which was used to investigate the extent and partitioning of molecular variation within and between the two groups.</p>
<p>While SNP discovery using whole genome sequence data is currently limited to a relatively small number of wheat and barley accessions, this situation is expected to rapidly change as sequencing costs continue to decrease. For example, 4M group 7 chromosome SNPs from 16 bread wheat accessions (<xref ref-type="bibr" rid="B22">Lai et al., 2015</xref>) and 36M whole genome SNPs from 18 bread wheat accessions (<xref ref-type="bibr" rid="B32">Montenegro et al., 2017</xref>) have previously been reported. The more recent publication of the whole genome sequence assemblies for 15 hexaploid wheat varieties from global breeding programs (<xref ref-type="bibr" rid="B47">Walkowiak et al., 2020</xref>) provides additional new resources for <italic>de novo</italic> whole genome SNP discovery and investigating structural variation within the wheat genome. In barley, <xref ref-type="bibr" rid="B17">Hill et al. (2020)</xref> used a combination of data sources including low coverage whole genome sequence of 632 genotypes representing major global barley breeding programs to investigate genomic selection signatures of breeding in modern varieties.</p>
<p>Increasing genomic resources and increased understanding of global and local population structure (<xref ref-type="bibr" rid="B20">Joukhadar et al., 2017</xref>) enable a shift from higher- to lower-density genotyping assays as a basis for undertaking genetic analyses for trait dissection and mapping. Where high-density data is still required, imputation can be effective to accurately infer higher marker density. Imputation uses statistical approaches to fill missing genotype data and increase low-density genotype data to genome-wide high-density data (<xref ref-type="bibr" rid="B31">Money et al., 2015</xref>). Imputation has been shown to increase the power of the detection of marker-trait associations in GWAS (<xref ref-type="bibr" rid="B19">Jordan et al., 2015</xref>; <xref ref-type="bibr" rid="B14">Fikere et al., 2020</xref>) and genomic selection (<xref ref-type="bibr" rid="B34">Nyine et al., 2019</xref>). Currently, hybridization-based SNP arrays are better suited for imputation, compared to genotyping-by-sequencing (GBS) approaches, due to their lower missing data rates and higher genotype calling accuracies (<xref ref-type="bibr" rid="B39">Rasheed et al., 2017</xref>; <xref ref-type="bibr" rid="B13">Elbasyoni et al., 2018</xref>).</p>
<p>To date, several hybridization-based SNP genotyping arrays providing genome-wide coverage have been developed for wheat and barley. <xref ref-type="bibr" rid="B7">Cavanagh et al. (2013)</xref> developed an Illumina iSelect array that genotyped 9,000 SNPs. The same technology was used a year later to design an array that assayed 90,000 SNPs (<xref ref-type="bibr" rid="B48">Wang et al., 2014</xref>), which was subsequently used to derive a breeder-oriented Infinium 15K array (<xref ref-type="bibr" rid="B43">Soleimani et al., 2020</xref>). <xref ref-type="bibr" rid="B52">Winfield et al. (2016)</xref> reported an Affymetrix Axiom 820K SNP array, which was also subsequently used to derive an Axiom 35K Wheat Breeders&#x2019; array that targeted applications in elite wheat germplasm (<xref ref-type="bibr" rid="B1">Allen et al., 2015</xref>). These genotyping arrays were largely based on genome sequence fragments from early Roche 454 and Illumina assemblies, or from exome capture sequence, and were generally enriched for gene-associated SNPs. More recently, <xref ref-type="bibr" rid="B40">Rimbert et al. (2018)</xref> reported an Axiom 280K SNP array based on content derived from the intergenic fraction of the wheat genome, which to date has been poorly exploited for SNP, while <xref ref-type="bibr" rid="B44">Sun et al. (2020)</xref> described an Axiom 660K array based on genome-specific markers from hexaploid and tetraploid wheat, emmer wheat, and <italic>Aegilops tauschii</italic>. In barley, two Infinium iSelect genotyping arrays comprising 9K and 50K SNPs have been reported (<xref ref-type="bibr" rid="B10">Comadran et al., 2012</xref>; <xref ref-type="bibr" rid="B4">Bayer et al., 2017</xref>).</p>
<p>While SNP genotyping arrays provide robust allele calling with high call rates and fast sample turnaround (typically about 3 days), they have high setup costs. The latter has presented significant challenges for the development of SNP arrays that can comprehensively serve both research and breeding applications; researchers have traditionally preferred high SNP density (which creates a high genotyping cost per sample but low cost per data point), while breeders typically only want a minimally sufficient marker density. This challenge drove us to develop a general approach to SNP array design that specifically takes into consideration the need for low-cost genotyping across a wide range of research and breeding applications, with the aim to seamlessly connect research to breeding.</p>
<p>Here, we present the design methodology and an example of its implementation in the Illumina Infinium Wheat Barley 40K SNP array Version 1.0, a new and highly optimized genotyping platform containing 25,363 wheat-specific and 14,261 barley-specific SNP, the vast majority of which behave as easily scored, single-copy biallelic markers. The SNP content was carefully selected to enable accurate imputation to high SNP density in globally diverse wheat and barley germplasm, as well as within the more restricted germplasm pools of breeding programs. The array is well connected to markers on other commonly used SNP arrays and to many existing genomic resources and provides high utility in research and breeding from germplasm resource characterization, GWAS, and genetic mapping to tracking introgressions from different sources, marker-assisted breeding, and genomic selection. In addition, the SNPs have been selected to enable joint hybridization of wheat and barley samples in the same assay, potentially halving costs for large-scale deployment. The array is available for use by the international wheat and barley community and is supported by the web tool, <italic>Pretzel</italic> (<xref ref-type="bibr" rid="B21">Keeble-Gagn&#x00E8;re et al., 2019</xref>)<sup><xref ref-type="fn" rid="footnote2">2</xref></sup>.</p>
</sec>
<sec id="S2" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="S2.SS1">
<title>Germplasm and Genomic Resources</title>
<p>Single nucleotide polymorphism genotypes for 1,041 exome-sequenced bread wheat accessions were used to select content for the Infinium Wheat Barley 40K SNP array. The accessions included 790 previously reported by <xref ref-type="bibr" rid="B15">He et al. (2019)</xref> to capture global wheat (<italic>T. aestivum</italic>) diversity, an additional 149 accessions selected from the global collection contained in the associated VCF file<sup><xref ref-type="fn" rid="footnote3">3</xref></sup> to expand the diversity captured and 102 historical breeding lines from the InterGrain commercial wheat breeding program<sup><xref ref-type="fn" rid="footnote4">4</xref></sup>. The first two sets of accessions maximally captured genetic diversity among 6,087 globally diverse wheat accessions comprising landraces, varieties, synthetic derivatives, and novel trait donor lines (<xref ref-type="bibr" rid="B15">He et al., 2019</xref>). The additional 149 accessions were selected to capture genetic diversity within synthetic derivative germplasm derived from crossing 100 primary synthetics (derived from interspecific hybridization of durum wheat with <italic>Ae. tauschii</italic>) to three Australian varieties: Yitpi, Annuello, and Correll (<xref ref-type="bibr" rid="B35">Ogbonnaya et al., 2007</xref>). The latter two sets of accessions were exome-capture sequenced as described in <xref ref-type="bibr" rid="B15">He et al. (2019)</xref>. SNP discovery was performed using the first two sets of accessions and the resulting SNP list was used to call SNP genotypes across all accessions.</p>
<p>The Infinium 90K wheat SNP genotypes reported by <xref ref-type="bibr" rid="B26">Maccaferri et al. (2019)</xref> for a globally diverse tetraploid wheat collection of 1,856 accessions comprising wild emmer (<italic>T. turgidum</italic> ssp. <italic>dicoccoides</italic>), domesticated emmer (<italic>T. turgidum</italic> ssp. <italic>dicoccocum</italic>), and <italic>T. turgidum</italic> genotypes including durum landraces and cultivars were used to select tetraploid wheat specific SNP.</p>
<p>A georeferenced landrace collection of 267 exome-sequenced barley accessions, including 2- and 6-rowed <italic>H. vulgare</italic> landraces as well as <italic>H. spontaneum</italic> (<xref ref-type="bibr" rid="B41">Russell et al., 2016</xref>), and 117 whole genome sequenced accessions representing historical breeding lines from the InterGrain commercial barley breeding program were used to select the content for the SNP array.</p>
</sec>
<sec id="S2.SS2">
<title>SNP Discovery</title>
<p>In wheat, SNP discovery and genotype calling were performed as described by <xref ref-type="bibr" rid="B15">He et al. (2019)</xref>, against IWGSC RefSeq v1.0 (<xref ref-type="bibr" rid="B46">The International Wheat Genome Sequencing Consortium [IWGSC], 2018</xref>). After filtering for &#x003E; 60% call rate and &#x003E; 1% MAF, 2.04M SNPs were used for LD analysis. To filter for nucleotide variation originating from <italic>Ae. tauschii</italic>, D-genome-specific SNP that had a MAF &#x003E; 0.1 in the synthetic derivative wheat and MAF &#x003C; 0.1 in the globally diverse wheat collection were identified. In addition, the top 2% of D-genome SNPs that showed differential allele frequencies between these two groups based on F<sub>st</sub> values (<xref ref-type="bibr" rid="B50">Weir and Cockerham, 1984</xref>) were selected. From these two SNP sets, SNP uniformly distributed across the D-genome were selected for inclusion as SNP content.</p>
<p>In barley, SNP discovery was performed as described by <xref ref-type="bibr" rid="B15">He et al. (2019)</xref> using the exome sequence data published by <xref ref-type="bibr" rid="B41">Russell et al. (2016)</xref>, against Morex v1.0 (<xref ref-type="bibr" rid="B27">Mascher et al., 2017</xref>). Following the removal of <italic>H. spontaneum-</italic>like accessions based on principal component analysis (PCA) clustering (which left 157 <italic>H. vulgare</italic>-like accessions), the resulting SNP list was used to call SNP genotypes in the 120 InterGrain historical breeding lines. After filtering for &#x003E; 60% call rate and &#x003E; 5% MAF (a higher cut-off was used in barley due to the smaller reference population), 932,098 SNPs were used for LD analysis.</p>
</sec>
<sec id="S2.SS3">
<title>Linkage Disequilibrium Analysis</title>
<p>Linkage disequilibrium analysis for the filtered SNP was performed using PLINK (<xref ref-type="bibr" rid="B38">Purcell et al., 2007</xref>) at the chromosome level within each species with a maximum window size of 2 Mb; i.e., all the SNPs in a tag SNP set (see &#x201C;Results&#x201D; for definition) had to be within a 2 Mb window. The squared correlation coefficient (<italic>r</italic><sup>2</sup>) based on the allele frequency in the global barley or wheat diversity panel (excluding the synthetic derivatives) between two SNPs was considered as a measure of LD.</p>
</sec>
<sec id="S2.SS4">
<title>Choice of SNP Probe Designs</title>
<p>To maximize the number of SNPs assayed for a given number of probes on the bead chip array, A/T and C/G variants (Infinium Type I SNP which require two probes) were avoided. To maximize SNP scorability and genotype calling accuracy, polymorphism underlying the 50-mer oligonucleotide SNP probe sequences were also avoided as they are known to cause shifts in SNP cluster position (<xref ref-type="bibr" rid="B48">Wang et al., 2014</xref>). For tagging SNPs (tSNPs), the probe sequences were required to align uniquely to the target genome and not aligned to the other genome; i.e., a wheat SNP probe had to align uniquely to the wheat genome and not to the barley genome, and vice versa. Finally, an Illumina Design Tool score of &#x2265; 0.6 was required for a probe to be included as array content. A relaxed set of criteria was also used (to tag SNP sets otherwise missed) which allowed up to three alignments to the target genome.</p>
</sec>
<sec id="S2.SS5">
<title>Selection of Tagging SNP for Imputation</title>
<p>A custom algorithm was used to select tSNP tagging LD blocks in each of the global collections and to facilitate imputation from the density of the SNP array. In brief, for each chromosome, the algorithm iteratively selected the most informative tSNPs passing all filters (based on its <italic>r</italic><sup>2</sup> value from the LD analysis), removed all SNPs linked to the selected tSNPs from the remaining list of SNPs, as well as all SNPs linked to any SNP in the selected tSNP set to avoid directly tagging any SNP at <italic>r</italic><sup>2</sup> &#x2265; 0.9 more than once, before repeating the process until a target number of tSNP was reached. This process ensured that the set of tSNP selected was the minimum set required to tag the most SNPs at <italic>r</italic><sup>2</sup> &#x2265; 0.90. Specifically, for a given set of SNPs <italic>S</italic> = {<italic>s<sub>1</sub></italic>,<italic>s</italic><sub>2</sub>,&#x2026;} and function <italic>r</italic><sup>2</sup>(<italic>s</italic><sub><italic>i</italic></sub>,<italic>s</italic><sub><italic>j</italic></sub>) defining the <italic>Pearson correlation coefficient squared</italic> &#x2200;<italic>s</italic><sub><italic>i</italic></sub>,<italic>s</italic><sub><italic>j</italic></sub> &#x2208; <italic>S</italic>, we define the tSNP set for <italic>s_i</italic> at <italic>q</italic> to be:</p>
<disp-formula id="S2.Ex1">
<mml:math id="M1">
<mml:mrow>
<mml:mrow>
<mml:mpadded width="+3.3pt">
<mml:msubsup>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi>q</mml:mi>
</mml:msubsup>
</mml:mpadded>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mpadded width="+5pt">
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mpadded>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">|</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo rspace="5.8pt">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo rspace="5.8pt">&#x2265;</mml:mo>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Rename the <italic>T^q</italic><sub><italic>s_i</italic></sub> and define <inline-formula><mml:math id="INEQ4"><mml:mrow><mml:mrow><mml:msubsup><mml:mi>T</mml:mi><mml:mrow><mml:mi>s</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>o</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>r</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>t</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>e</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>d</mml:mi></mml:mrow><mml:mi>q</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>T</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mi>q</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mpadded width="+3.3pt"><mml:mi>j</mml:mi></mml:mpadded><mml:mo rspace="5.8pt">=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mi>T</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mi>q</mml:mi></mml:msubsup></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msubsup><mml:mi>T</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mi>q</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>T</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mn>3</mml:mn></mml:msub><mml:mi>q</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mi mathvariant="normal">&#x2026;</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula> where <inline-formula><mml:math id="INEQ5"><mml:mrow><mml:mpadded width="+3.3pt"><mml:mi>i</mml:mi></mml:mpadded><mml:mo rspace="5.8pt">&#x2265;</mml:mo><mml:mpadded width="+3.3pt"><mml:mi>j</mml:mi></mml:mpadded><mml:mo>=</mml:mo><mml:mo rspace="5.8pt">&gt;</mml:mo><mml:mo>|</mml:mo><mml:msubsup><mml:mi>T</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mi>q</mml:mi></mml:msubsup><mml:mo rspace="5.8pt">|</mml:mo><mml:mo rspace="5.8pt">&#x2264;</mml:mo><mml:mo stretchy="false">|</mml:mo><mml:msubsup><mml:mi>T</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mi>q</mml:mi></mml:msubsup><mml:mo stretchy="false">|</mml:mo><mml:mo>.</mml:mo></mml:mrow></mml:math></inline-formula></p>
<p>In other words, <italic>T^q</italic><sub><italic>sorted</italic></sub> is an ordering of equivalent SNP sets, monotonically decreasing in size.</p>
<p>Let <italic>F</italic>&#x2282;<italic>S</italic> be a subset of filtered SNPs. Define <inline-formula><mml:math id="INEQ7"><mml:mrow><mml:mrow><mml:mi>F</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mi>T</mml:mi><mml:mrow><mml:mi>s</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>o</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>r</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>t</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>e</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>d</mml:mi></mml:mrow><mml:mi>q</mml:mi></mml:msubsup><mml:mo rspace="5.8pt">)</mml:mo></mml:mrow></mml:mrow><mml:mo rspace="5.8pt">=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:msubsup><mml:mpadded lspace="5pt" width="+5pt"><mml:mi>T</mml:mi></mml:mpadded><mml:msub><mml:mi>s</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mi>q</mml:mi></mml:msubsup><mml:mo maxsize="120%" minsize="120%">|</mml:mo><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo rspace="5.8pt">&#x2208;</mml:mo><mml:mi>F</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>.</p>
<p>We define <inline-formula><mml:math id="INEQ8"><mml:mrow><mml:mrow><mml:msubsup><mml:mi>T</mml:mi><mml:mrow><mml:mi>s</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>o</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>r</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>t</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>e</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>d</mml:mi></mml:mrow><mml:mi>q</mml:mi></mml:msubsup><mml:mo>-</mml:mo><mml:mpadded width="+3.3pt"><mml:msubsup><mml:mi>T</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mi>q</mml:mi></mml:msubsup></mml:mpadded></mml:mrow><mml:mo rspace="5.8pt">=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:msub><mml:mrow><mml:mo lspace="7.5pt" stretchy="false">(</mml:mo><mml:msubsup><mml:mi>T</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mi>q</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo maxsize="120%" minsize="120%">|</mml:mo><mml:mrow><mml:mrow><mml:mpadded width="+3.3pt"><mml:msubsup><mml:mi>T</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mi>q</mml:mi></mml:msubsup></mml:mpadded><mml:mo rspace="5.8pt">&#x2229;</mml:mo><mml:mpadded width="+3.3pt"><mml:msubsup><mml:mi>T</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mi>q</mml:mi></mml:msubsup></mml:mpadded></mml:mrow><mml:mo rspace="5.8pt">=</mml:mo><mml:mi mathvariant="normal">&#x2205;</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, <italic>h</italic><italic>e</italic><italic>a</italic><italic>d</italic>(<italic>L</italic>) to be the first element of the ordered sequence <italic>L</italic>, and select <inline-formula><mml:math id="INEQ10"><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>T</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mi>q</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> = <italic>s_i</italic>.</p>
<p>The algorithm is then:</p>
<disp-formula id="S2.Ex2">
<mml:math id="M2">
<mml:mrow>
<mml:mpadded width="+5pt">
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mpadded>
<mml:mo rspace="7.5pt">&#x2190;</mml:mo>
<mml:mi mathvariant="normal">&#x2205;</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="S2.Ex3">
<mml:math id="M3">
<mml:mrow>
<mml:mpadded width="+5pt">
<mml:mi>T</mml:mi>
</mml:mpadded>
<mml:mo>&#x2190;</mml:mo>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mi>q</mml:mi>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="S2.Ex4">
<mml:math id="M4">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mtext>while</mml:mtext>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo rspace="5.8pt">|</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo rspace="5.8pt">&#x2265;</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:mi/>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="S2.Ex5">
<mml:math id="M5">
<mml:mrow>
<mml:mrow>
<mml:mi> </mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+5pt">
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mpadded>
</mml:mrow>
<mml:mo>&#x2190;</mml:mo>
<mml:mrow>
<mml:mpadded width="+5pt">
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mpadded>
<mml:mo rspace="7.5pt">&#x222A;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="S2.Ex6">
<mml:math id="M6">
<mml:mrow>
<mml:mrow>
<mml:mi> </mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:msubsup>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mi>q</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo>&#x2190;</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mi>q</mml:mi>
</mml:msubsup>
<mml:mo>-</mml:mo>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="S2.Ex7">
<mml:math id="M7">
<mml:mrow>
<mml:mrow>
<mml:mi> </mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+5pt">
<mml:mi>T</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo>&#x2190;</mml:mo>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mi>q</mml:mi>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>For example, the above applied with <italic>q</italic> = 0.9,<italic>m</italic> = 10 defines the first iteration of tSNP selection.</p>
<p>To guard against possible loss of imputation accuracy due to SNP assays failing to provide reliable genotype calls, a level of redundancy was included in the tSNP sets for wheat and barley. Specifically, three tSNPs were chosen when the number of SNPs tagged was &#x2265; 50 and two tSNPs were selected when the number of SNPs tagged was &#x2265; 20. Single tSNP were selected when they tagged at least 10 SNPs. Some SNP sets could not be tagged because no probe passed all the filters; in this case, we ran the algorithm on the remaining sets allowing for SNP passing relaxed filters (up to three hits to the target genome were allowed). In addition, tSNPs were selected to tag genomic regions that had sparse SNP coverage but high LD; i.e., tagging &#x003C; 10 SNP within windows larger than 500 Kb in wheat and 1Mb in barley. Finally, SNPs were selected in regions still lacking SNPs after the previous steps.</p>
</sec>
<sec id="S2.SS6">
<title>Optimization of SNP Content</title>
<p>To ensure broader applicability of the SNP array in research and breeding, the content included SNP selected to specifically interlink germplasm resources, such as the 19,778 domesticated barley accessions with GBS genotypes described by <xref ref-type="bibr" rid="B29">Milner et al. (2019)</xref>. It also included SNP probes designed to interrogate published trait-linked markers in wheat and barley. Designs for these markers were based directly on published sequences or from the alignment of published primers or flanking sequences and inference of the targeted nucleotide variation. For all trait-linked markers, the best probe design was selected based solely on the Illumina quality score. Due to the difficulty of designing SNP probes targeting known alleles of phenology genes, we selected 293 exome SNPs around the genes reported by <xref ref-type="bibr" rid="B42">Shi et al. (2019)</xref>.</p>
</sec>
<sec id="S2.SS7">
<title>Imputation</title>
<p>The wheat and barley global diversity sets were used as reference haplotypes for imputation. For wheat, accessions clustering with the synthetic derivatives in a PCA analysis were excluded. For barley, only samples with &#x003C; 20% missing data were used. In both species, missing data were filled in using Beagle 4.1 (<xref ref-type="bibr" rid="B5">Browning and Browning, 2007</xref>) and phased with Eagle 2.4.1 (<xref ref-type="bibr" rid="B24">Loh et al., 2016</xref>). In total, 868 and 155 wheat and barley lines were used as reference haplotypes.</p>
<p>In wheat, SNP coordinates were converted to IWGSC v2.0 pseudomolecules<sup><xref ref-type="fn" rid="footnote5">5</xref></sup> (<xref ref-type="bibr" rid="B55">Zhu et al., 2021</xref>) before imputation. After transfer into the v2.0 assembly, there were 18,521 tSNPs before imputation, with 630,058, 549,003, and 352,947 tagged at <italic>r</italic><sup>2</sup> &#x2265; 0.50, 0.70, and 0.90, respectively.</p>
<p>To assess the accuracy of imputation into globally diverse germplasm, 100-fold cross validation was performed. A random subset of 100 wheat (or 10 barley) lines had their true genotypes masked, leaving only the tSNP. The remaining lines were then used as the reference population with Minimac3 software (<xref ref-type="bibr" rid="B11">Das et al., 2016</xref>) to impute back the missing genotypes for three different target SNP sets: the set of SNPs tagged at <italic>r</italic><sup>2</sup> &#x2265; 0.50, 0.70, and 0.90. The imputation accuracy for each line, measured as both correlation squared and concordance between the actual and imputed genotypes, was calculated from 100 repetitions of this process in each of wheat and barley. The correlation squared metric used was the Pearson correlation coefficient squared (<italic>r</italic><sup>2</sup>) between SNPs called in both genotypes being compared, while concordance was measured as the fraction of SNPs in agreement with those called in both genotypes being compared.</p>
</sec>
<sec id="S2.SS8">
<title>Genome-Wide Association Studies</title>
<p>Genome-wide association studies were performed using the GCTA software (<xref ref-type="bibr" rid="B53">Yang et al., 2011</xref>) using a mixed linear model with the SNP matrix fitted as a fixed effect and genomic relationship matrix (GRM) as a random effect. The GRM is a covariance matrix from the SNP information for each sample. Phenotype data for awned status (scored as a presence-absence trait) in 355 wheat accessions and row type (scored as two- or six-rowed) in 121 barley accessions were used. The number of SNPs used in the wheat GWAS, after transfer into the IWGSC v2.0 assembly (see imputation section above), were 18,515 (selected tSNP), 548,864 (imputed tSNP) and 1,086,408 (exome). The number of SNPs used in the barley GWAS were 13,518 (selected tSNP), 359,752 (imputed tSNP), and 1,719,837 (exome). An arbitrary threshold <italic>P</italic>-value of 1&#x00D7;10<sup>&#x2013;5</sup> was used as the significant threshold for declaring a marker-trait association.</p>
</sec>
<sec id="S2.SS9">
<title>SNP Assay and Genotype Calling</title>
<p>Samples were assayed following the protocol for Infinium XT bead chip technology (Illumina Ltd., CA, United States). SNP clustering and allele calling was performed using GenomeStudio Polyploid software (Illumina Ltd., CA, United States) using the Illumina-supplied wheat or barley SNP manifest file. The custom genotype calling pipeline described by <xref ref-type="bibr" rid="B26">Maccaferri et al. (2019)</xref> was also used.</p>
</sec>
<sec id="S2.SS10">
<title>Principal Component Analysis and Plots</title>
<p>Figures and plots were produced in R 3.6.1<sup><xref ref-type="fn" rid="footnote6">6</xref></sup> and ggplot2 (<xref ref-type="bibr" rid="B51">Wickham, 2016</xref>). For PCA plots, SNPRelate 1.20.1 (<xref ref-type="bibr" rid="B54">Zheng et al., 2012</xref>) was used.</p>
</sec>
</sec>
<sec id="S3" sec-type="results">
<title>Results</title>
<sec id="S3.SS1">
<title>Overview of the Design Approach</title>
<p>The central idea of the design concept is to exploit LD using the <italic>r</italic><sup>2</sup> measure to define sets of SNPs that can be considered equivalent; for a given SNP (referred to as a tSNP), we define its tag SNP (or tSNP) set as the set of SNPs tagged by this SNP at <italic>r</italic><sup>2</sup> &#x2265; 0.9. This metric provides a measure of equivalence as well as a natural ranking of SNP by their informativeness, as defined by the size of their tSNP set. We assume that the relationship is symmetrical; i.e., if SNP A is in the tSNP set of SNP B, then SNP B should be in the tSNP set of SNP A. The original set of SNPs are then filtered using technology and application-specific criteria (see section &#x201C;Materials and Methods&#x201D;) while maintaining connectivity to SNPs that fail the filters via the tSNP that pass the filters.</p>
<p>To design a genotyping array that has broader applicability in research and breeding, the SNPs should be discovered in diverse germplasm to avoid ascertainment bias (since LD is population-dependent) and with sufficient density to produce large tSNP sets. The latter helps ensure that at least one SNP in a tSNP set will pass all the design filters in most instances. Here, we used a globally diverse set of barley landrace accessions and a globally diverse set of wheat accessions that included landraces, varieties, novel trait donors, and historical breeding lines (<xref ref-type="fig" rid="F1">Figure 1</xref>). For array designs that are focused only on breeding applications, SNP discovery should aim to capture the genetic diversity within the breeding germplasm pool.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>Principal component analysis (PCA) plots showing genetic diversity of wheat and barley accessions used for SNP discovery. <bold>(A)</bold> About 6,087 wheat accessions were genotyped with the iSelect wheat 90K SNP array (<xref ref-type="bibr" rid="B48">Wang et al., 2014</xref>) (black), exome-sequenced accessions used for linkage disequilibrium (LD) analysis (red), and synthetic derivative accessions capturing D-genome diversity (blue); and <bold>(B)</bold> 19,778 barley accessions genotyped with GBS (<xref ref-type="bibr" rid="B29">Milner et al., 2019</xref>) (black), with exome-sequenced accessions used for LD analysis (red).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-756877-g001.tif"/>
</fig>
<p>A novel selection algorithm (described in &#x201C;Materials and Methods&#x201D;) is then used to select SNPs that maximize LD capture, while minimizing the number of SNPs assayed on the array, using only SNPs that pass the design filters.</p>
<p>The design concept can be applied to any animal or plant species. In addition to this set of SNPs, utility in research and breeding can be further enhanced by including context-relevant SNPs, such as trait-linked markers and markers that link germplasm resources across different genotyping technologies. The approach used to design the Wheat Barley 40K SNP array is summarized in <xref ref-type="supplementary-material" rid="FS5">Supplementary Figure 5</xref>.</p>
</sec>
<sec id="S3.SS2">
<title>SNP Discovery and Filtering</title>
<p>Filtering for a minimum minor allele frequency (MAF) of 1% and maximum missing rate of 40% using the 8,869,370 wheat SNP published in <xref ref-type="bibr" rid="B15">He et al. (2019)</xref> resulted in 2,037,434 high quality SNPs for downstream analysis. Of these, 122,799 SNPs had at least one array probe that passed all design filters. In barley, filtering of the 1,843,823 SNPs identified from our processing of exome capture sequence from the accessions from <xref ref-type="bibr" rid="B41">Russell et al. (2016)</xref> for MAF &#x003E; 5% and missing rate &#x003C; 40% resulted in 932,098 high quality SNPs for downstream analysis, of which 119,633 SNPs had at least one array probe passing all the filters. The filtered SNP matrices used in the subsequent analysis are available at <ext-link ext-link-type="uri" xlink:href="https://dataverse.harvard.edu/dataverse/WheatBarley40k_v1">https://dataverse.harvard.edu/dataverse/WheatBarley40k_v1</ext-link>.</p>
</sec>
<sec id="S3.SS3">
<title>Linkage Disequilibrium Analysis and Selection of Tagging SNP for Imputation</title>
<p>Based on LD values of <italic>r</italic><sup>2</sup> &#x2265; 0.9, a total of 1.07M wheat and 413,508 barley high quality SNPs were singletons; i.e., they had no SNP within 1Mb up and downstream with <italic>r</italic><sup>2</sup> &#x2265; 0.9. These SNPs were either genuine singletons or categorized as singletons due to the absence of additional SNPs within the surrounding 2Mb region. As singleton SNPs can only be tagged directly, which is not feasible on a low-density array, these SNPs were not considered further for inclusion on the array.</p>
<p>The custom selection algorithm grouped the 122,799 non-singleton wheat SNPs passing all design filters into 11,076 tSNP tagging SNP sets containing &#x2265; 10 SNP within a 2 Mb window. These tSNPs tagged 317,599, 538,326, and 652,476 SNPs at <italic>r</italic><sup>2</sup> &#x2265; 0.9, 0.7 and 0.5, respectively. Of the 119,633 non-singleton barley SNPs passing all filters, the selection algorithm identified 7,316 tSNPs which tagged a total of 150,096, 294,659, and 390,844 SNPs at <italic>r</italic><sup>2</sup> &#x2265; 0.9, 0.7, and 0.5, respectively. At the genome level, the rate of return per tSNP was surprisingly similar for wheat and barley and plateaued at about 15,000 tSNP at <italic>r</italic><sup>2</sup> &#x2265; 0.9 (<xref ref-type="fig" rid="F2">Figure 2</xref>). However, the rate of return per tSNP varied at the chromosome level (<xref ref-type="supplementary-material" rid="FS1">Supplementary Figure 1</xref>).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Cumulative number of SNPs tagged by tSNPs at <italic>r</italic><sup>2</sup> &#x2265; 0.9 (blue), 0.7 (green), and 0.5 (red), in wheat <bold>(A)</bold> and barley <bold>(B)</bold>. The curves are shown until the first singleton SNP (at <italic>r</italic><sup>2</sup> &#x2265; 0.90) is reached.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-756877-g002.tif"/>
</fig>
<p>In total, 21,012 wheat and 13,469 barley tSNPs were included as content on the array. This tally includes redundant SNPs selected to guard against the possible loss of imputation accuracy due to SNP assays that might fail; SNP passing a relaxed set of filters (allowing up to three alignments to the target genome) and tagging SNP sets untaggable with the strictly filtered SNP; and SNP to tag genomic regions that had sparse SNP coverage but high LD; i.e., tagging &#x003C; 10 SNPs within windows larger than 500 Kb in wheat and 1 Mb in barley. The latter SNPs are expected to support increased imputation density in these regions as higher density SNP datasets become available in the future. The wheat tSNPs tagged a total of 394,034, 636,641, and 758,452 SNPs at <italic>r</italic><sup>2</sup> &#x2265; 0.9, 0.70 and 0.50, respectively, while the barley tSNP tagged a total of 187,412, 361,012, and 471,645 SNPs, respectively. Importantly, the MAF distributions for the tSNP, tagged SNP, and filtered SNP from the globally diverse wheat and barley collections closely matched one another (<xref ref-type="fig" rid="F3">Figure 3</xref>). The distribution of the selected tSNP in the wheat and barley genomes is shown in <xref ref-type="supplementary-material" rid="FS4">Supplementary Figure 4</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>Minor allele frequency (MAF) distribution of all SNPs used for LD analysis, selected tSNPs, and the set of SNPs tagged by the tSNPs at <italic>r</italic><sup>2</sup> &#x2265; 0.70 in the globally diverse wheat <bold>(A)</bold> (<italic>n</italic> = 790) and barley <bold>(B)</bold> (<italic>n</italic> = 157) collections.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-756877-g003.tif"/>
</fig>
</sec>
<sec id="S3.SS4">
<title>Accuracy for Imputing Into Globally Diverse Germplasm</title>
<p>Cross validation (100-fold) was used to assess the accuracy for imputing from the tSNPs on the array to the sets of SNPs tagged at <italic>r</italic><sup>2</sup> &#x2265; 0.50, 0.70, and 0.90, in the globally diverse wheat and barley germplasm. This was achieved by randomly selecting 100 wheat (or 10 barley) lines and masking their true genotypes to leave only the tSNPs. Using the remaining lines as the reference population, the masked genotypes for each randomly selected line were imputed to the density of one of the target SNP sets. Accuracy was determined from the correlation squared and concordance between the imputed and actual genotypes for each wheat or barley line averaged over the occurrences of that sample within the 100 iterations.</p>
<p>As expected, all metrics were the highest when imputing to the set of SNPs tagged at <italic>r</italic><sup>2</sup> &#x2265; 0.90 and the lowest for those tagged at <italic>r</italic><sup>2</sup> &#x2265; 0.50 (<xref ref-type="table" rid="T1">Table 1</xref>). In wheat, only a small decrease in accuracy was observed for most accessions as the size of the tagged SNP set increased (i.e., <italic>r</italic><sup>2</sup> decreased), with reduced accuracy most evident in the bottom 50 accessions (<xref ref-type="fig" rid="F4">Figure 4</xref>). For these accessions, the difference in accuracy (both correlation squared and concordance) between comparisons including and excluding heterozygous genotype calls was almost 10%, suggesting the possibility of high error rates in the heterozygous exome SNP calls for these accessions. About 768 (88.5%) of the wheat accessions had accuracies &#x2265; 90% with the strictest correlation squared metric (which included heterozygous calls) for the set of SNPs tagged at <italic>r</italic><sup>2</sup> &#x2265; 0.50. When comparing only homozygous calls, the number of lines above this threshold rose to 866 (99.8%) (<xref ref-type="fig" rid="F4">Figure 4</xref>).</p>
<table-wrap position="float" id="T1">
<label>TABLE 1</label>
<caption><p>Average accuracies for imputing from the tSNPs on the array to the sets of SNPs tagged at <italic>r</italic><sup>2</sup> &#x2265; 0.50, 0.70, and 0.90, in wheat and barley.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left"></td>
<td valign="top" align="center">Set of SNP tagged at <italic>r</italic><xref ref-type="table-fn" rid="t1fn2"><sup>2</sup></xref></td>
<td valign="top" align="center">Wheat</td>
<td valign="top" align="center">Barley</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Correlation squared (including heterozygous calls)</td>
<td valign="top" align="center">0.50</td>
<td valign="top" align="center">93.7 (4.0)</td>
<td valign="top" align="center">86.0 (3.1)</td>
</tr>
<tr>
<td valign="top" align="justify"/><td valign="top" align="center">0.70</td>
<td valign="top" align="center">95.3 (3.8)</td>
<td valign="top" align="center">92.4 (2.6)</td>
</tr>
<tr>
<td valign="top" align="justify"/><td valign="top" align="center">0.90</td>
<td valign="top" align="center">97.0 (3.4)</td>
<td valign="top" align="center">96.8 (1.6)</td>
</tr>
<tr>
<td valign="top" align="left">Correlation squared (excluding heterozygous calls)</td>
<td valign="top" align="center">0.50</td>
<td valign="top" align="center">97.6 (1.3)</td>
<td valign="top" align="center">91.5 (2.9)</td>
</tr>
<tr>
<td valign="top" align="justify"/><td valign="top" align="center">0.70</td>
<td valign="top" align="center">98.7 (1.0)</td>
<td valign="top" align="center">96.9 (2.3)</td>
</tr>
<tr>
<td valign="top" align="justify"/><td valign="top" align="center">0.90</td>
<td valign="top" align="center">99.3 (0.7)</td>
<td valign="top" align="center">98.7 (1.3)</td>
</tr>
<tr>
<td valign="top" align="left">Concordance (including heterozygous calls)</td>
<td valign="top" align="center">0.50</td>
<td valign="top" align="center">96.9 (2.2)</td>
<td valign="top" align="center">92.8 (1.4)</td>
</tr>
<tr>
<td valign="top" align="justify"/><td valign="top" align="center">0.70</td>
<td valign="top" align="center">97.4 (2.1)</td>
<td valign="top" align="center">95.2 (1.2)</td>
</tr>
<tr>
<td valign="top" align="justify"/><td valign="top" align="center">0.90</td>
<td valign="top" align="center">98.3 (2.0)</td>
<td valign="top" align="center">98.1 (0.8)</td>
</tr>
<tr>
<td valign="top" align="left">Concordance (excluding heterozygous calls)</td>
<td valign="top" align="center">0.50</td>
<td valign="top" align="center">99.6 (0.2)</td>
<td valign="top" align="center">98.1 (0.7)</td>
</tr>
<tr>
<td valign="top" align="justify"/><td valign="top" align="center">0.70</td>
<td valign="top" align="center">99.8 (0.2)</td>
<td valign="top" align="center">99.3 (0.5)</td>
</tr>
<tr>
<td valign="top" align="justify"/><td valign="top" align="center">0.90</td>
<td valign="top" align="center">99.9 (0.1)</td>
<td valign="top" align="center">99.7 (0.2)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="t1fn2"><p><italic>Correlation squared is defined as the Pearson correlation coefficient squared (r<sup>2</sup>) between SNPs called in both genotypes being compared. Concordance is the fraction of SNPs in agreement between those called in both genotypes being compared. SDs are shown in brackets.</italic></p></fn>
</table-wrap-foot>
</table-wrap>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p>Imputation accuracy from the tSNPs on the array to the set of SNPs tagged at <italic>r</italic><sup>2</sup> &#x2265; 0.5, 0.7, and 0.9, in wheat <bold>(A)</bold> and barley <bold>(B)</bold>. Metrics plotted are correlation <italic>r</italic><sup>2</sup> including heterozygous calls (purple line), <italic>r</italic><sup>2</sup> excluding heterozygous calls (cyan line), concordance including heterozygous calls (green line), and concordance excluding heterozygous calls (orange line). The accessions are rank ordered based on the <italic>r</italic><sup>2</sup> including heterozygous calls.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-756877-g004.tif"/>
</fig>
<p>Reduced accuracy when imputing to higher tagged SNP numbers was more pronounced in barley. A difference of 10.8% (from 96.8 to 86%) was observed between the average correlation squared (which included heterozygous calls) for the set of SNPs tagged at <italic>r</italic><sup>2</sup> &#x2265; 0.90, compared to those tagged at <italic>r</italic><sup>2</sup> &#x2265; 0.50 (<xref ref-type="table" rid="T1">Table 1</xref>). As observed in wheat, the inclusion of heterozygous calls reduced the accuracy, particularly when imputing to the set of SNPs tagged at <italic>r</italic><sup>2</sup> &#x2265; 0.50, again suggesting possible erroneous heterozygous calls in the sequence genotypes (<xref ref-type="fig" rid="F4">Figure 4</xref>). The reduced accuracies observed in barley compared to wheat are also likely to be partly due to the reduced size of reference haplotypes (155 vs. 868). Accuracies in barley are likely to improve if the reference haplotype set is expanded.</p>
<p>To confirm that the selected tSNPs were useful for detecting marker-trait associations, we performed GWAS using phenotype data for awned status (scored as a presence-absence trait) in 355 wheat accessions and defining row type (scored as two- or six-rowed) in 121 barley accessions and the selected tSNPs before and after imputation. The results were compared with GWAS performed using the same phenotypic data and exome SNP genotypes (<xref ref-type="fig" rid="F5">Figure 5</xref>). Significant and completely overlapping GWAS signals were observed for the three analyses performed in both wheat and barley using the different datasets. The significant SNPs in each analysis were associated with genomic regions previously reported to be associated with the traits (<xref ref-type="bibr" rid="B41">Russell et al., 2016</xref>; <xref ref-type="bibr" rid="B18">Huang et al., 2020</xref>). While the significance of the associated SNPs differed across the three analyses for each trait, the GWAS results show that the selected tSNPs effectively tag common halotype block diversity in globally diverse germplasm.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption><p>Genome-wide association study (GWAS) for awned status and row type in wheat and barley, respectively, using: <bold>(A)</bold> selected tSNP; <bold>(B)</bold> selected tSNP after imputation to the <italic>r</italic><sup>2</sup> &#x2265; 0.7 target set (the sample being imputed was removed from the reference set); and <bold>(C)</bold> exome SNP. Note the -log10(p) axes are scaled to 10 which resulted in the most significant SNP (38.44) for the 5A locus in wheat being out of the range on the axis for the wheat exome SNP plot.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-756877-g005.tif"/>
</fig>
</sec>
<sec id="S3.SS5">
<title>Wheat Barley 40K SNP Array Content</title>
<p>The final array design comprised 34,481 tSNPs and two additional categories of context-specific SNPs (content summarized in <xref ref-type="table" rid="T2">Table 2</xref>; full details are in <xref ref-type="supplementary-material" rid="TS1">Supplementary Table 1</xref>).</p>
<table-wrap position="float" id="T2">
<label>TABLE 2</label>
<caption><p>SNP content of the Infinium Wheat Barley 40K SNP bead chip array.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left"></td>
<td valign="top" align="center">Wheat</td>
<td valign="top" align="center">Barley</td>
<td valign="top" align="center">Total</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Tagging SNP for imputation</td>
<td valign="top" align="center">21,012</td>
<td valign="top" align="center">13,469</td>
<td valign="top" align="center">34,481</td>
</tr>
<tr>
<td valign="top" align="left">Trait associated SNP</td>
<td valign="top" align="center">427</td>
<td valign="top" align="center">178</td>
<td valign="top" align="center">605</td>
</tr>
<tr>
<td valign="top" align="left">SNP linking germplasm resources</td>
<td valign="top" align="center">3,924</td>
<td valign="top" align="center">614</td>
<td valign="top" align="center">4,538</td>
</tr>
<tr>
<td valign="top" align="left"><bold>Total number of SNP</bold></td>
<td valign="top" align="center"><bold>25,363</bold></td>
<td valign="top" align="center"><bold>14,261</bold></td>
<td valign="top" align="center"><bold>39,624</bold></td>
</tr>
</tbody>
</table></table-wrap>
<p>The first context-specific category included 2,609 SNPs from the Infinium wheat 90K SNP array (<xref ref-type="bibr" rid="B48">Wang et al., 2014</xref>) that were selected based on allele differentiation to tag tetraploid wheat (A- and B-genome) diversity and to clearly delineate tetraploid wheat from other types of wheat, as well as to distinguish tetraploid species and subgroups from one another. The SNPs comprised the following four classes: (1) differentiating SNPs that represent the top 2% F<sub>st</sub> values in the study by <xref ref-type="bibr" rid="B26">Maccaferri et al. (2019)</xref> between the four subgroups of tetraploid species: wild emmer (<italic>Triticum turgidum</italic> ssp. <italic>dicoccoides</italic>), domesticated emmer (<italic>T. turgidum</italic> ssp. <italic>dicoccocum</italic>), and durum (<italic>T. turgidum</italic>) landraces, and durum cultivars; (2) subgroup-specific private SNPs that showed a MAF &#x2265; 0.1 in one of the subgroups and were either monomorphic or showed a MAF &#x003C; 0.05 in the other subgroups; (3) subgroup-specific high MAF SNPs that were present at &#x2265; 0.3 MAF in any one of the subgroups; and (4) neutral SNPs that did not show any signatures of selection, were polymorphic in all subgroups and showed an overall MAF of &#x2265; 0.4. The ability of these SNPs to reliably differentiate the tetraploid species subgroups as efficiently as the Infinium wheat 90K array is shown in <xref ref-type="supplementary-material" rid="FS2">Supplementary Figure 2</xref>. The distribution of these SNPs across the A- and B-genomes of wheat is shown in <xref ref-type="supplementary-material" rid="FS4">Supplementary Figure 4</xref>.</p>
<p>The second category included 1,206 exome SNPs tagging <italic>Ae. tauschii</italic> (D-genome) diversity present in backcross synthetic derivatives that originated from crosses involving 100 primary synthetic parents, which were selected for phenotypic and genetic diversity among approximately 400 primary synthetics developed at CIMMYT and imported to Australia in 2001. Each of the 100 primary synthetic parents was derived from a different <italic>Ae. tauschii</italic> accession. SNPs tagging diversity in <italic>Ae. tauschii</italic> were selected to provide high genome coverage in the D-genome (<xref ref-type="supplementary-material" rid="FS4">Supplementary Figure 4</xref>). They were also selected to clearly delineate bread wheat from other types of wheat. The SNP comprised two classes: (1) differentiating SNPs that represent the top 2% F<sub>st</sub> values between the global diversity wheat and synthetic derivative collections; and (2) D-genome diversity from <italic>Ae. tauschii</italic> that showed a MAF &#x2265; 0.1 in the synthetic derivative collection and MAF &#x2264; 0.1 in the globally diverse wheat collection. The ability of these SNPs to reliably differentiate synthetic wheat from common wheat as efficiently as the Infinium wheat 90K array is shown in <xref ref-type="supplementary-material" rid="FS3">Supplementary Figure 3</xref>.</p>
<p>The final category included linked SNPs for key breeding traits and SNPs linking major germplasm resources genotyped with different technologies. In total, 457 wheat and 178 barley SNPs corresponded to published trait-linked markers as well as 109 SNPs associated with agronomically important genes reported in published GWAS studies (<xref ref-type="bibr" rid="B45">Sun et al., 2017</xref>; <xref ref-type="bibr" rid="B49">Wang et al., 2017</xref>) (<xref ref-type="supplementary-material" rid="TS1">Supplementary Table 1</xref>). Another 614 SNPs provide a direct link to 19,778 GBS genotyped domesticated barley accessions (<xref ref-type="bibr" rid="B29">Milner et al., 2019</xref>).</p>
</sec>
<sec id="S3.SS6">
<title>Assay Performance&#x2013;Single Sample Hybridizations</title>
<p>A limitation of hybridization-based genotyping arrays is that their oligonucleotide probes hybridize both to the targeted locus and its homologs and paralogs, if present (<xref ref-type="bibr" rid="B7">Cavanagh et al., 2013</xref>; <xref ref-type="bibr" rid="B48">Wang et al., 2014</xref>). Consequently, the ratio of allele-specific fluorescent signals observed for an assay depends on the copy number of the locus in the genome, with increasing copy number reducing the allele-specific fluorescent signal ratio and separation of SNP allele clusters. Further, SNP assay scorability and genotype calling can be confounded by the presence of mutations that modify oligonucleotide annealing such that different cluster patterns are observed across germplasm (<xref ref-type="bibr" rid="B48">Wang et al., 2014</xref>). An ideal assay design for a hybridization-based genotyping array is therefore an oligonucleotide probe that binds at only one locus in the genome and has no known nucleotide variation underlying the probe hybridization site. Theoretically, this should ensure three distinct clusters corresponding to the genotypic states (REF, HET, and ALT) expected from a single-copy biallelic SNP. The increasing availability of genomic resources now allows for this historical problem to be addressed. Hence, we used the combination of reference genome assemblies and genotypic data for large globally diverse wheat and barley collections to specifically target the design of single copy biallelic SNP assays.</p>
<p>For the purpose of evaluating the performance of the array, the wheat and barley diversity populations were used to define cluster positions for SNP genotype calling. The vast majority (98%) of the 39,654 SNP assays on the array produced scorable cluster patterns when hybridized with a barley or wheat sample; about 91% (12,949/14,261) of the barley and 83% (20,090/24,598) of the wheat SNP assays could be reliably scored as single-copy biallelic markers, with the REF and ALT clusters having theta values close to 0 and 1 in GenomeStudio SNP plots (<xref ref-type="fig" rid="F6">Figure 6</xref>). While the remaining SNP could typically be reliably scored as biallelic markers, they showed cluster compression indicative of multiple loci. Few assays showed complex clustering patterns, indicating the success of designing probes without any underlying polymorphism. About 5 and 7% of wheat and barley assays showed a clustering pattern typical for the presence of a null allele. The occurrence of assays not behaving as single-copy biallelic markers reflects current knowledge gaps for structural variation in the genomes of wheat and barley including both copy number variation and presence-absence variation (<xref ref-type="bibr" rid="B48">Wang et al., 2014</xref>; <xref ref-type="bibr" rid="B3">Balfourier et al., 2019</xref>; <xref ref-type="bibr" rid="B47">Walkowiak et al., 2020</xref>).</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption><p>Cluster positions and theta separation of SNP in single sample hybridization assays. Scatter plot of cluster positions (left) and density plot of the difference in theta value between REF and ALT clusters (right) for <bold>(A)</bold> 14,261 barley and <bold>(B)</bold> 24,598 wheat SNP revealing polymorphism in the globally diverse wheat and barley populations.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-756877-g006.tif"/>
</fig>
<p>The concordance between called and actual genotypes was exceptionally high for both wheat and barley. The average genotype concordance and correlation squared were 99.5 and 98.1%, respectively, in wheat when heterozygous genotype calls were excluded, and 97.6 and 95.7%, respectively, when heterozygous calls were included. Similarly, 99.8% concordance and 99.2% correlation squared were observed in barley when heterozygous calls were excluded, and 98.2 and 97.2% were observed with heterozygous calls included. The average missing data rates were 4.8 and 3.8% in wheat and barley, respectively.</p>
</sec>
<sec id="S3.SS7">
<title>Assay Performance&#x2014;Dual Sample Hybridizations</title>
<p>The design process specifically aimed to select species-specific SNP probes and thus it should be theoretically possible to jointly hybridize a wheat and barley sample to the same bead chip array (dual hybridization) without the loss of genotype calling accuracy. Cross-hybridization between species is expected to confound genotype calling accuracy by creating shifts in SNP cluster positions and/or complex clustering patterns that cannot be easily scored.</p>
<p>To evaluate the assay performance of a dual hybridization, samples from the InterGrain commercial barley and wheat breeding programs were used to define cluster positions and call SNP genotypes for 576 dual hybridization assays. The same samples were also assayed in single sample hybridization assays to enable genotype calling accuracy between dual and single hybridization assays to be directly compared.</p>
<p>Most of the barley and wheat SNPs in dual hybridization assays produced scorable cluster patterns. Shifts in cluster positions were observed, which indicated either that some oligonucleotide probes showed a degree of cross-species hybridization or that deviation from the standard amount of sample DNA (200 ng per sample) recommended for the bead chip assay affected signal-to-noise. Through empirical testing, we found that the quantity of genomic DNA per sample was a major factor causing shifts in cluster position (data not shown) and could be minimized by adjusting the input DNA for each sample to match the ratio of the genome size for each species; e.g., 200 ng barley DNA and 600 ng wheat DNA; the bread wheat genome is about three times larger than that of barley.</p>
<p>For the purpose of assessing genotype calling accuracy for dual hybridization assays, only SNPs that revealed polymorphism among the 576 wheat and barley samples assayed were considered. Of the 9,826 barley and 9,118 wheat SNPs showing polymorphism, the vast majority were easily scored as biallelic markers and had good cluster separation, indicating that oligonucleotide probe cross-species hybridization was minimal (<xref ref-type="fig" rid="F7">Figure 7</xref>). The average concordance between genotypes calls for the same wheat and barley samples in single and dual sample hybridization assays were 99.9, 96.7, and 99.8%, for the REF, HET, and ALT alleles, respectively. The average missing data rate across the wheat and barley samples was similar for both assay types, with 4.7 and 2.0% in dual and single hybridization assays, respectively.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption><p>Cluster positions and theta separation of SNPs in dual hybridization assays. Scatter plot of cluster positions (left) and density plot of the difference in theta value between REF and ALT clusters (right) for <bold>(A)</bold> 9,826 barley and <bold>(B)</bold> 9,118 wheat SNP revealing polymorphism among 576 wheat and barley breeding lines.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-756877-g007.tif"/>
</fig>
</sec>
</sec>
<sec id="S4" sec-type="discussion">
<title>Discussion</title>
<p>High-throughput, low-cost and flexible genotyping platforms are required for both research and breeding applications. Compared to GBS and PCR-based marker systems, array-based genotyping platforms are highly commercialized and highly customizable, both for the number of markers and the samples assayed. They also have low genotype error and missing data rates compared to GBS technologies (<xref ref-type="bibr" rid="B39">Rasheed et al., 2017</xref>). Consequently, SNP arrays are widely utilized and several low-density SNP genotyping arrays have been developed for wheat and barley. Here, we described a novel approach that is applicable to any animal or plant species for the design of cost-effective, imputation-based SNP genotyping arrays with broad utility and that support the hybridization of multiple samples to the same SNP array. The utility of the approach was demonstrated through the development of the Infinium Wheat Barley 40K SNP array.</p>
<p>The key difference between the Infinium Wheat Barley 40K SNP array and previously reported array-based genotyping assays is a paradigm shift in the logic underpinning its design. To date, commonly used low-density genotyping arrays are comprised of the most scorable and informative markers from higher density arrays. For example, the Infinium Wheat 15K SNP array (<xref ref-type="bibr" rid="B43">Soleimani et al., 2020</xref>) and Axiom Wheat Breeders&#x2019; 35K SNP array (<xref ref-type="bibr" rid="B1">Allen et al., 2015</xref>) are derived from the Infinium Wheat 90K SNP array (<xref ref-type="bibr" rid="B48">Wang et al., 2014</xref>) and Axiom Wheat 820K SNP array (<xref ref-type="bibr" rid="B52">Winfield et al., 2016</xref>). SNPs on the Infinium 90K SNP array were derived from the transcriptome sequence of 19 bread wheat accessions and 18 tetraploid accessions, while those on the Axiom 820K arrays were based on exome capture sequence from 43 bread wheat and wild species accessions representing the primary, secondary, and tertiary gene pools. While these derived low-density arrays are affordable for routine deployment in breeding and research, their content is breeder-oriented and has limited utility outside the primary gene pool of hexaploid wheat.</p>
<p>The design implemented in the Infinium Wheat Barley 40K SNP array is based on the hugely expanded genotypic and genomic resources now available for wheat and barley. By using these resources, we were able to identify species-specific single-copy tSNPs that capture a large proportion of the haplotypic diversity in globally diverse germplasm, and are highly scorable for accurate genotype calling, minimize ascertainment bias, and enable accurate imputation to high SNP density. In the case of wheat, this included the use of 2.04M SNPs identified from exome sequence data of 1,041 accessions selected to maximally capture genetic diversity among a global collection of 6,087 accessions genotyped using the Infinium 90K SNP array (<xref ref-type="bibr" rid="B15">He et al., 2019</xref>; <xref ref-type="fig" rid="F1">Figure 1A</xref>). The global collection included landraces, released varieties, synthetic derivatives, and novel trait donor and historical breeding lines. For barley, this included 932,098 SNPs identified from exome sequence data of 267 accessions selected to maximally capture geographic diversity among landraces (<xref ref-type="bibr" rid="B41">Russell et al., 2016</xref>; <xref ref-type="fig" rid="F1">Figure 1B</xref>), as well as SNPs identified from target capture sequencing of 174 flowering time-related genes performed in 895 worldwide accessions (<xref ref-type="bibr" rid="B16">Hill et al., 2019</xref>). The latter dataset included globally diverse cultivated and landrace germplasm.</p>
<p>By selecting tSNPs that enable accurate imputation of common haplotype block diversity in globally diverse germplasm, the Infinium Wheat Barley 40K array is expected to maintain power for GWAS, genetic mapping, and genomic selection (<xref ref-type="bibr" rid="B19">Jordan et al., 2015</xref>; <xref ref-type="bibr" rid="B15">He et al., 2019</xref>; <xref ref-type="bibr" rid="B33">Negro et al., 2019</xref>; <xref ref-type="bibr" rid="B34">Nyine et al., 2019</xref>). Haplotype blocks are essentially fixed stretches of DNA sequence that show little historical evidence of recombination and are effectively inherited as genetic units that are shuffled and assembled during breeding. The univariate LD metric <italic>r</italic><sup>2</sup> has been used in many tSNP algorithms as it is a major determinant of imputation accuracy and has a simple inverse relationship with the sample size required to detect associations in GWAS (<xref ref-type="bibr" rid="B6">Carlson et al., 2004</xref>; <xref ref-type="bibr" rid="B12">Ding and Kullo, 2007</xref>). By selecting tSNPs with an <italic>r</italic><sup>2</sup> &#x2265; 0.9 cut-off, we aimed to retain most of the information content in the original SNP set and to balance the power loss with the effort needed to compensate with increased sample numbers in downstream GWAS (&#x223C;11%; i.e., 1/0.9). This aspect of the array design was confirmed by performing GWAS for awned status in wheat and row type in barley (<xref ref-type="fig" rid="F5">Figure 5</xref>). A significant advantage when using <italic>r</italic><sup>2</sup> is that it allows for a high degree of flexibility in the composition of the final tSNP set, thereby enabling other design criteria to be applied without compromising the overall tagging efficiency. This was especially important for implementing array design principles such as selecting species-specific, single-copy SNP targets that had no nucleotide variation underlying the probes to both maximize SNP scorability and support dual sample hybridization assays. The success of our approach was confirmed by &#x003E; 97% accuracy (as measured by both correlation squared and concordance between the imputed and actual SNP genotypes) when imputing the set of SNPs tagged at <italic>r</italic><sup>2</sup> &#x2265; 0.9 (inclusive of heterozygous calls) in both wheat and barley. Importantly, imputation accuracy was also high for the set of SNPs tagged at <italic>r</italic><sup>2</sup> &#x2265; 0.5 (<xref ref-type="table" rid="T1">Table 1</xref>). To futureproof the array design, we added tSNP tagging genomic regions in wheat and barley that had sparse exome SNP coverage but high LD. We expect this content will similarly support accurate imputation to whole genome sequence once genomic resources needed to achieve this are available.</p>
<p>In emphasizing the design focus on selecting tSNPs for imputation, we also point out the limitations it has for fully capturing the haplotype diversity in global wheat and barley germplasm. First, we did not tag LD blocks comprised of fewer than 10 SNPs since this would have required an order of magnitude more SNP assays on the array; about 30,000 tSNP per species was required to tag about half of the non-singleton exome SNP at <italic>r</italic><sup>2</sup> &#x2265; 0.9 in each of wheat and barley (<xref ref-type="fig" rid="F2">Figure 2</xref>). This presents a limitation for trait mapping using GWAS (but not genetic mapping) since trait loci located in untagged LD blocks will become increasingly harder to detect as their LD with SNPs on the array decreases. This limitation can be partly overcome by increasing the sample size but is an unavoidable consequence of low-density arrays, despite our tSNP selection algorithm ensuring that we maximized the number of SNP tagged in LD. And second, the set of SNPs and LD relationships between them is still limited by the data currently available. As exome capture sequencing assays target only 2&#x2013;3% of the genome, the SNPs discovered represent just a fraction of the true SNP density. It is therefore possible that SNPs were not selected simply because the haplotype they represent was only sampled by a small number of SNP in that region and was below our selection thresholds. This limitation will only be overcome by large-scale whole genome sequencing efforts which are just beginning to become affordable for large genome-sized species. It should be noted that the LD patterns detected in this study will remain valid even with higher density sequencing and that the majority of the tagged LD haplotypes span across the captured regions and so the number of SNPs in high LD with the selected tSNPs will only increase as higher density SNP data becomes available.</p>
<p>An argued advantage for GBS assays is that they are free from ascertainment bias. Ascertainment bias can result in rare alleles being missed and genetic diversity being underestimated in non-ascertained populations (<xref ref-type="bibr" rid="B9">Clark et al., 2005</xref>), with its impact dependent on the study being undertaken. Increasing marker density and low MAF markers in GWAS boosts the power for quantitative trait loci (QTL) detection (<xref ref-type="bibr" rid="B33">Negro et al., 2019</xref>; <xref ref-type="bibr" rid="B14">Fikere et al., 2020</xref>). <xref ref-type="bibr" rid="B8">Chu et al. (2020)</xref> reported that very low frequency markers (MAF &#x003C; 0.05) contributed to an improvement of genomic prediction accuracy in 378 winter bread wheat genotypes, and combined with the expectation that valuable novel diversity is most likely rare (<xref ref-type="bibr" rid="B28">Mascher et al., 2019</xref>), suggests that rare markers deserve careful consideration. Our tSNP selection algorithm prioritizes haplotypes that diverge significantly from the reference genome used for SNP discovery in order to maximize the number of SNP tagged in LD; it is agnostic to the MAF of individual SNP (beyond the MAF cut-offs of 1 and 5% in wheat and barley, respectively). Consequently, the MAF spectrum of the wheat and barley tSNPs closely resembled that observed for both the sets of tagged SNPs and the filtered SNPs in the globally diverse collections (<xref ref-type="fig" rid="F3">Figure 3</xref>). Hence, we suggest that the Infinium Wheat Barley 40K array has minimal ascertainment bias. Since tagging all minor variants is not feasible using low-density arrays, a better solution is to add minor variants into future versions of the array as trait associations are discovered, essentially as we have currently done for published trait linked markers.</p>
<p>To drive efficiencies for large-scale genotyping in commercial breeding programs, we explored the limits of the Infinium bead chip technology. One advantage of this technology is that each oligonucleotide assay probe has a unique physical position on the bead chip. This allows for SNP arrays to be designed to genotype multiple crop species, with a user-defined number of SNPs assigned to each species. The Infinium Wheat Barley 40K array assays 25,363 SNPs in wheat and 14,261 SNPs in barley. To the best of our knowledge, multispecies SNP arrays have only been used to assay a single sample at a time. Here, we demonstrated that through careful selection of species-specific oligonucleotide probes, it is possible to jointly hybridize a wheat and barley sample to the same bead chip array, without substantial loss of genotype calling accuracy (<xref ref-type="fig" rid="F7">Figure 7</xref>). The selection of such probes is facilitated by our design concept which exploits LD to identify SNPs that can be considered equivalent for the purpose of genotyping. From a deployment perspective in a commercial breeding program, dual hybridization doubles genotyping throughput, since twice as many samples can be processed given the same amount of time and resources. Dual hybridization genotyping is potentially a game changing option for the adoption of genomics technologies by breeding companies with large numbers of samples that can be coordinated into genotyping.</p>
<p>To ensure broad utility in research and breeding, we added SNP-content capturing genetic diversity in the secondary and tertiary gene pools of wheat. This included 2,609 SNPs from the Infinium 90K SNP array (<xref ref-type="bibr" rid="B48">Wang et al., 2014</xref>) tagging tetraploid wheat (A- and B-genome) diversity and clearly delineating tetraploid wheat from other types of wheat, as well as tetraploid species and subgroups from one another. Each SNP is a single copy in tetraploid wheat and has been genetically and physically mapped (<xref ref-type="bibr" rid="B26">Maccaferri et al., 2019</xref>). It also included 1,206 single-copy SNP tagging <italic>Ae. tauschii</italic> (D-genome) diversity represented in 100 primary synthetic wheats, where each primary synthetic was derived from a different <italic>Ae. tauschii</italic> accession. Collectively, these SNPs provide broad utility ranging from the differentiation and genetic characterization of tetraploid and synthetic wheat (as well as other secondary and tertiary gene pools of wheat) to the tracking of introgressed genomic segments during breeding. Also included are SNPs that directly link to the Infinium 90K (<xref ref-type="bibr" rid="B48">Wang et al., 2014</xref>) and 15K (<xref ref-type="bibr" rid="B43">Soleimani et al., 2020</xref>) wheat arrays to ensure connectivity with legacy genotypic datasets and research. For barley, we included 685 SNPs that overlap with SNP reported for 19,778 GBS genotyped accessions from the IPK Genebank (<xref ref-type="bibr" rid="B29">Milner et al., 2019</xref>) to provide a direct anchor to that resource, and 1,239 SNPs that overlap with the Infinium 50K barley SNP array (<xref ref-type="bibr" rid="B4">Bayer et al., 2017</xref>) which link to 21,606 common SNPs following imputation. Finally, we included trait-linked SNPs and SNP tagging GWAS signals for key breeding and research targets reported in the published literature.</p>
<p>The overall array design makes it ideal for a wide range of research and breeding applications, from germplasm resource characterization, GWAS and genetic mapping to tracking introgressions from different sources, marker-assisted breeding and genomic selection. Its utility is further enhanced through the web-based tool, <italic>Pretzel</italic> (<xref ref-type="bibr" rid="B21">Keeble-Gagn&#x00E8;re et al., 2019</xref>; see text footnote 2) which enables the content of the array to be visualized and interrogated in real-time in the context of numerous genetic and genomic resources. For example, the SNPs can be visualized relative to the genetic and physical positions of other DNA marker types (e.g., SSRs, DArT), SNP on other genotyping arrays, trait loci, annotated genes, and syntenic positions in the genomes of other crops and model species. The ability to upload and visualize data in <italic>Pretzel</italic> allows breeders and researchers to seamlessly link and interrogate their own data in the context of publicly available datasets hosted in <italic>Pretzel</italic>. When combined with <italic>Pretzel</italic>, the Infinium Wheat Barley 40K array enables legacy and current research to seamlessly connect to breeding.</p>
</sec>
<sec id="S5" sec-type="conclusion">
<title>Conclusion</title>
<p>In conclusion, we have described a novel approach applicable to any animal or plant species for designing cost-effective imputation-enabled SNP genotyping arrays that have broad applicability in research and industry applications (e.g., GWAS, genomic prediction, and operational breeding) and support the hybridization of multiple samples to the same array. The utility of this design approach was demonstrated through its implementation to develop a new Infinium Wheat Barley 40K SNP array. In addition to supporting broad utility in research and breeding, this array can be used as a resource to connect genetic and genomic datasets generated across germplasm pools and time. The array is further supported by the publicly available web-tool <italic>Pretzel</italic> and is available for purchase by the international wheat and barley community from Illumina Ltd. (CA, United States), the manufacturer of the Infinium bead chip technology.</p>
</sec>
<sec id="S6" sec-type="data-availability">
<title>Data Availability Statement</title>
<p>Exome data used from <xref ref-type="bibr" rid="B41">Russell et al. (2016)</xref> and <xref ref-type="bibr" rid="B15">He et al. (2019)</xref> are accessible under EBI ENA project accession numbers <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="PRJEB8044">PRJEB8044</ext-link> and <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="PRJEB31218">PRJEB31218</ext-link>, respectively. The filtered set of exome genotype calls for accessions and SNP underpinning the LD analysis and tag SNP selection for wheat (<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.7910/DVN/5LVYI1">10.7910/DVN/5LVYI1</ext-link>) and barley (<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.7910/DVN/CUPAXD">10.7910/DVN/CUPAXD</ext-link>) as well as the D-genome synthetic derivative-enriched SNP matrix (<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.7910/DVN/0QEASF">10.7910/DVN/0QEASF</ext-link>) are available through Dataverse at <ext-link ext-link-type="uri" xlink:href="https://dataverse.harvard.edu/dataverse/WheatBarley40k_v1">https://dataverse.harvard.edu/dataverse/WheatBarley40k_v1</ext-link>. Information about the status of each SNP, including tag SNP set ID and whether the SNP passed design filters, is included in the INFO column. Illumina 90k iSelect genotypes for the accessions used to select tetraploid-specific content is available at <ext-link ext-link-type="uri" xlink:href="https://figshare.com/articles/dataset/Durum_Wheat_cv_Svevo_annotation/6984035">https://figshare.com/articles/dataset/Durum_Wheat_cv_Svevo_annotation/6984035</ext-link> (<xref ref-type="bibr" rid="B26">Maccaferri et al., 2019</xref>).</p>
</sec>
<sec id="S7">
<title>Author Contributions</title>
<p>RP performed LD analysis. GK-G selected tagging SNP, performed imputation analyses, and produced the final designs. KF and DW performed exome and whole genome sequencing, Infinium Wheat Barley 40K assays, and genotype calling. JT performed sequence alignments and genotype calling. HR, JG, AR, DMo, and DMu selected non-tagging SNPs and provided wheat and barley germplasm. TW, HD, JT, and MH conceived the project. GK-G and MH wrote the manuscript. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="conf1" sec-type="COI-statement">
<title>Conflict of Interest</title>
<p>HR, JG, AR, DMo, DMu, and TW were employed by InterGrain. The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="pudiscl1" sec-type="disclaimer">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<sec id="S8" sec-type="supplementary-material"><title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2021.756877/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpls.2021.756877/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.zip" id="FS1" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 1</label>
<caption><p>Cumulative number of SNPs tagged by tSNPs at <italic>r</italic><sup>2</sup> &#x2265; 0.90 in each chromosome in wheat and barley. Curves are shown until the first singleton SNP is reached on each chromosome.</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Data_Sheet_1.zip" id="FS2" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 2</label>
<caption><p>Principal component analysis (PCA) based on <bold>(A)</bold> 17,600 SNPs described by <xref ref-type="bibr" rid="B26">Maccaferri et al. (2019)</xref> from the Infinium wheat 90K SNP array and <bold>(B)</bold> 2,609 SNPs selected for inclusion on the Infinium Wheat Barley 40K SNP array showing differentiation among 1,856 tetraploid wheat accessions representing wild emmer wheat from North Eastern Fertile Crescent (WEW-NE), wild emmer wheat from Southern Levant Fertile Crescent (WEW-SL), domesticated emmer wheat (DEW), domesticated emmer wheat from Ethiopia (DEW-ETH), durum wheat landraces (DWL), and durum wheat cultivars (DWC).</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Data_Sheet_1.zip" id="FS3" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 3</label>
<caption><p>PCA based on <bold>(A)</bold> 37,105 called SNPs from the Infinium wheat 90K SNP array, and <bold>(B)</bold> 20,665 SNPs on the Infinium Wheat Barley 40K SNP array showing differentiation among bread wheat (green), synthetic derivatives (blue), and hexaploid wheat derived from crosses between bread and durum accessions (red) (number of accessions = 1,219).</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Data_Sheet_1.zip" id="FS4" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 4</label>
<caption><p>Distribution of selected SNP content across the wheat and barley genomes. Selected tSNPs (green), tetraploid wheat-specific SNPs shown with positions as reported in the durum genome by <xref ref-type="bibr" rid="B26">Maccaferri et al., 2019</xref> (blue) and synthetic wheat-derived SNPs (red).</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Data_Sheet_1.zip" id="FS5" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 5</label>
<caption><p>Overview of the design approach for the Wheat Barley 40K SNP array.</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Data_Sheet_1.zip" id="TS1" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Table 1</label>
<caption><p>Detailed description of Infinium Wheat Barley 40K SNP array content.</p></caption>
</supplementary-material>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Allen</surname> <given-names>A. M.</given-names></name> <name><surname>Winfield</surname> <given-names>M. O.</given-names></name> <name><surname>Burridge</surname> <given-names>A. J.</given-names></name> <name><surname>Downie</surname> <given-names>R. C.</given-names></name> <name><surname>Benbow</surname> <given-names>H. R.</given-names></name> <name><surname>Barker</surname> <given-names>G. L. A.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>Characterization of a Wheat Breeders&#x2019; Array suitable for high-throughput SNP genotyping of global accessions of hexaploidy bread wheat (<italic>Triticum aestivum</italic>).</article-title> <source><italic>Plant Biotechnol. J.</italic></source> <volume>15</volume> <fpage>390</fpage>&#x2013;<lpage>401</lpage>.</citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Avni</surname> <given-names>R.</given-names></name> <name><surname>Nave</surname> <given-names>M.</given-names></name> <name><surname>Barad</surname> <given-names>O.</given-names></name> <name><surname>Baruch</surname> <given-names>K.</given-names></name> <name><surname>Twardziok</surname> <given-names>S. O.</given-names></name> <name><surname>Gundlach</surname> <given-names>H.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>Wild emmer genome architecture and diversity elucidate wheat evolution and domestication.</article-title> <source><italic>Science</italic></source> <volume>357</volume> <fpage>93</fpage>&#x2013;<lpage>97</lpage>.</citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Balfourier</surname> <given-names>F.</given-names></name> <name><surname>Bouchet</surname> <given-names>S.</given-names></name> <name><surname>Robert</surname> <given-names>S.</given-names></name> <name><surname>De Oliveira</surname> <given-names>R.</given-names></name> <name><surname>Rimbert</surname> <given-names>H.</given-names></name> <name><surname>Kitt</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Worldwide phylogeography and history of wheat genetic diversity.</article-title> <source><italic>Science Adv.</italic></source> <volume>5</volume>:<issue>eaav0536</issue>. <pub-id pub-id-type="doi">10.1126/sciadv.aav0536</pub-id> <pub-id pub-id-type="pmid">31149630</pub-id></citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bayer</surname> <given-names>M. M.</given-names></name> <name><surname>Rapazote-Flores</surname> <given-names>P.</given-names></name> <name><surname>Ganal</surname> <given-names>M.</given-names></name> <name><surname>Hedley</surname> <given-names>P. E.</given-names></name> <name><surname>Macaulay</surname> <given-names>M.</given-names></name> <name><surname>Plieske</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>Development and evaluation of a barley 50k iSelect SNP array.</article-title> <source><italic>Front. Plant Sci.</italic></source> <volume>8</volume>:<issue>1792</issue>. <pub-id pub-id-type="doi">10.3389/fpls.2017.01792</pub-id> <pub-id pub-id-type="pmid">29089957</pub-id></citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Browning</surname> <given-names>S. R.</given-names></name> <name><surname>Browning</surname> <given-names>B. L.</given-names></name></person-group> (<year>2007</year>). <article-title>Rapid and accurate haplotype phasing and missing-data inference for whole-genome association studies by use of localized haplotype clustering.</article-title> <source><italic>Am. J. Hum. Genet.</italic></source> <volume>81</volume> <fpage>1084</fpage>&#x2013;<lpage>1097</lpage>. <pub-id pub-id-type="doi">10.1086/521987</pub-id> <pub-id pub-id-type="pmid">17924348</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Carlson</surname> <given-names>C. S.</given-names></name> <name><surname>Eberle</surname> <given-names>M. A.</given-names></name> <name><surname>Rieder</surname> <given-names>M. J.</given-names></name> <name><surname>Yi</surname> <given-names>Q.</given-names></name> <name><surname>Kruglyak</surname> <given-names>L.</given-names></name> <name><surname>Nickerson</surname> <given-names>D. A.</given-names></name></person-group> (<year>2004</year>). <article-title>Selecting a maximally informative set of single-nucleotide polymorphisms for association analyses using linkage disequilibrium.</article-title> <source><italic>Am. J. Hum. Genet.</italic></source> <volume>74</volume> <fpage>106</fpage>&#x2013;<lpage>120</lpage>. <pub-id pub-id-type="doi">10.1086/381000</pub-id> <pub-id pub-id-type="pmid">14681826</pub-id></citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cavanagh</surname> <given-names>C.</given-names></name> <name><surname>Chao</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Huang</surname> <given-names>B. E.</given-names></name> <name><surname>Stephen</surname> <given-names>S.</given-names></name> <name><surname>Kianic</surname> <given-names>S.</given-names></name><etal/></person-group> (<year>2013</year>). <article-title>Genome-wide comparative diversity uncovers multiple targets of selection for improvement in hexaploid wheat landraces and cultivars.</article-title> <source><italic>Proc. Natl. Acad. Sci. U. S. A.</italic></source> <volume>110</volume> <fpage>8057</fpage>&#x2013;<lpage>8062</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1217133110</pub-id> <pub-id pub-id-type="pmid">23630259</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chu</surname> <given-names>J.</given-names></name> <name><surname>Zhao</surname> <given-names>Y.</given-names></name> <name><surname>Beier</surname> <given-names>S.</given-names></name> <name><surname>Schulthess</surname> <given-names>A. W.</given-names></name> <name><surname>Stein</surname> <given-names>N.</given-names></name> <name><surname>Philipp</surname> <given-names>N.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Suitability of single-nucleotide polymorphism arrays versus genotyping-by-sequencing for genebank genomics in wheat.</article-title> <source><italic>Front. Plant Sci.</italic></source> <volume>14</volume>:<issue>42</issue>. <pub-id pub-id-type="doi">10.3389/fpls.2020.00042</pub-id> <pub-id pub-id-type="pmid">32117381</pub-id></citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Clark</surname> <given-names>A. G.</given-names></name> <name><surname>Hubisz</surname> <given-names>M. J.</given-names></name> <name><surname>Bustamante</surname> <given-names>C. D.</given-names></name> <name><surname>Williamson</surname> <given-names>S. H.</given-names></name> <name><surname>Nielsen</surname> <given-names>R.</given-names></name></person-group> (<year>2005</year>). <article-title>Ascertainment bias in studies of human genome-wide polymorphism.</article-title> <source><italic>Genome Res.</italic></source> <volume>15</volume> <fpage>1496</fpage>&#x2013;<lpage>1502</lpage>. <pub-id pub-id-type="doi">10.1101/gr.4107905</pub-id> <pub-id pub-id-type="pmid">16251459</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Comadran</surname> <given-names>J.</given-names></name> <name><surname>Kilian</surname> <given-names>B.</given-names></name> <name><surname>Russell</surname> <given-names>J.</given-names></name> <name><surname>Ramsay</surname> <given-names>L.</given-names></name> <name><surname>Stein</surname> <given-names>N.</given-names></name> <name><surname>Ganal</surname> <given-names>M.</given-names></name><etal/></person-group> (<year>2012</year>). <article-title>Natural variation in a homolog of Antirrhinum CENTRORADIALIS contributed to spring growth habit and environmental adaptation in cultivated barley.</article-title> <source><italic>Nat. Genet.</italic></source> <volume>44</volume> <fpage>1388</fpage>&#x2013;<lpage>1392</lpage>.</citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Das</surname> <given-names>S.</given-names></name> <name><surname>Forer</surname> <given-names>L.</given-names></name> <name><surname>Sch&#x00F6;nherr</surname> <given-names>S.</given-names></name> <name><surname>Sidore</surname> <given-names>C.</given-names></name> <name><surname>Locke</surname> <given-names>A.</given-names></name> <name><surname>Kwong</surname> <given-names>A.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>Next-generation genotype imputation service and methods.</article-title> <source><italic>Nat. Genet.</italic></source> <volume>48</volume> <fpage>1284</fpage>&#x2013;<lpage>1287</lpage>. <pub-id pub-id-type="doi">10.1038/ng.3656</pub-id> <pub-id pub-id-type="pmid">27571263</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>K.</given-names></name> <name><surname>Kullo</surname> <given-names>I. J.</given-names></name></person-group> (<year>2007</year>). <article-title>Methods for the selection of tagging SNPs: a comparison of tagging efficiency and performance.</article-title> <source><italic>Eur. J. Hum. Genet.</italic></source> <volume>15</volume> <fpage>228</fpage>&#x2013;<lpage>236</lpage>. <pub-id pub-id-type="doi">10.1038/sj.ejhg.5201755</pub-id> <pub-id pub-id-type="pmid">17164795</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Elbasyoni</surname> <given-names>I. S.</given-names></name> <name><surname>Lorenz</surname> <given-names>A. J.</given-names></name> <name><surname>Guttieri</surname> <given-names>M.</given-names></name> <name><surname>Frels</surname> <given-names>K.</given-names></name> <name><surname>Baenziger</surname> <given-names>P. S.</given-names></name> <name><surname>Poland</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>A comparison between genotyping-by-sequencing and array-based scoring of SNPs for genomic prediction accuracy in winter wheat.</article-title> <source><italic>Plant Sci. J.</italic></source> <volume>270</volume> <fpage>123</fpage>&#x2013;<lpage>130</lpage>. <pub-id pub-id-type="doi">10.1016/j.plantsci.2018.02.019</pub-id> <pub-id pub-id-type="pmid">29576064</pub-id></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fikere</surname> <given-names>M.</given-names></name> <name><surname>Barbulescu</surname> <given-names>D. M.</given-names></name> <name><surname>Malmberg</surname> <given-names>M. M.</given-names></name> <name><surname>Spangenberg</surname> <given-names>G. C.</given-names></name> <name><surname>Cogan</surname> <given-names>N. O. I.</given-names></name> <name><surname>Daetwyler</surname> <given-names>H. D.</given-names></name></person-group> (<year>2020</year>). <article-title>Meta-analysis of GWAS in canola blackleg (<italic>Leptosphaeria maculans</italic>) disease traits demonstrates increased power from imputed whole-genome sequence.</article-title> <source><italic>Sci. Rep.</italic></source> <volume>10</volume>:<issue>14300</issue>. <pub-id pub-id-type="doi">10.1038/s41598-020-71274-6</pub-id> <pub-id pub-id-type="pmid">32868838</pub-id></citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>F.</given-names></name> <name><surname>Pasam</surname> <given-names>R.</given-names></name> <name><surname>Shi</surname> <given-names>F.</given-names></name> <name><surname>Kant</surname> <given-names>S.</given-names></name> <name><surname>Keeble-Gagnere</surname> <given-names>G.</given-names></name> <name><surname>Kay</surname> <given-names>P.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Exome sequencing highlights the role of wild-relative introgression in shaping the adaptive landscape of the wheat genome.</article-title> <source><italic>Nat. Genet.</italic></source> <volume>51</volume> <fpage>896</fpage>&#x2013;<lpage>904</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-019-0382-2</pub-id> <pub-id pub-id-type="pmid">31043759</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hill</surname> <given-names>C. B.</given-names></name> <name><surname>Angessa</surname> <given-names>T.</given-names></name> <name><surname>McFawn</surname> <given-names>L.-A.</given-names></name> <name><surname>Wong</surname> <given-names>D.</given-names></name> <name><surname>Tibbits</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>X.-Q.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Hybridisation-based target enrichment of phenology genes to dissect the genetic basis of yield and adaptation in barley.</article-title> <source><italic>Plant Biotechnol. J.</italic></source> <volume>17</volume> <fpage>932</fpage>&#x2013;<lpage>944</lpage>. <pub-id pub-id-type="doi">10.1111/pbi.13029</pub-id> <pub-id pub-id-type="pmid">30407713</pub-id></citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hill</surname> <given-names>C. B.</given-names></name> <name><surname>Angessa</surname> <given-names>T. T.</given-names></name> <name><surname>Zhang</surname> <given-names>X.-Q.</given-names></name> <name><surname>Chen</surname> <given-names>K.</given-names></name> <name><surname>Zhou</surname> <given-names>G.</given-names></name> <name><surname>Tan</surname> <given-names>C.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>A global barley panel revealing genomic signatures of breeding in modern cultivars.</article-title> <source><italic>BioRxiv</italic></source> [<comment>Preprint</comment>]. <pub-id pub-id-type="doi">10.1101/2020.03.04.976324</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>D.</given-names></name> <name><surname>Zheng</surname> <given-names>Q.</given-names></name> <name><surname>Melchkart</surname> <given-names>T.</given-names></name> <name><surname>Bekkaoui</surname> <given-names>Y.</given-names></name> <name><surname>Konkin</surname> <given-names>D. J. F.</given-names></name> <name><surname>Kagale</surname> <given-names>S.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Dominant inhibition of awn development by a putative zinc-finger transcriptional repressor expressed at the B1 locus in wheat.</article-title> <source><italic>New Phytol.</italic></source> <volume>225</volume> <fpage>340</fpage>&#x2013;<lpage>355</lpage>. <pub-id pub-id-type="doi">10.1111/nph.16154</pub-id> <pub-id pub-id-type="pmid">31469444</pub-id></citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jordan</surname> <given-names>K. W.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Lun</surname> <given-names>Y.</given-names></name> <name><surname>Gardiner</surname> <given-names>L.-J.</given-names></name> <name><surname>MacLauchlan</surname> <given-names>R.</given-names></name> <name><surname>Hucl</surname> <given-names>P.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>A haplotype map of allohexaploid wheat reveals distinct patterns of selection on homoeologous genomes.</article-title> <source><italic>Genome Biol.</italic></source> <volume>16</volume>:<issue>48</issue>. <pub-id pub-id-type="doi">10.1186/s13059-015-0606-4</pub-id> <pub-id pub-id-type="pmid">25886949</pub-id></citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Joukhadar</surname> <given-names>R.</given-names></name> <name><surname>Daetwyler</surname> <given-names>H. D.</given-names></name> <name><surname>Bansal</surname> <given-names>U. K.</given-names></name> <name><surname>Gendall</surname> <given-names>A. R.</given-names></name> <name><surname>Hayden</surname> <given-names>M. J.</given-names></name></person-group> (<year>2017</year>). <article-title>Genetic diversity, population structure and ancestral origin of Australian wheat.</article-title> <source><italic>Front. Plant Sci.</italic></source> <volume>8</volume>:<issue>2115</issue>. <pub-id pub-id-type="doi">10.3389/fpls.2017.02115</pub-id> <pub-id pub-id-type="pmid">29312381</pub-id></citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Keeble-Gagn&#x00E8;re</surname> <given-names>G.</given-names></name> <name><surname>Isdale</surname> <given-names>D.</given-names></name> <name><surname>Suchecki</surname> <given-names>R.</given-names></name> <name><surname>Kruger</surname> <given-names>A.</given-names></name> <name><surname>Lomas</surname> <given-names>K.</given-names></name> <name><surname>Carroll</surname> <given-names>D.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Integrating past, present and future wheat research with Pretzel.</article-title> <source><italic>BioRxiv</italic></source> [<comment>Preprint</comment>]. <pub-id pub-id-type="doi">10.1101/517953</pub-id></citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lai</surname> <given-names>K.</given-names></name> <name><surname>Lorenc</surname> <given-names>M. T.</given-names></name> <name><surname>Lee</surname> <given-names>H. C.</given-names></name> <name><surname>Berkman</surname> <given-names>P. J.</given-names></name> <name><surname>Bayer</surname> <given-names>P. E.</given-names></name> <name><surname>Visendi</surname> <given-names>P.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>Identification and characterization of more than 4 million intervarietal SNPs across the group 7 chromosomes of bread wheat.</article-title> <source><italic>Plant Biotechnol. J.</italic></source> <volume>13</volume> <fpage>97</fpage>&#x2013;<lpage>104</lpage>. <pub-id pub-id-type="doi">10.1111/pbi.12240</pub-id> <pub-id pub-id-type="pmid">25147022</pub-id></citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ling</surname> <given-names>H. Q.</given-names></name> <name><surname>Ma</surname> <given-names>B.</given-names></name> <name><surname>Shi</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Dong</surname> <given-names>L.</given-names></name> <name><surname>Sun</surname> <given-names>H.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>Genome sequence of the progenitor of wheat A subgenome <italic>Triticum urartu</italic>.</article-title> <source><italic>Nature</italic></source> <volume>557</volume> <fpage>424</fpage>&#x2013;<lpage>428</lpage>.</citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Loh</surname> <given-names>P.-R.</given-names></name> <name><surname>Danecek</surname> <given-names>P.</given-names></name> <name><surname>Palamara</surname> <given-names>P. F.</given-names></name> <name><surname>Fuchsberger</surname> <given-names>C.</given-names></name> <name><surname>Reshef</surname> <given-names>Y. A.</given-names></name> <name><surname>Finucane</surname> <given-names>H. K.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>Reference-based phasing using the Haplotype Reference Consortium panel.</article-title> <source><italic>Nat. Genet.</italic></source> <volume>48</volume> <fpage>1443</fpage>&#x2013;<lpage>1448</lpage>. <pub-id pub-id-type="doi">10.1038/ng.3679</pub-id> <pub-id pub-id-type="pmid">27694958</pub-id></citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>M. C.</given-names></name> <name><surname>Gu</surname> <given-names>Y. Q.</given-names></name> <name><surname>Puiu</surname> <given-names>D.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Twardziok</surname> <given-names>S. O.</given-names></name> <name><surname>Deal</surname> <given-names>K. R.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>Genome sequence of the progenitor of the wheat D genome <italic>Aegilops tauschii</italic>.</article-title> <source><italic>Nature</italic></source> <volume>551</volume> <fpage>498</fpage>&#x2013;<lpage>502</lpage>.</citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Maccaferri</surname> <given-names>M.</given-names></name> <name><surname>Harris</surname> <given-names>N. S.</given-names></name> <name><surname>Twardziok</surname> <given-names>S. O.</given-names></name> <name><surname>Pasam</surname> <given-names>R. K.</given-names></name> <name><surname>Gundlach</surname> <given-names>H.</given-names></name> <name><surname>Spannagl</surname> <given-names>M.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Durum wheat genome highlights past domestication signatures and future improvement targets.</article-title> <source><italic>Nat. Genet.</italic></source> <volume>51</volume>:<issue>885</issue>. <pub-id pub-id-type="doi">10.1038/s41588-019-0381-3</pub-id> <pub-id pub-id-type="pmid">30962619</pub-id></citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mascher</surname> <given-names>M.</given-names></name> <name><surname>Gundlach</surname> <given-names>H.</given-names></name> <name><surname>Himmelbach</surname> <given-names>A.</given-names></name> <name><surname>Beier</surname> <given-names>S.</given-names></name> <name><surname>Twardziok</surname> <given-names>S. O.</given-names></name> <name><surname>Wicker</surname> <given-names>T.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>A chromosome conformation capture ordered sequence of the barley genome.</article-title> <source><italic>Nature</italic></source> <volume>544</volume> <fpage>427</fpage>&#x2013;<lpage>433</lpage>.</citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mascher</surname> <given-names>M.</given-names></name> <name><surname>Schreiber</surname> <given-names>M.</given-names></name> <name><surname>Scholz</surname> <given-names>U.</given-names></name> <name><surname>Graner</surname> <given-names>A.</given-names></name> <name><surname>Reif</surname> <given-names>J. C.</given-names></name> <name><surname>Stein</surname> <given-names>N.</given-names></name></person-group> (<year>2019</year>). <article-title>Genebank genomics bridges the gap between the conservation of crop diversity and plant breeding.</article-title> <source><italic>Nat. Genet.</italic></source> <volume>51</volume> <fpage>1076</fpage>&#x2013;<lpage>1091</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-019-0443-6</pub-id> <pub-id pub-id-type="pmid">31253974</pub-id></citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Milner</surname> <given-names>S. G.</given-names></name> <name><surname>Jost</surname> <given-names>M.</given-names></name> <name><surname>Taketa</surname> <given-names>S.</given-names></name> <name><surname>Maz&#x00F3;n</surname> <given-names>E. R.</given-names></name> <name><surname>Himmelbach</surname> <given-names>A.</given-names></name> <name><surname>Oppermann</surname> <given-names>M.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Genebank genomics highlights the diversity of a global barley collection.</article-title> <source><italic>Nat. Genet.</italic></source> <volume>51</volume> <fpage>319</fpage>&#x2013;<lpage>326</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-018-0266-x</pub-id> <pub-id pub-id-type="pmid">30420647</pub-id></citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Monat</surname> <given-names>C.</given-names></name> <name><surname>Padmarasu</surname> <given-names>S.</given-names></name> <name><surname>Lux</surname> <given-names>T.</given-names></name> <name><surname>Wicker</surname> <given-names>T.</given-names></name> <name><surname>Gundlach</surname> <given-names>H.</given-names></name> <name><surname>Himmelbach</surname> <given-names>A.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>TRITEX: chromosome-scale sequence assembly of Triticeae genomes with open-source tools.</article-title> <source><italic>Genome Biol.</italic></source> <volume>20</volume>:<issue>284</issue>.</citation></ref>
<ref id="B31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Money</surname> <given-names>D.</given-names></name> <name><surname>Gardner</surname> <given-names>K.</given-names></name> <name><surname>Migicovsky</surname> <given-names>Z.</given-names></name> <name><surname>Schwaninger</surname> <given-names>H.</given-names></name> <name><surname>Zhong</surname> <given-names>G.-Y.</given-names></name> <name><surname>Myles</surname> <given-names>S.</given-names></name></person-group> (<year>2015</year>). <article-title>LinkImpute: fast and accurate genotype imputation for non-model organisms.</article-title> <source><italic>G3</italic></source> <volume>5</volume> <fpage>2383</fpage>&#x2013;<lpage>2390</lpage>. <pub-id pub-id-type="doi">10.1534/g3.115.021667</pub-id> <pub-id pub-id-type="pmid">26377960</pub-id></citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Montenegro</surname> <given-names>J. D.</given-names></name> <name><surname>Golicz</surname> <given-names>A. A.</given-names></name> <name><surname>Bayer</surname> <given-names>P. E.</given-names></name> <name><surname>Hurgobin</surname> <given-names>B.</given-names></name> <name><surname>Lee</surname> <given-names>H.</given-names></name> <name><surname>Chan</surname> <given-names>C.-K. K.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>The pangenome of hexaploidy bread wheat.</article-title> <source><italic>Plant J.</italic></source> <volume>90</volume> <fpage>1007</fpage>&#x2013;<lpage>1013</lpage>. <pub-id pub-id-type="doi">10.1111/tpj.13515</pub-id> <pub-id pub-id-type="pmid">28231383</pub-id></citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Negro</surname> <given-names>S. S.</given-names></name> <name><surname>Millet</surname> <given-names>E. J.</given-names></name> <name><surname>Madur</surname> <given-names>D.</given-names></name> <name><surname>Bauland</surname> <given-names>C.</given-names></name> <name><surname>Combes</surname> <given-names>V.</given-names></name> <name><surname>Welcker</surname> <given-names>C.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Genotyping-by-sequencing and SNP-arrays are complementary for detecting quantitative trait loci by tagging different haplotypes in association studies.</article-title> <source><italic>BMC Plant Biol.</italic></source> <volume>19</volume>:<issue>318</issue>. <pub-id pub-id-type="doi">10.1186/s12870-019-1926-4</pub-id> <pub-id pub-id-type="pmid">31311506</pub-id></citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nyine</surname> <given-names>M.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Kiani</surname> <given-names>K.</given-names></name> <name><surname>Jordan</surname> <given-names>K.</given-names></name> <name><surname>Liu</surname> <given-names>S.</given-names></name> <name><surname>Byrne</surname> <given-names>P.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Genotype imputation in winter wheat using first-generation haplotype map SNPs improves genome-wide association mapping and genomic prediction of traits.</article-title> <source><italic>G3</italic></source> <volume>9</volume> <fpage>125</fpage>&#x2013;<lpage>133</lpage>. <pub-id pub-id-type="doi">10.1534/g3.118.200664</pub-id> <pub-id pub-id-type="pmid">30420469</pub-id></citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ogbonnaya</surname> <given-names>F. C.</given-names></name> <name><surname>Ye</surname> <given-names>G.</given-names></name> <name><surname>Trethowan</surname> <given-names>R.</given-names></name> <name><surname>Dreccer</surname> <given-names>F.</given-names></name> <name><surname>Lush</surname> <given-names>D.</given-names></name> <name><surname>Shepperd</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2007</year>). <article-title>Yield of synthetic backcross-derived lines in rainfed environments of Australia.</article-title> <source><italic>Euphytica</italic></source> <volume>157</volume> <fpage>321</fpage>&#x2013;<lpage>336</lpage>. <pub-id pub-id-type="doi">10.1007/s10681-007-9381-y</pub-id></citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pasam</surname> <given-names>R. P.</given-names></name> <name><surname>Bansal</surname> <given-names>U.</given-names></name> <name><surname>Daetwyler</surname> <given-names>H. D.</given-names></name> <name><surname>Forrest</surname> <given-names>K. L.</given-names></name> <name><surname>Wong</surname> <given-names>D.</given-names></name> <name><surname>Petkowski</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>Detection and validation of genomic regions associated with three rust resistances to rust diseases in a worldwide hexaploid wheat landrace collection using BayesR and Mixed Linear Model approaches.</article-title> <source><italic>Theor. Appl. Genet.</italic></source> <volume>130</volume> <fpage>777</fpage>&#x2013;<lpage>793</lpage>. <pub-id pub-id-type="doi">10.1007/s00122-016-2851-7</pub-id> <pub-id pub-id-type="pmid">28255670</pub-id></citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pont</surname> <given-names>C.</given-names></name> <name><surname>Leroy</surname> <given-names>T.</given-names></name> <name><surname>Seidel</surname> <given-names>M.</given-names></name> <name><surname>Tondelli</surname> <given-names>A.</given-names></name> <name><surname>Duchemin</surname> <given-names>W.</given-names></name> <name><surname>Armisen</surname> <given-names>D.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Tracing the ancestry of modern bread wheats.</article-title> <source><italic>Nat. Genet.</italic></source> <volume>51</volume> <fpage>905</fpage>&#x2013;<lpage>911</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-019-0393-z</pub-id> <pub-id pub-id-type="pmid">31043760</pub-id></citation></ref>
<ref id="B38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Purcell</surname> <given-names>S.</given-names></name> <name><surname>Neale</surname> <given-names>B.</given-names></name> <name><surname>Todd-Brown</surname> <given-names>K.</given-names></name> <name><surname>Thomas</surname> <given-names>L.</given-names></name> <name><surname>Ferreira</surname> <given-names>M. A. R.</given-names></name> <name><surname>Bender</surname> <given-names>D.</given-names></name><etal/></person-group> (<year>2007</year>). <article-title>PLINK: a tool set for whole-genome association and population-based linkage analyses.</article-title> <source><italic>Am. J. Hum. Genet.</italic></source> <volume>81</volume> <fpage>559</fpage>&#x2013;<lpage>575</lpage>. <pub-id pub-id-type="doi">10.1086/519795</pub-id> <pub-id pub-id-type="pmid">17701901</pub-id></citation></ref>
<ref id="B39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rasheed</surname> <given-names>A.</given-names></name> <name><surname>Hao</surname> <given-names>Y.</given-names></name> <name><surname>Xia</surname> <given-names>X.</given-names></name> <name><surname>Khan</surname> <given-names>A.</given-names></name> <name><surname>Xu</surname> <given-names>Y.</given-names></name> <name><surname>Varshney</surname> <given-names>R. K.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>Crop breeding chips and genotyping platforms: progress, challenges, and perspectives.</article-title> <source><italic>Mol. Plant</italic></source> <volume>10</volume> <fpage>1047</fpage>&#x2013;<lpage>1064</lpage>. <pub-id pub-id-type="doi">10.1016/j.molp.2017.06.008</pub-id> <pub-id pub-id-type="pmid">28669791</pub-id></citation></ref>
<ref id="B40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rimbert</surname> <given-names>H.</given-names></name> <name><surname>Darrier</surname> <given-names>B.</given-names></name> <name><surname>Navarro</surname> <given-names>J.</given-names></name> <name><surname>Kitt</surname> <given-names>J.</given-names></name> <name><surname>Choulet</surname> <given-names>F.</given-names></name> <name><surname>Leveugle</surname> <given-names>M.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>High throughput SNP discovery and genotyping in hexaploid wheat.</article-title> <source><italic>PLoS One</italic></source> <volume>13</volume>:<issue>e0186329</issue>. <pub-id pub-id-type="doi">10.1371/journal.pone.0186329</pub-id> <pub-id pub-id-type="pmid">29293495</pub-id></citation></ref>
<ref id="B41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Russell</surname> <given-names>J.</given-names></name> <name><surname>Mascher</surname> <given-names>M.</given-names></name> <name><surname>Dawson</surname> <given-names>I. K.</given-names></name> <name><surname>Kyriakidis</surname> <given-names>S.</given-names></name> <name><surname>Calixto</surname> <given-names>C.</given-names></name> <name><surname>Freund</surname> <given-names>F.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>Exome sequencing of geographically diverse barley landraces and wild relatives gives insights into environmental adaptation.</article-title> <source><italic>Nat. Genet.</italic></source> <volume>48</volume> <fpage>1024</fpage>&#x2013;<lpage>1030</lpage>. <pub-id pub-id-type="doi">10.1038/ng.3612</pub-id> <pub-id pub-id-type="pmid">27428750</pub-id></citation></ref>
<ref id="B42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shi</surname> <given-names>C.</given-names></name> <name><surname>Zhao</surname> <given-names>L.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Lv</surname> <given-names>G.</given-names></name> <name><surname>Pan</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>F.</given-names></name></person-group> (<year>2019</year>). <article-title>Gene regulatory network and abundant genetic variation play critical roles in heading stage of polyploidy wheat.</article-title> <source><italic>BMC Plant Biol.</italic></source> <volume>19</volume>:<issue>6</issue>. <pub-id pub-id-type="doi">10.1186/s12870-018-1591-z</pub-id> <pub-id pub-id-type="pmid">30606101</pub-id></citation></ref>
<ref id="B43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Soleimani</surname> <given-names>B.</given-names></name> <name><surname>Lehnert</surname> <given-names>H.</given-names></name> <name><surname>Keilwagen</surname> <given-names>J.</given-names></name> <name><surname>Plieske</surname> <given-names>J.</given-names></name> <name><surname>Ordon</surname> <given-names>F.</given-names></name> <name><surname>Naseri Rad</surname> <given-names>S.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Comparison between core set selection methods using different Illumina marker platforms: a case study of assessment of diversity in wheat.</article-title> <source><italic>Front. Plant Sci.</italic></source> <volume>11</volume>:<issue>1040</issue>. <pub-id pub-id-type="doi">10.3389/fpls.2020.01040</pub-id> <pub-id pub-id-type="pmid">32754184</pub-id></citation></ref>
<ref id="B44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>C.</given-names></name> <name><surname>Dong</surname> <given-names>Z.</given-names></name> <name><surname>Zhao</surname> <given-names>L.</given-names></name> <name><surname>Ren</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>N.</given-names></name> <name><surname>Chen</surname> <given-names>F.</given-names></name></person-group> (<year>2020</year>). <article-title>The Wheat 660K SNP array demonstrates great potential for marker-assisted selection in polyploid wheat.</article-title> <source><italic>Plant Biotechnol. J.</italic></source> <volume>18</volume> <fpage>1354</fpage>&#x2013;<lpage>1360</lpage>. <pub-id pub-id-type="doi">10.1111/pbi.13361</pub-id> <pub-id pub-id-type="pmid">32065714</pub-id></citation></ref>
<ref id="B45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>C.</given-names></name> <name><surname>Zhang</surname> <given-names>F.</given-names></name> <name><surname>Yan</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Dong</surname> <given-names>Z.</given-names></name> <name><surname>Cui</surname> <given-names>D.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>Genome-wide association study for 13 agronomic traits reveals distribution of superior alleles in bread wheat from the Yellow and Huai Valley of China</article-title>. <source><italic>Plant Biotechnol. J.</italic></source> <volume>15</volume>, <fpage>953</fpage>&#x2013;<lpage>969</lpage>. <pub-id pub-id-type="doi">10.1111/pbi.12690</pub-id> <pub-id pub-id-type="pmid">28055148</pub-id></citation></ref>
<ref id="B46"><citation citation-type="journal"><collab>The International Wheat Genome Sequencing Consortium [IWGSC]</collab> (<year>2018</year>). <article-title>Shifting the limits in wheat research and breeding using a fully annotated reference genome.</article-title> <source><italic>Science</italic></source> <volume>361</volume>:<issue>eaar7191</issue>.</citation></ref>
<ref id="B47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Walkowiak</surname> <given-names>S.</given-names></name> <name><surname>Gao</surname> <given-names>L.</given-names></name> <name><surname>Monat</surname> <given-names>C.</given-names></name> <name><surname>Haberer</surname> <given-names>G.</given-names></name> <name><surname>Kassa</surname> <given-names>M. T.</given-names></name> <name><surname>Brinton</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Multiple wheat genomes reveal global variation in modern breeding.</article-title> <source><italic>Nature</italic></source> <volume>588</volume> <fpage>277</fpage>&#x2013;<lpage>283</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-020-2961-x</pub-id> <pub-id pub-id-type="pmid">33239791</pub-id></citation></ref>
<ref id="B48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Wong</surname> <given-names>D.</given-names></name> <name><surname>Forrest</surname> <given-names>K.</given-names></name> <name><surname>Allen</surname> <given-names>A.</given-names></name> <name><surname>Chao</surname> <given-names>S.</given-names></name> <name><surname>Huang</surname> <given-names>B. E.</given-names></name><etal/></person-group> (<year>2014</year>). <article-title>Characterization of polyploid wheat genomic diversity using a high-density 90,000 single nucleotide polymorphism array.</article-title> <source><italic>Plant Biotechnol. J.</italic></source> <volume>12</volume> <fpage>787</fpage>&#x2013;<lpage>796</lpage>. <pub-id pub-id-type="doi">10.1111/pbi.12183</pub-id> <pub-id pub-id-type="pmid">24646323</pub-id></citation></ref>
<ref id="B49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>S. X.</given-names></name> <name><surname>Zhu</surname> <given-names>Y. L.</given-names></name> <name><surname>Zhang</surname> <given-names>D. X.</given-names></name> <name><surname>Shao</surname> <given-names>H.</given-names></name> <name><surname>Liu</surname> <given-names>P.</given-names></name> <name><surname>Hu</surname> <given-names>J. B.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>Genome-wide association study for grain yield and related traits in elite wheat varieties and advanced lines using SNP markers</article-title>. <source><italic>PLoS One.</italic></source> <volume>12</volume>:<issue>e0188662</issue>. <pub-id pub-id-type="doi">10.1371/journal.pone.0188662</pub-id> <pub-id pub-id-type="pmid">29176820</pub-id></citation></ref>
<ref id="B50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Weir</surname> <given-names>B. S.</given-names></name> <name><surname>Cockerham</surname> <given-names>C. C.</given-names></name></person-group> (<year>1984</year>). <article-title>Estimating F-statistics for the analysis of population structure.</article-title> <source><italic>Evolution</italic></source> <volume>38</volume> <fpage>1358</fpage>&#x2013;<lpage>1370</lpage>. <pub-id pub-id-type="doi">10.2307/2408641</pub-id></citation></ref>
<ref id="B51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wickham</surname> <given-names>H.</given-names></name></person-group> (<year>2016</year>). <source><italic>ggplot2: Elegant Graphics for Data Analysis.</italic></source> <publisher-loc>New York</publisher-loc>: <publisher-name>Springer-Verlag</publisher-name>.</citation></ref>
<ref id="B52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Winfield</surname> <given-names>M. O.</given-names></name> <name><surname>Allen</surname> <given-names>A. M.</given-names></name> <name><surname>Burridge</surname> <given-names>A. J.</given-names></name> <name><surname>Barker</surname> <given-names>G. L.</given-names></name> <name><surname>Benbow</surname> <given-names>H. R.</given-names></name> <name><surname>Wilkinson</surname> <given-names>P. A.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>High-density SNP genotyping array for hexaploid wheat and its secondary and tertiary gene pool.</article-title> <source><italic>Plant Biotechnol. J.</italic></source> <volume>14</volume> <fpage>1195</fpage>&#x2013;<lpage>1206</lpage>. <pub-id pub-id-type="doi">10.1111/pbi.12485</pub-id> <pub-id pub-id-type="pmid">26466852</pub-id></citation></ref>
<ref id="B53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>J.</given-names></name> <name><surname>Lee</surname> <given-names>S. H.</given-names></name> <name><surname>Goddard</surname> <given-names>M. E.</given-names></name> <name><surname>Visscher</surname> <given-names>P. M.</given-names></name></person-group> (<year>2011</year>). <article-title>GCTA: a tool for genome-wide complex trait analysis.</article-title> <source><italic>Am. J. Hum. Genet.</italic></source> <volume>88</volume> <fpage>76</fpage>&#x2013;<lpage>82</lpage>.</citation></ref>
<ref id="B54"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname> <given-names>X.</given-names></name> <name><surname>Levine</surname> <given-names>D.</given-names></name> <name><surname>Shen</surname> <given-names>J.</given-names></name> <name><surname>Gogarten</surname> <given-names>S.</given-names></name> <name><surname>Laurie</surname> <given-names>C.</given-names></name> <name><surname>Weir</surname> <given-names>B.</given-names></name></person-group> (<year>2012</year>). <article-title>A High-performance Computing Toolset for Relatedness and Principal Component Analysis of SNP Data.</article-title> <source><italic>Bioinformatics</italic></source> <volume>28</volume> <fpage>3326</fpage>&#x2013;<lpage>3328</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bts606</pub-id> <pub-id pub-id-type="pmid">23060615</pub-id></citation></ref>
<ref id="B55"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>T.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name> <name><surname>Rimbert</surname> <given-names>H.</given-names></name> <name><surname>Rodriguez</surname> <given-names>J. C.</given-names></name> <name><surname>Deal</surname> <given-names>K. R.</given-names></name> <name><surname>De Oliveira</surname> <given-names>R.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Optical maps refine the bread wheat Triticum aestivum cv. Chinese Spring genome assembly.</article-title> <source><italic>Plant J.</italic></source> <volume>107</volume> <fpage>303</fpage>&#x2013;<lpage>314</lpage>. <pub-id pub-id-type="doi">10.1111/tpj.15289</pub-id> <pub-id pub-id-type="pmid">33893684</pub-id></citation></ref>
</ref-list>
<fn-group>
<fn id="footnote1">
<label>1</label>
<p><ext-link ext-link-type="uri" xlink:href="https://www.earthbiogenome.org/">https://www.earthbiogenome.org/</ext-link></p></fn>
<fn id="footnote2">
<label>2</label>
<p><ext-link ext-link-type="uri" xlink:href="https://plantinformatics.io/">https://plantinformatics.io/</ext-link></p></fn>
<fn id="footnote3">
<label>3</label>
<p><ext-link ext-link-type="uri" xlink:href="http://wheatgenomics.plantpath.ksu.edu/1000EC/">http://wheatgenomics.plantpath.ksu.edu/1000EC/</ext-link></p></fn>
<fn id="footnote4">
<label>4</label>
<p><ext-link ext-link-type="uri" xlink:href="http://www.intergrain.com">www.intergrain.com</ext-link></p></fn>
<fn id="footnote5">
<label>5</label>
<p><ext-link ext-link-type="uri" xlink:href="https://urgi.versailles.inra.fr/download/iwgsc/IWGSC_RefSeq_Assemblies/v2.0/">https://urgi.versailles.inra.fr/download/iwgsc/IWGSC_RefSeq_Assemblies/v2.0/</ext-link></p></fn>
<fn id="footnote6">
<label>6</label>
<p><ext-link ext-link-type="uri" xlink:href="https://www.r-project.org/">https://www.r-project.org/</ext-link></p></fn>
</fn-group>
</back>
</article>
