<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2023.1103857</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>An improved reference genome for <italic>Trifolium subterraneum</italic> L. provides insight into molecular diversity and intra-specific phylogeny</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Shirasawa</surname>
<given-names>Kenta</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/343803"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Moraga</surname>
<given-names>Roger</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1831975"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ghelfi</surname>
<given-names>Andrea</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/462447"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hirakawa</surname>
<given-names>Hideki</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/469098"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Nagasaki</surname>
<given-names>Hideki</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/473937"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ghamkhar</surname>
<given-names>Kioumars</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/384189"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Barrett</surname>
<given-names>Brent A.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Griffiths</surname>
<given-names>Andrew G.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<xref ref-type="author-notes" rid="fn004">
<sup>&#x2021;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1828014"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Isobe</surname>
<given-names>Sachiko N.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<xref ref-type="author-notes" rid="fn004">
<sup>&#x2021;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/96272"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Frontier Research and Development, Kazusa DNA Research Institute</institution>, <addr-line>Kisarazu</addr-line>, <country>Japan</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>AgResearch, Grasslands Research Centre</institution>, <addr-line>Palmerston North</addr-line>, <country>New Zealand</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Tea Break Bioinformatics Limited</institution>, <addr-line>Palmerston North</addr-line>, <country>New Zealand</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Bioinformation and DDBJ Center, National Institute of Genetics</institution>, <addr-line>Mishima</addr-line>, <country>Japan</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Pierre Sourdille, INRAE Clermont-Auvergne-Rh&#xf4;ne-Alpes, France</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Fr&#xe9;d&#xe9;ric Choulet, INRA UMR1095 G&#xe9;n&#xe9;tique, Diversit&#xe9;, Ecophysiologie des C&#xe9;r&#xe9;ales (GDEC), France; Leif Skot, Aberystwyth University, United Kingdom</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Andrew G. Griffiths, <email xlink:href="mailto:andrew.griffiths@agresearch.co.nz">andrew.griffiths@agresearch.co.nz</email>; Sachiko N. Isobe, <email xlink:href="mailto:sisobe@kazusa.or.jp">sisobe@kazusa.or.jp</email>
</p>
</fn>
<fn fn-type="equal" id="fn003">
<p>&#x2020;These authors have contributed equally to this work</p>
</fn>
<fn fn-type="other" id="fn004">
<p>&#x2021;These authors share last authorship</p>
</fn>
<fn fn-type="other" id="fn002">
<p>This article was submitted to Plant Breeding, a section of the journal Frontiers in Plant Science</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>15</day>
<month>02</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1103857</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>11</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>30</day>
<month>01</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Shirasawa, Moraga, Ghelfi, Hirakawa, Nagasaki, Ghamkhar, Barrett, Griffiths and Isobe</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Shirasawa, Moraga, Ghelfi, Hirakawa, Nagasaki, Ghamkhar, Barrett, Griffiths and Isobe</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Subterranean clover (<italic>Trifolium subterraneum</italic> L., Ts) is a geocarpic, self-fertile annual forage legume with a compact diploid genome (n = x = 8, 544 Mb/1C). Its resilience and climate adaptivity have made it an economically important species in Mediterranean and temperate zones. Using the cultivar Daliak, we generated higher resolution sequence data, created a new genome assembly TSUd_3.0, and conducted molecular diversity analysis for copy number variant (CNV) and single-nucleotide polymorphism (SNP) among 36 cultivars. TSUd_3.0 substantively improves prior genome assemblies with new Hi-C and long-read sequence data, covering 531 Mb, containing 41,979 annotated genes and generating a 94.4% BUSCO score. Comparative genomic analysis among select members of the tribe Trifolieae indicated TSUd 3.0 corrects six assembly-error inversion/duplications and confirmed phylogenetic relationships. Its synteny with <italic>T. pratense</italic>, <italic>T. repens</italic>, <italic>Medicago truncatula</italic> and <italic>Lotus japonicus</italic> genomes were assessed, with the more distantly related <italic>T. repens</italic> and <italic>M</italic>. <italic>truncatula</italic> showing higher levels of co-linearity with Ts than between Ts and its close relative <italic>T. pratense</italic>. Resequencing of 36 cultivars discovered 7,789,537 SNPs subsequently used for genomic diversity assessment and sequence-based clustering. Heterozygosity estimates ranged from 1% to 21% within the 36 cultivars and may be influenced by admixture. Phylogenetic analysis supported subspecific genetic structure, although it indicates four or five groups, rather than the three recognized subspecies. Furthermore, there were incidences where cultivars characterized as belonging to a particular subspecies clustered with another subspecies when using genomic data. These outcomes suggest that further investigation of Ts sub-specific classification using molecular and morpho-physiological data is needed to clarify these relationships. This upgraded reference genome, complemented with comprehensive sequence diversity analysis of 36 cultivars, provides a platform for future gene functional analysis of key traits, and genome-based breeding strategies for climate adaptation and agronomic performance. Pangenome analysis, more in-depth intra-specific phylogenomic analysis using the Ts core collection, and functional genetic and genomic studies are needed to further augment knowledge of <italic>Trifolium</italic> genomes.</p>
</abstract>
<kwd-group>
<kwd>whole genome assembly</kwd>
<kwd>intra-specific phylogenomics</kwd>
<kwd>germplasm accession integrity</kwd>
<kwd>legume macrosynteny</kwd>
<kwd>subterranean clover (<italic>Trifolium subterraneum</italic> L.)</kwd>
</kwd-group>
<contract-sponsor id="cn001">Ministry of Business, Innovation and Employment<named-content content-type="fundref-id">10.13039/501100003524</named-content>
</contract-sponsor>
<counts>
<fig-count count="6"/>
<table-count count="1"/>
<equation-count count="0"/>
<ref-count count="81"/>
<page-count count="17"/>
<word-count count="10827"/>
</counts>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<title>Introduction</title>
<p>Legumes have a distinctive floral structure, podded fruit, and the ability by 88% of the species examined to date to form nodules with rhizobia and fix atmospheric nitrogen (<xref ref-type="bibr" rid="B16">Chaulagain and Frugoli, 2021</xref>). Food and feed legumes are the second most cultivated group of plants, after cereals, occupying 11% of the world&#x2019;s agricultural land. They are used for food, such as soybean (<italic>Glycine ma</italic>x (L.) Merr.), beans (<italic>Phaseolus vulgaris</italic> L. or <italic>Vicia faba</italic> L.), and lentil (<italic>Lens culinaris</italic> Medik.), as well as animal feed such as lucerne (<italic>Medicago sativa</italic> L., Ms) and many species of clover (<italic>Trifolium</italic> L.).</p>
<p>Subterranean clover (<italic>Trifolium subterraneum</italic> L.; Ts) is endemic to the Mediterranean region, West Asia and the Atlantic coast of Western Europe (<xref ref-type="bibr" rid="B57">Morley and Katznelson, 1965</xref>; <xref ref-type="bibr" rid="B81">Zohary and Heller, 1984</xref>). It has been introduced successfully as a forage in many countries with similar climates including Argentina, Australia, Chile, New Zealand, South Africa, Uruguay and the USA (<xref ref-type="bibr" rid="B60">Nichols et&#xa0;al., 2013</xref>). The geocarpic ability to bury its seed-containing burrs and the grazing tolerance of Ts under suboptimal environmental conditions (<xref ref-type="bibr" rid="B60">Nichols et&#xa0;al., 2013</xref>) makes it a popular pasture in countries with production zones ill-suited to perennial pasture cover. It exhibits genetic variation for methanogenic potential (<xref ref-type="bibr" rid="B4">Banik et&#xa0;al., 2013</xref>; <xref ref-type="bibr" rid="B59">Muir et&#xa0;al., 2020</xref>), enabling lower-emission farm systems. It is also a source of diversity for adaptive response to different environmental conditions and stress, especially in environments where perennial pastures fail or are marginal.</p>
<p>Genome information enables species characterization, comparative analysis, and genetic improvement for breeding. Two Ts linkage maps have been developed using microsatellite (<xref ref-type="bibr" rid="B31">Ghamkhar et&#xa0;al., 2012</xref>) and single-nucleotide polymorphism (SNP) (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>) markers. Both span ~2,000 centimorgans, with SNP markers increasing map resolution. Additional resources available to improve this species or target varieties for different uses includes: a core collection of Ts germplasm providing a genetically representative subset of global material is available (<xref ref-type="bibr" rid="B1">Abdi et&#xa0;al., 2020</xref>); and a diverse range of over 40 cultivars with valuable traits for specific environments (<xref ref-type="bibr" rid="B60">Nichols et&#xa0;al., 2013</xref>).</p>
<p>Developing an enhanced Ts genome sequence extends the genomic resources for genus <italic>Trifolium</italic>. These genomic resources will be particularly important for elucidating the genetic basis of geocarpy, abiotic stress adaptation, methanogenesis, and animal nutrition. Compared with red clover (n=2x=14, <italic>T. pratense</italic>, Tp) and white clover (n=2x=16, <italic>T. repens</italic>, Tr), which are outbreeding perennials, Ts has a simpler genetic and genomic structure. It is an annual autogamous diploid (2n = 2x = 16), with a genome size of ~ 544 Mb/1C (<xref ref-type="bibr" rid="B74">Vi&#x17e;intin et&#xa0;al., 2006</xref>).</p>
<p>Previous Ts reference genome assemblies showed a stepwise improvement in quality metrics as a representation of the Ts genome (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>). A draft Ts assembly (TSUd_r1.1) reported by <xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al. (2016)</xref> was based on sequence data comprising Roche 454 and Illumina HiSeq reads with a range of insert sizes, which were assembled, ordered by linkage mapping, and contained approximately 43,000 annotated genes. The assembly accounts for 90% (472 Mb) of the estimated Ts genome size of 544 Mb/1C and 401 Mb was placed into eight constructed chromosomes. Synteny reported by <xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al. (2016)</xref> included the genome sequences of the forage legume <italic>Medicago truncatula</italic> (Mt), Tp and <italic>Lotus japonicus</italic> (Lj). While a significant resource, the absence of physical mapping resources resulted in an assembly with syntenic inconsistencies. This was particularly noticeable with Mt which highlighted inversions suggesting possible genome assembly duplications in Chr2, 4, 5 and 7 of the TSUd_r1.1 Ts assembly relative to Mt (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>An overview showing the relationship among existing <italic>Trifolium subterraneum</italic> (Ts) reference genomes (TSUd_r1.1; Tsub_Refv2.0; TrSub3) and the strategy for sequencing and assembling the reference genome (TSUd_r3.0) in the current study. The percent genome encompassed by each assembly is calculated on a Ts size of 544 Mb/1C (<xref ref-type="bibr" rid="B74">Vi&#x17e;intin et&#xa0;al., 2006</xref>) which may differ from the value used in the various assembly publications. &#x2018;Daliak&#x2019; = Ts cultivar used for developing the reference genome; BUSCO = Benchmarking Universal Single-Copy Orthologues scores to assess assembly gene coverage based on 1,440 reference genes; Chr 0 = sequence data not assigned to the eight assembled chromosomes; Gb, Gigabase; Hi-C, high-throughput chromosome conformation capture analysis; MP, Illumina mate-pair sequences; PE, Illumina paired-end sequences; SE, Illumina single-end sequences; TSLR, Illumina TruSeq Synthetic Long-read sequence.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1103857-g001.tif"/>
</fig>
<p>The next assembly, Tsub_Refv2.0 (<xref ref-type="bibr" rid="B43">Kaur et&#xa0;al., 2017b</xref>) was an extension of TSUd_r1.1 where optical mapping was used to increase scaffold size and generated a draft genome covering 94% of the genome (512 Mb) of which 443 Mb were assembled into eight chromosomes (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>). Annotations for approximately 32,000 genes were imported from TSUd_r1.1. The assembly was used to compare two cultivars at the genome level (<xref ref-type="bibr" rid="B79">Yuan et&#xa0;al., 2018</xref>), but did not address the potential scaffold order issues in the original TSUd_r1.1 draft genome from which it was derived.</p>
<p>The most recent Ts draft genome (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>), TrSub3, was developed from the original TSUd_r1.1 assembly by ordering the scaffolds using Hi-C genomic proximity analysis (<xref ref-type="bibr" rid="B23">Dudchenko et&#xa0;al., 2018</xref>). The resulting draft encompassed 72% (403 Mb) of the genome with 393 Mb assembled into eight chromosomes. This assembly is not annotated and while there was improved alignment with the Mt genome, there was a note of caution that errors in the input assembly (TSUd_r1.1) could remain in the final genome (<xref ref-type="bibr" rid="B23">Dudchenko et&#xa0;al., 2018</xref>).</p>
<p>The current Ts assemblies described above indicate the need for a comprehensive reference genome that can be used as a resource for dissecting key agronomic traits, providing tools for genetic improvement of Ts as well as a platform for exploration of synteny with other <italic>Trifolium</italic> species. Syntenic analysis with other legumes, particularly within <italic>Trifolium</italic>, gives insight into evolutionary journeys among species. Availability of the Tr allotetraploid genome sequence and its diploid progenitor species genomes <italic>T. occidentale</italic> and <italic>T. pallescens</italic> (<xref ref-type="bibr" rid="B36">Griffiths et&#xa0;al., 2019</xref>) and assemblies of other <italic>Trifolium</italic> species provide a resource for a fresh investigation of synteny within the genus. Furthermore, a comprehensive Ts reference genome is a useful tool for investigating high resolution genomic diversity among Ts cultivars, an unexplored research topic. A Ts core collection (<xref ref-type="bibr" rid="B60">Nichols et&#xa0;al., 2013</xref>; <xref ref-type="bibr" rid="B1">Abdi et&#xa0;al., 2020</xref>) comprising a representative subset of the genetic diversity of this species complex is a valuable resource for this analysis. The currently accepted taxonomy of Ts recognizes three subspecies with adaptation to different environmental, soil and climatic conditions (<xref ref-type="bibr" rid="B32">Ghamkhar et&#xa0;al., 2015</xref>). Exploring germplasm and cultivar diversity can provide a resource to investigate genome-by-environment interaction. This will enable our understanding of species adaptation to different environments, and to future climate change scenarios.</p>
<p>In this study, we develop and provide an improved Ts reference genome (TSUd_r3.0) based on a comprehensive assembly, scaffolding with new Hi-C data, and further refinements using new long-read sequence data. We also provide comparative phylogenomics among five pasture legumes and <italic>Arabidopsis thaliana</italic> (At), and genome analysis of Ts with other members of Trifolieae including, for the first time, Tr. Lastly, we provide a genomic diversity assessment among 36 and within four Ts cultivars across the three subspecies. This is based on resequencing data to give insight into among and within Ts subspecific relationships and highlight potential issues for germplasm maintenance in germplasm centres.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<title>Materials and methods</title>
<sec id="s2_1">
<title>Repository</title>
<p>Configuration files for assembly and sequence quality tools that were used beyond default parameters have been placed at <uri xlink:href="https://github.com/Lanilen/Subclover_genome">https://github.com/Lanilen/Subclover_genome</uri>.</p>
</sec>
<sec id="s2_2">
<title>Whole genome sequencing</title>
<p>Following a strategy outlined in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>, Illumina single-end (SE), paired-end (PE) and mate-pair (MP) sequences obtained in the previous study (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>) were used in this study (PRJDB2012, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S1</bold>
</xref>). Additional long read sequences were also generated in this study (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Tables S1, S2</bold>
</xref>). High molecular weight cellular DNA of subterranean clover cultivar &#x2018;Daliak&#x2019; was extracted from young leaves with a Genomic-Tip (QIAGEN, Hilden, Germany) and used for construction of TruSeq synthetic long-read (TSLR) and PacBio single molecule real-time (SMRT) sequencing libraries. The TSLR library was constructed by a TruSeq synthetic long-read DNA library prep kit (Illumina, San Diego, CA) and sequences were generated by Illumina HiSeq2000 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Tables S1, S2</bold>
</xref>). The derived TSLR reads were assembled though the standard TruSPADES pipeline (<xref ref-type="bibr" rid="B5">Bankevich and Pevzner, 2016</xref>) pipeline. The SMRT library (PacBio, CA, USA) was sequenced using PacBio RS II platform.</p>
</sec>
<sec id="s2_3">
<title>Data quality control and trimming</title>
<p>Previously generated (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>) Illumina single-end, paired-end, and mate-pair data (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S1</bold>
</xref>; <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>) were assessed with the FastQC software package v0.10.1 (<uri xlink:href="http://www.bioinformatics.babraham.ac.uk/projects/fastqc/">http://www.bioinformatics.babraham.ac.uk/projects/fastqc/</uri>) (<xref ref-type="bibr" rid="B3">Andrews, 2010</xref>) to provide metrics of sequencing quality which informed next steps for data curation. Data were then trimmed to remove adapters. For data filtering we employed Skewer (<xref ref-type="bibr" rid="B41">Jiang et&#xa0;al., 2014</xref>), an Illumina-only read trimming and filtering tool. All Illumina short read datasets were filtered as follows: Discard reads with a mean quality lower than 20, trim ends to end quality of 30, discard reads shorter than 54bp.</p>
<p>Total and frequency of kmers (n=17) was counted in unassembled Illumina 180 bp insert paired-end sequence data for white clover and its progenitors using Jellyfish v2.2.0 (<xref ref-type="bibr" rid="B54">Mar&#xe7;ais and Kingsford, 2011</xref>) with default parameters. The data were plotted to determine maximum read depth for each specific 17-mer, and genome size was estimated as total 17-mer number divided by peak depth.</p>
<p>Kmer abundance graphs for all pair-end libraries were drawn with kmergenie v1.7051 (<xref ref-type="bibr" rid="B17">Chikhi and Medvedev, 2013</xref>; <xref ref-type="bibr" rid="B18">Crusoe et&#xa0;al., 2015</xref>) software package. After examination of the resulting graphs, two libraries were selected from the whole genome sequence (WGS) sets (DRX016491 and DRX028980) for assembly, as these two libraries showed the cleaner distribution of kmer abundances and provided sufficient coverage for assembly.</p>
<p>The khmer package (<xref ref-type="bibr" rid="B18">Crusoe et&#xa0;al., 2015</xref>) was then used for <italic>in silico</italic> digital normalization of WGS reads based on kmer abundance. We employed a different workflow than the ones recommended by the software authors. The general pipeline described by the authors removes high coverage kmers as well as low coverage kmers. This can lead to under-representation of repeat sequences in the final assembly. While de-Bruijn graph assemblers tend to collapse repeats in high coverage contigs, many of these repeats can be properly solved. Thus, khmer was used only to filter low-abundance kmer coverage reads to reduce noise. The normalize-by-median package was used to create a hash of 31 bp kmer abundances in the paired-end and single-end Illumina WGS libraries, and this hash was subsequently used with the filter-abund module to exclude reads with median kmer coverage of two or less. This adapted method allows for a reduction in the complexity of the graph assembly without reducing the representation of high coverage sequences, such as those from transposable elements, duplicated genomic regions, or closely related paralogs. All scripts and parameters used beyond default settings are located in a github repository described above.</p>
</sec>
<sec id="s2_4">
<title>Contig assembly</title>
<p>Assembly of paired-end data was done with two different graph-based assemblers. Initially, the best kmer was chosen based on the output from kmergenie. A kmer size of 45 was selected as it was long enough to contain a high number of unique kmers compared to shorter lengths, but still showed a clear separation between the haploid kmer coverage peak and the bottom sequencing error peak.</p>
<p>Initial assemblies were done using Velvet (<xref ref-type="bibr" rid="B80">Zerbino and Birney, 2008</xref>), employing both HiSeq and MiSeq paired-end libraries, as well as the Illumina TSLR data. Fixed insert lengths were an input for the paired-end libraries, and a low cutoff coverage of four was specified to discard low-coverage contigs based on the kmer abundance graph. Finally, the &#x2013;conserveLong option was employed to conserve all contigs containing an Illumina TSLR.</p>
<p>A second set of assemblies was done using Meraculous 2.2.4 (<xref ref-type="bibr" rid="B15">Chapman et&#xa0;al., 2011</xref>), using HiSeq and MiSeq paired-end data, as well as the shortest mate-pair Illumina library (2kb insert length). As the software does not process reads longer than 500 bp, none of the long-read libraries were used during assembly. The program was configured to run in Haploid mode, as there was no detectable heterozygosity in the kmer graph.</p>
<p>The addition of a single, short insert (2 kb) mate-pair library allowed for an increase of total assembly size during the gap filling step of the Meraculous pipeline, when compared to identical runs excluding all mate-pair reads. Assembly quality was assessed based on total assembly size, N50 statistics, as well as contiguity when aligning the assembled contigs with the single-read Illumina TSLR data, and whole genome alignment between the assemblies produced by the two tools. Results from both assemblers were similar, but Meraculous produced better contiguity.</p>
</sec>
<sec id="s2_5">
<title>Scaffolding</title>
<p>To improve the Meraculous assembly scaffolds, contigs were processed using a mix of Illumina mate-pair data and long-read sequencing data (PacBio Sequel, Illumina TSLRs).</p>
<p>Evaluation of mate-pair library quality was done by mapping all mate-pair reads to the assembled contigs using NextGenMap (<xref ref-type="bibr" rid="B64">Sedlazeck et&#xa0;al., 2013</xref>). Discarding non-unique aligning reads and with a minimum identity of 95%, this alignment was then used to estimate mate pair insert size, size distribution, and number of outliers. One library was discarded due to a large number of inconsistent read pairs, and the rest were deemed acceptable.</p>
<p>PacBio long reads were error-corrected with paired-end Illumina data using the <italic>proovread</italic> v2.14.0 (<uri xlink:href="https://github.com/BioInf-Wuerzburg/proovread">https://github.com/BioInf-Wuerzburg/proovread</uri>) software package (<xref ref-type="bibr" rid="B38">Hackl et&#xa0;al., 2014</xref>).</p>
<p>Illumina mate-pair, PacBio, and Illumina SLR data were collected and processed using the software package Opera-LG (<xref ref-type="bibr" rid="B30">Gao et&#xa0;al., 2016</xref>). For each mate-pair library, the expected number of connections (based on coverage), insert length, and insert length standard deviation (based on previously described estimates) were specified.</p>
</sec>
<sec id="s2_6">
<title>Gap filling</title>
<p>To improve assembly completeness, a two-step gap filling procedure was employed using both short and long reads. A first round of gap filling was done using Illumina paired-end short reads and the software GapFiller (<xref ref-type="bibr" rid="B10">Boetzer and Pirovano, 2012</xref>) using default parameters. Results were then run through Jelly2 and a second round of gap filling was done using PacBio long reads (<xref ref-type="bibr" rid="B28">English et&#xa0;al., 2012</xref>). Parameters were adjusted for the blasr step to tweak candidates and maximum score, but otherwise used the standard protocol (<uri xlink:href="https://sourceforge.net/p/pb-jelly/wiki/Home/">https://sourceforge.net/p/pb-jelly/wiki/Home/</uri>).</p>
</sec>
<sec id="s2_7">
<title>Error correction</title>
<p>To complete the assembly, scaffolds were corrected for errors at the nucleotide level with the use of an in-house workflow (<uri xlink:href="https://github.com/Lanilen/SemHelpers">https://github.com/Lanilen/SemHelpers</uri>). Illumina TSLRs were mapped against the assembly using LAST (<xref ref-type="bibr" rid="B44">Kie&#x142;basa et&#xa0;al., 2011</xref>), and a recursive approach was used to verify assembly bases from reads mapped with over 99% identity down to 90% identity from the resulting minor allele frequency alignment files. The recursive process was run to ensure that any single region of the reference genome underwent error-correction based on the highest scoring Illumina TSLR-genome high-scoring pair and did so only once.</p>
</sec>
<sec id="s2_8">
<title>Hi-C scaffolding</title>
<p>A Hi-C library was constructed from the young leaves of cv. Daliak as described (<xref ref-type="bibr" rid="B53">Lieberman-Aiden et&#xa0;al., 2009</xref>). Chromatin conformation capture data was generated using a Proximo Hi-C Plant Kit (Phase Genomics, Seattle, WA). Intact cells crosslinked by a formaldehyde solution were digested using the Sau3AI restriction enzyme. The resulting library was sequenced by Illumina NextSeq500 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>). The 105,525,717 paired-end (PE) reads were aligned to the previously assembled 27,474 scaffolds in TSUd_r1.1 (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>) by BWA (<xref ref-type="bibr" rid="B50">Li and Durbin, 2010</xref>). The read pairs with an unmapped mate were removed by SAMtools view using the -F 12 filtering (<xref ref-type="bibr" rid="B51">Li et&#xa0;al., 2009</xref>). Chromosome-scale scaffolds were created based on the 27,424 TSUd_r1.1 scaffolds by Proximo Hi-C genome scaffolding platform (Phase Genomics) in a method similar to that described (<xref ref-type="bibr" rid="B9">Bickhart et&#xa0;al., 2017</xref>). A linkage map of the genome (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>) was also used during scaffolding to prevent contigs originating from different linkage groups being placed on the same chromosome during the clustering process. Approximately 132,000 separate Proximo runs were performed to optimize the number of scaffolds and scaffold construction, in order to make the scaffolds as concordant with the observed Hi-C data as possible. Juicebox (<xref ref-type="bibr" rid="B61">Rao et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B24">Durand et&#xa0;al., 2016</xref>) was then used to correct scaffolding errors.</p>
</sec>
<sec id="s2_9">
<title>Gene prediction and annotation</title>
<p>To support gene modelling, transcript sequences were obtained from leaves, roots and seedlings of cv. Daliak (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>). cDNA libraries were constructed using a TruSeq RNA Library Prep Kit (Illumina), from which 301 nucleotide paired-end (PE) sequences were generated by Illumina MiSeq. PacBio full-length cDNA sequences (Iso-Seq) were also generated from seedlings of cv. Daliak. Quality control was performed using Rcorrector (<uri xlink:href="https://github.com/mourisl/Rcorrector">https://github.com/mourisl/Rcorrector</uri>) (<xref ref-type="bibr" rid="B71">Song and Florea, 2015</xref>), and Trim Galore! (<uri xlink:href="http://www.bioinformatics.babraham.ac.uk/projects/trim_galore/">http://www.bioinformatics.babraham.ac.uk/projects/trim_galore/</uri>). FastQC v0.10.1 (<xref ref-type="bibr" rid="B3">Andrews, 2010</xref>) was used to examine the quality metrics of reads.</p>
<p>RepeatModeler v1.0.11 (<xref ref-type="bibr" rid="B69">Smit and Hubley, 2008-2015</xref>) was used to generate a species-specific library of repetitive sequences. USEARCH v10.0.240 (<xref ref-type="bibr" rid="B26">Edgar, 2010</xref>) was used to align the library output to the UniProtKB (The UniProt Consortium 2018) (taxonomy Embryophyta) to predict protein existence at evidence level 1 (Experimental evidence at protein level), level 2 (Experimental evidence at transcript level) and level 3 (Protein inferred by homology). Any repeat sequence with a significant hit to UniProtKB was removed from the RepeatModeler library to minimize the number of repetitive protein sequences included in the library for repeat masking. RepeatMasker v4.0.7 (<xref ref-type="bibr" rid="B70">Smit et&#xa0;al., 2013-2015</xref>) was used to mask repetitive genomic sequence using the RepeatModeler library.</p>
<p>Gene prediction was performed on the masked genome as described in the gene prediction and functional annotation flowchart (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref>). Evidence-based gene prediction was performed with transcriptome data (RNA-Seq from MiSeq; Iso-Seq sequences) and legume peptide sequences. The RNA-Seq reads were assembled by tissue (seedling root and leaf) using a genome-guided implementation of Trinity v2.4.0 (<xref ref-type="bibr" rid="B37">Haas et&#xa0;al., 2013</xref>) after alignment to the masked subterranean clover pseudomolecules using STAR (<xref ref-type="bibr" rid="B22">Dobin et&#xa0;al., 2012</xref>). PacBio SMRT Link was used to generate high-quality (HQ) Iso-Seq reads. The known peptide sequences were obtained from the Legume Information System (<uri xlink:href="https://legumeinfo.org/">https://legumeinfo.org/</uri>) for the following thirteen species: <italic>Medicago truncatula</italic>, <italic>Arachis duranensis</italic>, <italic>Arachis hypogaea</italic>, <italic>Arachis ipaensis</italic>, <italic>Cajanus cajan</italic>, <italic>Cicer arietinum</italic>, <italic>Glycine max</italic>, <italic>Lotus japonicus</italic>, <italic>Lupinus angustifolius</italic>, <italic>Phaseolus vulgaris</italic>, <italic>Trifolium pratense</italic>, <italic>Vigna angularis</italic>, and <italic>Vigna radiata</italic>. The Trinity-assembled RNA-Seq sequences, HQ Iso-Seq sequences and the peptide sequences of these 13 legumes were then aligned independently with the assembled subterranean clover genome by Exonerate (<xref ref-type="bibr" rid="B68">Slater and Birney, 2005</xref>) for homology-based gene prediction using the following parameters: &#x2013;model protein2genome &#x2013;showtargetgff TRUE &#x2013;softmaskquery yes &#x2013;query target_prot.fasta &#x2013;target target_dna.fasta &#x2013;score 200. Meanwhile, Meanwhile, <italic>ab initio</italic> gene prediction was performed by Augustus (<xref ref-type="bibr" rid="B72">Stanke and Morgenstern, 2005</xref>) using <italic>Arabidopsis thaliana</italic> as the training set. In order to remove pseudogenes, the quality of the gene sequences predicted from RNA-Seq, Iso-Seq, the legume peptide sequences and the Augustus-derived <italic>ab initio</italic> predictions was then assessed based on the alignment to an in-house database incorporating three major databases (OrthoDB, UniProtKB and RefSeq from Viridiplantae) using Hayai-annotation with a parameter of 40% sequence (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref>). For the first validation step (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref>), the filtered gene sequences were aligned to the assembled subterranean clover genome sequence again using STARlong (<xref ref-type="bibr" rid="B22">Dobin et&#xa0;al., 2012</xref>). When multiple gene sequences predicted by different approaches were aligned on the same region, the longest sequences were selected, and after this second validation step (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref>) the final set of gene sequences was created.</p>
<p>Hayai-Annotation Plants v1.0.1 (<xref ref-type="bibr" rid="B34">Ghelfi et&#xa0;al., 2019</xref>) using the UniProtKB/Embriophyta database was used for gene annotation of TSUd_r3.0, TSUd_r1.1, <italic>A. thaliana</italic> (Araport11, <uri xlink:href="https://www.araport.org/data/araport11/">https://www.araport.org/data/araport11/</uri>), <italic>L. japonicus</italic> (Lj3.0, <uri xlink:href="http://www.kazusa.or.jp/lotus/index.html">http://www.kazusa.or.jp/lotus/index.html</uri> (<xref ref-type="bibr" rid="B63">Sato et&#xa0;al., 2008</xref>), <italic>M. truncatula</italic> (Mt4.0v2, <uri xlink:href="http://www.medicagogenome.org/">http://www.medicagogenome.org/</uri>), and <italic>T. pratense</italic> TGACv2 (<xref ref-type="bibr" rid="B21">De Vega et&#xa0;al., 2015</xref>). The output tables of Hayai-Annotation Plants, GO_BP_table.csv, GO_MF_table.csv and EC_table.csv, for each species were used to generate the graphic comparing all four legume species, the former version of <italic>T. subterraneum</italic> (TSUd_r1.1), and <italic>A. thaliana</italic>. The graphics were performed using Rscript. Note that a recent high-quality assembly of the <italic>T. pratense</italic> genome (ARS_RCv1.1; <xref ref-type="bibr" rid="B8">Bickhart et&#xa0;al., 2022</xref>) was not used in this analysis as it has yet to be annotated.</p>
</sec>
<sec id="s2_10">
<title>Macrosyntenic analysis</title>
<p>The predicted protein-coding gene sequences were clustered with those of <italic>A. thaliana</italic> (Araport11), <italic>L. japonicus</italic> (Lj3.0), <italic>M. truncatula</italic> (Mt4.0v2), <italic>T. pratense</italic> (TGACv2) and <italic>T. repens</italic> (<xref ref-type="bibr" rid="B36">Griffiths et&#xa0;al., 2019</xref>) using the CD-HIT program (<xref ref-type="bibr" rid="B29">Fu et&#xa0;al., 2012</xref>). A range of cluster sequence identity (c) and alignment length (aL) parameters were assessed. Parameters c = 0.6 and aL = 0.5 were selected to optimise clustering, and all other parameters were default values.</p>
<p>Macrosynteny between <italic>T. subterraneum</italic>, and <italic>T. pratense</italic>, <italic>T. repens</italic>, <italic>M. truncatula</italic>, and <italic>L. japonicas</italic> was investigated based on homologous translated protein sequences using BLAST searches with an E-value cut-off of 1E-100. For this macrosyntenic analysis, a recent high-quality <italic>T. pratense</italic> assembly (ARS_RCv1.1; <xref ref-type="bibr" rid="B8">Bickhart et&#xa0;al., 2022</xref>) was used. As <italic>T. repens</italic> is an allotetraploid, the two subgenomes were merged into single consensus homoeologous groups, hence eight chromosomes rather than 16. Synteny plots were drawn using the gnuplot program (<uri xlink:href="http://www.gnuplot.info">http://www.gnuplot.info</uri>). To create the Circos diagrams, synteny blocks were constructed by whole genome alignment using LASTZ (<xref ref-type="bibr" rid="B39">Harris, 2007</xref>) and comprised matches &gt;33 kb in length. These alignments were then filtered to choose best-matching regions based on maximizing total aligned base pairs from ascending/descending sorted matches. Plots were then generated using Circa (<uri xlink:href="http://omgenomics.com/circa">http://omgenomics.com/circa</uri>) and blocks within 100 kb windows were merged and represented by a single line.</p>
</sec>
<sec id="s2_11">
<title>Molecular phylogeny</title>
<p>The translated protein sequences of TSUd_r3.0 were compared by OrthoMCL (<xref ref-type="bibr" rid="B52">Li et&#xa0;al., 2003</xref>) with those of <italic>A. thaliana</italic> (Araport11), <italic>L. japonicus</italic> (Lj3.0), <italic>M. truncatula</italic> (Mt4.0v2), <italic>T. pratense</italic> (TGACv2) and <italic>T. repens</italic> (<xref ref-type="bibr" rid="B36">Griffiths et&#xa0;al., 2019</xref>) with the parameters c = 0.6 and aL = 0.9. The single copy genes in each cluster commonly conserved among the seven species were applied to multiple alignment by MUSCLE 3.8.31 (<xref ref-type="bibr" rid="B25">Edgar, 2004</xref>). The indels in the aligned sequences of the single copy genes were eliminated by Gblocks 0.91b (<xref ref-type="bibr" rid="B14">Castresana, 2000</xref>). The sequences of conserved blocks in the single copy genes were concatenated for each species and used for construction of phylogenetic tree by Maximum-Likelihood algorithm using MEGA X 10.0.5 (<xref ref-type="bibr" rid="B47">Kumar et&#xa0;al., 2018</xref>) with the Jones-Taylor-Thornton substitution model. In this step, <italic>A. thaliana</italic> was selected as outgroup. According to the TIMETREE (<uri xlink:href="http://www.timetree.org">http://www.timetree.org</uri>), the divergence time between <italic>L. japonicus</italic> and <italic>M. truncatula</italic> is estimated as 59 MYA, and this value was used for the calibration.</p>
</sec>
<sec id="s2_12">
<title>Cultivar resequencing</title>
<p>A total of 35 Ts cultivars in addition to Daliak were used for whole-genome diversity analysis (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>) using resequencing of a pooled sample of at least 10 individuals per accession. The materials were obtained from the Margot Forde Germplasm Centre in New Zealand. Of the 35 cultivars, 32 were resequenced using one accession whereas cvs &#x2018;Denmark&#x2019;, &#x2018;Leura&#x2019;, and &#x2018;Coolamon&#x2019; had two different accessions each. In total, 38 Ts accessions in addition to Daliak were resequenced. Paired-end reads were generated by Illumina HiSeq2000 (101 nt) or HiSeqX (151 nt) with coverage depth ranging from 12.4 - 41.7&#xd7; based on an estimated genome size of 552.4 Mb (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>). The reads were mapped onto the assembled <italic>T. subterraneum</italic> genome (TSUd_r3.0) by Bowtie2 (<xref ref-type="bibr" rid="B49">Langmead and Salzberg, 2012</xref>) and variant calls were performed by SAMtools v0.1.19 (<xref ref-type="bibr" rid="B51">Li et&#xa0;al., 2009</xref>) and VarScan 2.3 (<xref ref-type="bibr" rid="B45">Koboldt et&#xa0;al., 2012</xref>) as summarized in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S3</bold>
</xref>. Population structure of the resequenced lines was investigated by Admixture 1.3.0 (<xref ref-type="bibr" rid="B2">Alexander et&#xa0;al., 2009</xref>). Principal component analysis (PCA) was performed by using TASSEL 5 (<xref ref-type="bibr" rid="B12">Bradbury et&#xa0;al., 2007</xref>). Nucleotide diversity (Pi) was calculated on a per-site basis with a 500 kb sliding window using VCFtools (<xref ref-type="bibr" rid="B19">Danecek et&#xa0;al., 2011</xref>), whereas Copy Number Variations (CNV), calculated per cultivar relative to Daliak, was determined using CNV-Seq (<xref ref-type="bibr" rid="B77">Xie and Tammi, 2009</xref>) with a 500 kb window.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<title>Results</title>
<sec id="s3_1">
<title>TSUd_3.0 - an improved subterranean clover genome</title>
<p>In this study, a new subterranean clover (<italic>Trifolium subterraneum</italic>; Ts) genome assembly was generated by <italic>de novo</italic> assembly of existing short read data with inclusion of PacBio-derived long-read and TruSeq Synthetic Long-Read (TSLR) sequence data as detailed in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>. This assembly was enhanced by high-throughput chromosome conformation capture (Hi-C) analysis. The Hi-C analysis generated 105,525,717 paired-end 80 bp reads (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>) and formed 8,927 scaffolds which clustered into eight chromosome-scale scaffolds with a combined length of 441.9 Mb. This represents 93.7% of the total length of the prior assembly TSUd_r1.1 (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>). Alignment between the Hi-C-derived assembly and TSUd_r1.1 revealed inconsistencies (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S2A</bold>
</xref>).</p>
<p>Our initial assembly was augmented by the addition of 895,541 TSLR (5&#xd7; coverage) and 477,975 PacBio Sequel (10&#xd7; coverage) reads with mean lengths of 3,225 bp and 10,671 bp, respectively (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Tables S1, S2</bold>
</xref>). This combined sequence resource provided 331&#xd7; coverage. Integrating the TSLR data with reassembled pre-existing Illumina short-read data from <xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al. (2016)</xref> generated contigs that were scaffolded using the PacBio data resulting in eight pseudomolecules after alignment with both the Hi-C clustering, described above, and a genetic linkage map (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>) (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures S2B, C</bold>
</xref>). This improved assembly encompassed 531 Mb accounting for 97.8% of the estimated genome size (<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). A total of 80% of this was ordered into pseudomolecules with an average length of 54.4 Mb. Whole genome shotgun Illumina sequence reads representing 57&#xd7; depth were mapped back to the assembly to assess genome coverage. A mapping rate of 97.7% for these reads further indicated the assemblies encompass a high proportion of the genome. The final assembly was designated TSUd_r3.0.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Genome assembly statistics for the TSUd_r3.0 assembly of subterranean clover (<italic>Trifolium subterranean</italic>) compared with previous drafts.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="bottom" align="left"/>
<th valign="bottom" align="center">TSUd_r1.1<sup>1</sup>
</th>
<th valign="bottom" align="center">Tsub_Refv2.0<sup>2</sup>
</th>
<th valign="bottom" align="center">TSUd_r3.0<sup>3</sup>
</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="bottom" colspan="4" align="left">Estimated Genome Size</th>
</tr>
<tr>
<td valign="bottom" align="left">
<italic>K</italic>-mer (<italic>K</italic> = 17)<sup>4</sup>
</td>
<td valign="bottom" align="center">552,423,008</td>
<td valign="bottom" align="center">552,423,008</td>
<td valign="bottom" align="center">552,423,008</td>
</tr>
<tr>
<td valign="bottom" align="left">1C Genome size<sup>5</sup>
</td>
<td valign="bottom" align="center">544,000,000</td>
<td valign="bottom" align="center">544,000,000</td>
<td valign="bottom" align="center">544,000,000</td>
</tr>
<tr>
<td valign="bottom" align="left">Total Assembly Length (including N)</td>
<td valign="bottom" align="center">471,834,188</td>
<td valign="bottom" align="center">512,439,000</td>
<td valign="bottom" align="center">531,039,069</td>
</tr>
<tr>
<td valign="bottom" align="left">Estimated Genome Coverage</td>
<td valign="bottom" align="center">86.7%</td>
<td valign="bottom" align="center">94.2%</td>
<td valign="bottom" align="center">97.6%</td>
</tr>
<tr>
<th valign="bottom" colspan="4" align="left">Assembly Metrics</th>
</tr>
<tr>
<td valign="bottom" align="left">Ps 1-8 (<bold>
<italic>scaffolds in Ps-0</italic>
</bold>)</td>
<td valign="bottom" align="center">8 (<bold>
<italic>27,416</italic>
</bold>)</td>
<td valign="bottom" align="center">1,545</td>
<td valign="bottom" align="center">8 (<bold>
<italic>13,109</italic>
</bold>)</td>
</tr>
<tr>
<td valign="bottom" align="left">
<bold>
<italic>Total Ps + scaffolds</italic>
</bold>
</td>
<td valign="bottom" align="center">
<bold>
<italic>27,424</italic>
</bold>
</td>
<td valign="bottom" align="center">
<bold>
<italic>1,545</italic>
</bold>
</td>
<td valign="bottom" align="center">
<bold>
<italic>13,117</italic>
</bold>
</td>
</tr>
<tr>
<td valign="bottom" align="left">%genome in pseudomolecules (excl Ps-0)</td>
<td valign="bottom" align="center">73.9%</td>
<td valign="bottom" align="center">80.2%</td>
<td valign="bottom" align="center">80.1%</td>
</tr>
<tr>
<td valign="bottom" align="left">Scaffold N<sub>50</sub> (bp)</td>
<td valign="bottom" align="center">47,721,588</td>
<td valign="bottom" align="center">410,493</td>
<td valign="bottom" align="center">56,229,069</td>
</tr>
<tr>
<th valign="bottom" colspan="4" align="left">Pseudomolecule Metrics</th>
</tr>
<tr>
<td valign="bottom" align="left">Ps-1 length (bp)</td>
<td valign="bottom" align="center">47,645,759</td>
<td valign="bottom" align="center">49,039,259</td>
<td valign="bottom" align="center">54,345,684</td>
</tr>
<tr>
<td valign="bottom" align="left">Ps-2 length (bp)</td>
<td valign="bottom" align="center">63,731,624</td>
<td valign="bottom" align="center">67,952,282</td>
<td valign="bottom" align="center">60,869,117</td>
</tr>
<tr>
<td valign="bottom" align="left">Ps-3 length (bp)</td>
<td valign="bottom" align="center">44,866,005</td>
<td valign="bottom" align="center">53,122,998</td>
<td valign="bottom" align="center">49,046,471</td>
</tr>
<tr>
<td valign="bottom" align="left">Ps-4 length (bp)</td>
<td valign="bottom" align="center">56,437,177</td>
<td valign="bottom" align="center">55,565,095</td>
<td valign="bottom" align="center">56,229,069</td>
</tr>
<tr>
<td valign="bottom" align="left">Ps-5 length (bp)</td>
<td valign="bottom" align="center">47,721,588</td>
<td valign="bottom" align="center">56,909,348</td>
<td valign="bottom" align="center">52,082,161</td>
</tr>
<tr>
<td valign="bottom" align="left">Ps-6 length (bp)</td>
<td valign="bottom" align="center">49,553,705</td>
<td valign="bottom" align="center">53,633,078</td>
<td valign="bottom" align="center">53,133,211</td>
</tr>
<tr>
<td valign="bottom" align="left">Ps-7 length (bp)</td>
<td valign="bottom" align="center">42,658,284</td>
<td valign="bottom" align="center">46,363,611</td>
<td valign="bottom" align="center">50,781,203</td>
</tr>
<tr>
<td valign="bottom" align="left">Ps-8 length (bp)</td>
<td valign="bottom" align="center">48,533,994</td>
<td valign="bottom" align="center">60,596,234</td>
<td valign="bottom" align="center">58,605,038</td>
</tr>
<tr>
<td valign="bottom" align="left">
<bold>
<italic>Total length Ps 1-8 (bp)</italic>
</bold>
</td>
<td valign="bottom" align="center">
<italic>401,148,136</italic>
</td>
<td valign="bottom" align="center">
<italic>443,181,905</italic>
</td>
<td valign="bottom" align="center">
<italic>435,091,954</italic>
</td>
</tr>
<tr>
<td valign="bottom" align="left">
<italic>GC (GATC) content (%);</italic> <bold>
<italic>Gap Ratio</italic>
</bold> <italic>(%)</italic>
</td>
<td valign="bottom" align="center">
<italic>33.0;</italic> <bold>
<italic>13.7</italic>
</bold>
</td>
<td valign="bottom" align="center">
<italic>33.3;</italic> <bold>
<italic>11.3</italic>
</bold>
</td>
<td valign="bottom" align="center">
<italic>33.3;</italic> <bold>
<italic>7.1</italic>
</bold>
</td>
</tr>
<tr>
<td valign="bottom" align="left">Ps-0 length (bp)</td>
<td valign="bottom" align="center">70,686,052</td>
<td valign="bottom" align="center">69,257,095</td>
<td valign="bottom" align="center">95,947,115</td>
</tr>
<tr>
<td valign="bottom" align="left">
<italic>&#x2003;Ps-0 scaffolds</italic>
</td>
<td valign="bottom" align="center">
<italic>27,416</italic>
</td>
<td valign="bottom" align="center">&#x2013;</td>
<td valign="bottom" align="center">
<italic>13,109</italic>
</td>
</tr>
<tr>
<td valign="bottom" align="left">
<italic>&#x2003;Average length Ps-0 scaffolds (bp)</italic>
</td>
<td valign="bottom" align="center">&#x2013;</td>
<td valign="bottom" align="center">&#x2013;</td>
<td valign="bottom" align="center">
<italic>6,319</italic>
</td>
</tr>
<tr>
<td valign="bottom" align="left">
<italic>&#x2003;Ps-0 scaffold max length (bp)</italic>
</td>
<td valign="bottom" align="center">&#x2013;</td>
<td valign="bottom" align="center">&#x2013;</td>
<td valign="bottom" align="center">
<italic>3,018,486</italic>
</td>
</tr>
<tr>
<td valign="bottom" align="left">
<italic>&#x2003;Ps-0 scaffold min length (bp)</italic>
</td>
<td valign="bottom" align="center">
<italic>300</italic>
</td>
<td valign="bottom" align="center">
<italic>300</italic>
</td>
<td valign="bottom" align="center">
<italic>500</italic>
</td>
</tr>
<tr>
<td valign="bottom" align="left">
<italic>&#x2003;Ps-0 scaffold N<sub>50</sub> length</italic>
</td>
<td valign="bottom" align="center">&#x2013;</td>
<td valign="bottom" align="center">&#x2013;</td>
<td valign="bottom" align="center">
<italic>259,414</italic>
</td>
</tr>
<tr>
<td valign="bottom" align="left">
<italic>GC (GATC) content (%);</italic> <bold>
<italic>Gap Ratio</italic>
</bold> <italic>(%)</italic>
</td>
<td valign="bottom" align="center">
<italic>34.8;</italic> <bold>
<italic>22.6</italic>
</bold>
</td>
<td valign="bottom" align="center">&#x2013;</td>
<td valign="bottom" align="center">
<italic>34.2;</italic> <bold>
<italic>12.8</italic>
</bold>
</td>
</tr>
<tr>
<td valign="bottom" align="left">
<bold>
<italic>Total length Ps 1-8 + 0</italic>
</bold>
</td>
<td valign="bottom" align="center">
<bold>
<italic>471,834,188</italic>
</bold>
</td>
<td valign="bottom" align="center">
<bold>
<italic>512,439,000</italic>
</bold>
</td>
<td valign="bottom" align="center">
<bold>
<italic>531,039,069</italic>
</bold>
</td>
</tr>
<tr>
<th valign="bottom" colspan="4" align="left">Annotation Summary</th>
</tr>
<tr>
<td valign="bottom" align="left">Number of predicted genes</td>
<td valign="bottom" align="center">42,706</td>
<td valign="bottom" align="center">32,333</td>
<td valign="bottom" align="center">41,979</td>
</tr>
<tr>
<td valign="bottom" align="left">Total length of predicted genes (bp)</td>
<td valign="bottom" align="center">47,965,017</td>
<td valign="bottom" align="center">34,758,167</td>
<td valign="bottom" align="center">55,177,719</td>
</tr>
<tr>
<td valign="bottom" align="left">Mean length of predicted genes (bp)</td>
<td valign="bottom" align="center">1,123</td>
<td valign="bottom" align="center">1,075</td>
<td valign="bottom" align="center">1,314</td>
</tr>
<tr>
<td valign="bottom" align="left">Length of genes (bp): Max; <bold>
<italic>Min</italic>
</bold>
</td>
<td valign="bottom" align="center">15,417; <bold>
<italic>150</italic>
</bold>
</td>
<td valign="bottom" align="center">15,309; <bold>
<italic>201</italic>
</bold>
</td>
<td valign="bottom" align="center">15,309; <bold>
<italic>90</italic>
</bold>
</td>
</tr>
<tr>
<td valign="bottom" align="left">N<sub>50</sub> of predicted genes (bp)</td>
<td valign="bottom" align="center">1,548</td>
<td valign="bottom" align="center">1,437</td>
<td valign="bottom" align="center">1,767</td>
</tr>
<tr>
<td valign="bottom" align="left">Proportion genes &#x2265; 1 kb</td>
<td valign="bottom" align="center">42.1%</td>
<td valign="bottom" align="center">&#x2013;</td>
<td valign="bottom" align="center">53.7%</td>
</tr>
<tr>
<th valign="bottom" colspan="4" align="left">BUSCO<sup>6</sup> Scores of Assembly (based on 1,440 reference genes)</th>
</tr>
<tr>
<td valign="bottom" align="left">Complete genes - total</td>
<td valign="bottom" align="center">1,108 (77.0%)</td>
<td valign="bottom" align="center">1,261 (87.6%)</td>
<td valign="bottom" align="center">1,358 (94.4%)</td>
</tr>
<tr>
<td valign="bottom" align="left">
<italic>&#x2003;Complete genes - single copy</italic>
</td>
<td valign="bottom" align="center">
<italic>983</italic> (<italic>68.3%</italic>)</td>
<td valign="bottom" align="center">
<italic>1111</italic> (<italic>77.2%</italic>)</td>
<td valign="bottom" align="center">
<italic>1,255</italic> (<italic>87.2%</italic>)</td>
</tr>
<tr>
<td valign="bottom" align="left">
<italic>&#x2003;Complete genes - duplicated</italic>
</td>
<td valign="bottom" align="center">
<italic>125</italic> (<italic>8.7%</italic>)</td>
<td valign="bottom" align="center">
<italic>150</italic> (<italic>10.4%</italic>)</td>
<td valign="bottom" align="center">
<italic>103</italic> (<italic>7.2%</italic>)</td>
</tr>
<tr>
<td valign="bottom" align="left">Fragmented genes</td>
<td valign="bottom" align="center">138 (9.6%)</td>
<td valign="bottom" align="center">59 (4.1%)</td>
<td valign="bottom" align="center">25 (<italic>1.7%</italic>)</td>
</tr>
<tr>
<td valign="bottom" align="left">Missing genes</td>
<td valign="bottom" align="center">194 (3.9%)</td>
<td valign="bottom" align="center">120 (8.3%)</td>
<td valign="bottom" align="center">57 (<italic>3.9%</italic>)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Ps, pseudomolecule; bp, base pairs; <sup>1</sup>
<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>; <sup>2</sup>
<xref ref-type="bibr" rid="B42">Kaur et&#xa0;al., 2017a</xref>; <sup>3</sup>This study.</p>
</fn>
<fn>
<p>
<sup>4</sup>These assemblies are based on the same Illumina paired-end sequence that was used to derive the K-mer estimate of genome size.</p>
</fn>
<fn>
<p>
<sup>5</sup>Flow cell cytometry (<xref ref-type="bibr" rid="B6">Bennett and Leitch, 2011</xref>) or 1C = 0.55 pg DNA; (<xref ref-type="bibr" rid="B74">Vi&#x17e;intin et&#xa0;al., 2006</xref>).</p>
</fn>
<fn>
<p>
<sup>6</sup>Benchmarking Universal Single-Copy Orthologues (BUSCO) assessment (<xref ref-type="bibr" rid="B65">Sim&#xe3;o et&#xa0;al., 2015</xref>).</p>
</fn>
<fn>
<p>Predicted genes excludes introns; length of genes is based on cDNA.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_2">
<title>Pseudomolecule scaffold order enhanced compared to previous assemblies</title>
<p>To assess relative integrity of the assembled pseudomolecules, the TSUd_r3.0 assembly was aligned with TSUd_r1.1 (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>). Significant inversions and putative assembly duplications were identified on six of the eight pseudomolecules (Chr2, Chr3, Chr4, Chr5 Chr7 and Chr8; <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>). As the genome assemblies were constructed from individuals of the same cultivar (&#x2018;Daliak&#x2019;), these structural differences suggested assembly discrepancies. The veracity of the TSUd_r3.0 assembly was investigated further using graphical genotypes of the 155 F2-derived F4 bulked DNA used for linkage map construction (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>). Scaffolds on Chr2 were selected based on location discrepancies when comparing the TSUd_r3.0 and TSUd_r1.1 assemblies (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3A</bold>
</xref>). A scaffold at the inflexion point in the Chr2 assembly alignment (light pink box) and two scaffolds that were adjacent in assembly TSUd_r3.0 but in different chromosome arms in the original assembly (TSUd_r1.1) (light blue and light green boxes), were analyzed for graphical genotypes.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>A matrix plot based on NUCmer-derived (<xref ref-type="bibr" rid="B48">Kurtz et&#xa0;al., 2004</xref>) alignment showing synteny between chromosome sequences of the original [TSUd_r1.1; (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>)] and current (TSUd_r3.0) subterranean clover assemblies.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1103857-g002.tif"/>
</fig>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Validation of TSUd_r3.0 Chr2 assembly relative to TSUd_r1.1 using graphical genotype data from 155 F2 individuals. <bold>(A)</bold> Corresponding positions in megabase pairs (Mb) in chromosome 2 (Chr2) of scaffold sequences common to assemblies TSUd_r3.0 and TSUd_r1.1. The light blue and light green boxes identify scaffolds that are in close proximity in assembly TSUd_r3.0 but located at either end of Chr2 in assembly TSUd_r1.1. The light pink box denotes a Chr2 scaffold at the inflexion point of alignment of the two assemblies. <bold>(B)</bold> Graphical genotypes of the 155 F<sub>2</sub> individuals (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>) of five SNPs located on each of three Chr2 scaffolds Tsud_sc00312.00 (light blue), Tsud_sc00006.10 (pink), and Tsud_sc00816.00 (light green). In the TSUd_r1.1 assembly, these scaffolds are positioned at approximately 0.8 Mb, 22.5 Mb, and 50.1 Mb, respectively. In the TSUd_r3.0 assembly, these scaffolds are positioned at approximately 48 Mb, 0 Mb and 48 Mb, respectively. The genotype at each locus for each of the 155 individuals is represented by a colored vertical bar. Red, yellow, blue and white bars represent homozygous reference allele, heterozygous (reference and alternate) and homozygous alternate allele, and missing data, respectively. SNP 01 &#x2013; 05 = scaffold Tsud_sc00312.00 SNPs 10311, 15589, 164181, 164286 and 166200, respectively. SNP 06 &#x2013; 10 = scaffold Tsud_sc00816.00 SNPs 136492, 144450, 28682, 29782, and 46273, respectively. SNP 11 &#x2013; 15 = scaffold Tsud_sc0006.10 SNPs 126296, 155381, 164679, 180487, and 182924, respectively.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1103857-g003.tif"/>
</fig>
<p>Scaffolds in closer genomic proximity exhibit consistency of graphical genotypes for SNP loci within individuals due to the reduced likelihood of recombination. The two scaffolds identified by the light blue and green boxes in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref> showed similar graphical genotypes for homozygosity and heterozygosity of reference and alternate alleles for each of five SNP loci per scaffold in each of the 155 genotyped individuals. This indicated these two scaffolds comprised DNA likely in close genomic proximity, supporting the placement in the current TSUd_r3.0 assembly. As expected, there was little conservation of graphical genotypes between these two scaffolds and the one identified by the light pink box. The latter was separated by ~48 Mb in assembly TSUd_r3.0, and ~20 to 30 Mb in the TSUd_r1.1 assembly (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>).</p>
<p>In summary, the analysis indicated scaffolds Tsud_sc00312.00 (light blue) and Tsud_sc00816.00 (light green) are in close genomic proximity, and in assembly TSUd&#xad;r_3.0, these two scaffold sequences were adjacent to each other at positions 47.0 to 47.6 Mb and 47.6 to 47.7 Mb, respectively. All other points of discrepancy with the previous assemblies were investigated in a similar way and the pseudomolecule structure of TSUd_r3.0 was corroborated in each case by genotype data.</p>
</sec>
<sec id="s3_3">
<title>Comprehensive gene prediction and annotation</title>
<p>A total of 41,979 protein-coding gene sequences were predicted and annotated in assembly TSUd_r3.0 (<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>), a number close to TSUd_r1.1 (42,706) and more than Tsub_Refv2.0 (32,333). Based on source data, the distribution of the best models showed that of the total predicted genes, 43% (18,051 sequences) were based on transcriptome sequence (RNA-Seq and Iso-Seq; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>) assemblies, 30% (12,594) were homology-based supported by 13 legume species, and the remaining 27% (11,334<italic>) ab initio</italic> predictions. It is important to note that genes predicted by Augustus (<italic>ab initio</italic> prediction software) were often corroborated by the transcriptome data, however the gene prediction and annotation workflow (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref>) prioritized the longest genes in the final validation step. Therefore, the 27% <italic>ab initio</italic> proportion of predicted gene models is a likely overestimate.</p>
<p>On average, 99.8% of the MiSeq raw reads (leaf, root, and seedling) and 82.0% of IsoSeq data (seedling) mapped to the assembly. Even though there was a high mapping rate, the transcriptome sequences accounted for approximately 43% of predicted genes. Further evidence of assembly quality and gene set completeness of TSUd_r3.0 was the marked improvement in BUSCO scores compared with previous assembly iterations (<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>).</p>
<p>Functional gene annotation was performed using Hayai-annotation Plants (<xref ref-type="bibr" rid="B34">Ghelfi et&#xa0;al., 2019</xref>), and assembly quality further assessed by analysis of predicted gene function by assignment of Gene Ontology (GO) terms. Assembly TSUd_r3.0 had a greater proportion of predicted genes in most of the categories (~75%) within the Biological Process and Molecular Function domains, than the TSUd_r1.1 assembly (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures S3A, B</bold>
</xref>). Assigning Enzyme Commission (EC) numbers to the predicted genes revealed that transferases (2.x), hydrolases (3.x) with some oxidoreductases (1.x) and isomerases (5.x) were the most common (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;3C</bold>
</xref>). When comparing across a panel of species comprising white clover (<italic>Trifolium repens</italic>; Tr), red clover (<italic>T. pratense</italic>: Tp), <italic>Medicago truncatula</italic> (Mt), <italic>Lotus japonicus</italic> (Lj) and <italic>Arabidopsis thaliana</italic> (At), the Trifoleae species (Ts+Tp+Tr+Mt) had similar proportions of genes in each EC class except for 3.6.4.12 (DNA helicase) where Tr was under-represented, and Mt had a greater proportion than Ts. Among the legumes, Lj generally had a greater proportion of predicted genes in each EC class, and the model plant At had an even higher proportion in most EC classes (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S3C</bold>
</xref>).</p>
</sec>
<sec id="s3_4">
<title>Comparative analysis with other legume species shows fragmented alignment with red clover</title>
<p>Additional evidence of assembly improvement was gained from gene sequence comparisons. The set of the 41,979 predicted Ts protein-coding genes were clustered with predicted genes from a range of plant species comprising Tp, Tr, Mt, Lj and At and resulted in 34,929 Ts genes aligning with at least one other gene either in Ts or another species. Approximately 50% of the Ts genes clustered across all six species, 13% were shared within Family Leguminosae (Ts+Tp+Tr+Mt+Lj), 8% within Tribe Trifoliae (Ts+Tp+Tr+Mt) and 2% shared only among members of the genus <italic>Trifolium</italic> (Ts+Tp+Tr) (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S4</bold>
</xref>). While a group of 1,306 genes (4%) were shared only with the close relative Tp, and a smaller set of 799 genes (2%) only with the more distantly related Tr, less than a tenth (8%; 2,791 genes) were exclusive to Ts. Other combinations of groupings accounted for the remaining 13% of the number of clustered genes (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S4</bold>
</xref>). There was a similar pattern across these groupings when focussing on the number of gene clusters (total = 28,724 clusters) rather than clustered genes, although the percentages relative to the Ts cluster number were half those of the gene number comparisons (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S5</bold>
</xref>). Furthermore, the proportion of clusters exclusive to each species ranged from 506 (Ts) to 2,821 (Lj) with Ts having the lowest proportion of exclusive clusters (1.4%), similar to Tp (1.9%), and less than those of Tr (3.2%), Mt (4.1%), At (4.2%), and Lj (7.9%) (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S5</bold>
</xref>).</p>
<p>The enhanced Ts genome assembly has provided a platform for investigating phylogenetic and macrosyntenic analysis among legumes. Focusing on 723 conserved single-copy genes, and calibrating divergence time based on the separation of Lj and Mt, the data suggest that Lj (Robinoid clade) and the inverted repeat-lacking clade containing <italic>Medicago</italic> and <italic>Trifolium</italic> genera diverged 59 million years ago (MYA), whereas the ancestors of Tr diverged from Ts and its sister clade species 11.24 MYA. Red clover, meanwhile, diverged from Ts 9.23 MYA (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref>).</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Estimated phylogeny and genome alignments among Papilionoideae legumes. <bold>(A)</bold> a phylogenetic tree based on 723 common single copy genes showing divergence (millions of years ago (MYA)) of subterranean clover (<italic>Trifolium subterraneum</italic>; Ts), red clover (<italic>T. pratense</italic>; Tp), white clover (<italic>T. repens</italic>; Tr), <italic>Medicago truncatula</italic> (Mt), <italic>Lotus japonicus</italic> (Lj) from each other and <italic>Arabidopsis thaliana</italic> (At) as an outgroup. IRLC identifies the inverted repeat-lacking clade of the Papilionoideae subfamily of Leguminosae which contains Vicioids such as the <italic>Medicago</italic> and <italic>Trifolium</italic> genus. <bold>(B&#x2013;E)</bold>. Circos diagrams showing inter-chromosome relationships between the genomes of Ts and those of Tp, Tr, Mt and Lj, respectively. The ring represents pseudomolecules to scale within each circus figure. Colored lines represent synteny blocks constructed by whole genome alignment using the program LASTZ (<xref ref-type="bibr" rid="B39">Harris, 2007</xref>) comprising matches &gt;33 kb in length. Blocks within 100 kb windows were merged and represented by a single line. The Ts genome is delineated by the arc and comprises chromosomes Ts1 (54 megabases (Mb)), Ts2 (61 Mb), Ts3 (49 Mb), Ts4 (56 Mbp), Ts5 (52 Mbp), Ts6 (53 Mb), Ts7 (51 Mb), Ts8 (59 Mb). The Tr genome is presented as a merged consensus of the two subgenomes into single homoeologous groups, hence eight chromosomes rather than 16.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1103857-g004.tif"/>
</fig>
<p>Red clover (Tp), a close relative of Ts, has seven chromosome pairs rather than the eight typical of genus <italic>Trifolium</italic>. Our analysis using a recent high-quality Tp genome (ARS_RCv1.1; <xref ref-type="bibr" rid="B8">Bickhart et&#xa0;al., 2022</xref>) showed numerous rearrangements between Ts and Tp, with only Chr1 maintaining substantive synteny with Ts (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref>). Ts Chr2 showed synteny with Tp Chr3 and 7, and only half of Ts Chr3 aligned with Tp on Tp Chr3 and 7. Ts Chr4 aligned with portions of Tp Chr2, 4, 5 and 6, whereas Ts Chr 5 was syntenic with Tp Chr2 and 4. Ts Chr6 aligned with portions of Tp Chr3 and 4, whereas Ts Chr7 showed synteny with Tp Chr 3, 6 and 7. The greatest portion of Ts Chr 8 aligned with most of Tp Chr 5 and 7 (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref>). Despite the close relationship between Ts and Tp, there were large blocks with no synteny with Tp, for example Ts Chr3, 5, 6, 7 and 8 (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref>).</p>
<p>When comparing Ts with Tr, a more distant <italic>Trifolium</italic> relative than Tp (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref>), there was a lack of synteny with Ts/Tr Chr3, similar to what was seen with Ts/Tp Chr3 (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4C</bold>
</xref>). There were also gaps in synteny between Ts Chr4, 5, 6, and 8 and the Tr genome. Despite the phylogenetic distance, however, Ts and Tr Chr1 had substantial synteny, and Ts Chr3 and 5 aligned with Tr Chr 3 and 5, respectively. Ts Chr2 aligned with Tr Chr 4 and 6; Ts Chr4 aligned with Tr Chr 7 and 8. With Ts Chr6, the few syntenic regions aligned with Tr Chr4 and 6. Ts Chr7 aligned with Tr Chr3 and 7, whereas the small portions of Ts/Tr synteny in Ts Chr8 aligned with Tr Chr2 (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref>).</p>
<p>By contrast to the previous comparisons, Ts showed large regions of synteny across the genome with Mt, an even more distant relative than Tr (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4D</bold>
</xref>). As with the previous comparisons, Ts Chr1 aligns well with Mt Chr1, as does Ts/Mt Chr5. Ts Chr2 aligned with Mt Chr2, 4 and 8, whereas most of Ts Chr3 aligned with Mt3 except for a portion with synteny to Mt Chr6 (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4D</bold>
</xref>). Portions of Ts Chr4 align with Mt Chr4, 7 and 8. Most of Ts Chr6 aligned with Mt Chr6, although small portions showed synteny with Mt Chr2 and 4. Ts Chr7 aligned with Mt Chr7 as well as Mt Chr3 and a small section of Mt Chr2, whereas Ts Chr8 aligned with both Mt Chr2 and 8 (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4D</bold>
</xref>). The dotplot of genome alignment between Ts and Mt (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S6</bold>
</xref>) echoes the circos diagram and shows the improvement in co-linearity of the enhanced current assembly compared with TrSUd_r1.1 (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>). Alignment of Ts with the six chromosomes of the more distantly related Lj showed Ts Chr1, 2 and 5 exhibited syntenic conservation with Lj Chr5, 3 and 2, respectively, whereas Ts Chr4 was split between Lj Chr1 and 4 (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4E</bold>
</xref>).</p>
</sec>
<sec id="s3_5">
<title>Cultivar resequencing, heterozygosity and diversity analysis</title>
<p>To explore genetic diversity, heterozygosity and copy number variation in Ts, a pool of 10 individuals from each of 35 cultivars, in addition to Daliak, was resequenced on an Illumina platform with up to 47&#xd7; coverage. Three cultivars, Coolamon, Denmark and Leura, were each represented by two different accessions; therefore 38 pools were sequenced plus Daliak. Population structure based on 7,789,537 single nucleotide polymorphism (SNP) variants among the 39 pools identified four groups (K=4) (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S3</bold>
</xref>). Further analysis using the Neighbor-Joining (NJ) algorithm indicated that one group comprised two closely linked sub-groups (Groups 1a and 1b; <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>) that aligned well with known pedigree. Each group was comprised predominantly of a single subspecies with <italic>T. subterraneum</italic> ssp. <italic>subterraneum</italic> placed into Groups 1a, 1b and 2, whereas <italic>T. subterraneum</italic> ssp. <italic>yanninicum</italic> and <italic>T. subterraneum</italic> ssp. <italic>brachycalicinum</italic> formed Groups 3 and 4, respectively. There were only three examples (Group 1a &#x2013; Larisa; Group 3 &#x2013; Mt Barker; Group 4 Woogenellup) where a cultivar classified as a particular subspecies was placed in a group comprised of another subspecies (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>). The cultivars with multiple accessions showed dissimilarities between accessions that were greater than differences between two cultivars, such as Uniwager and Geraldton.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>A neighbor-joining phylogenetic tree and a population structure analyzed by Admixture (K=4) of the resequenced 39 individuals representing 36 subterranean clover cultivars, three of which had multiple accessions, including &#x2018;Daliak&#x2019; (Group 2). The AK references are identifiers of seed accessions stored in the Margot Forde Germplasm Center located at AgResearch, Palmerston North, New Zealand. The 36 cultivars were classified into the four groups and underwent PCA analysis (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S7</bold>
</xref>). Ts-s = <italic>Trifolium subterraneum</italic> ssp. <italic>subterraneum</italic>; Ts-b = <italic>T. subterraneum</italic> ssp. <italic>brachycalicinum</italic>; Ts-y = <italic>T. subterraneum</italic> ssp. <italic>yanninicum</italic>. These identifiers indicate that all individuals within the Group have been classified taxonomically as that subspecies. Exceptions are identified individually within the Group.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1103857-g005.tif"/>
</fig>
<p>Principal component (PC) analysis of groups identified in the population structure analysis revealed clear differentiation by PC1 and PC2 separating Groups 3 and 4 from Groups 1 and 2 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S7</bold>
</xref>). This supports the results from the NJ tree and structure analysis. Group 2 separated from Group 1 with PC3 whereas the Group 1 sub-groups resolved with PC4 and PC5.</p>
<p>SNP variant distribution among the 39 pools was analyzed across the genome and showed a general pattern of reduction towards the center of the pseudomolecule, indicating likely centromeric regions (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S8A, B</bold>
</xref>). In some cases, such as Chr2, 3, and 4, there was a marked reduction in variants at the telomeric regions (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S8A</bold>
</xref>). Nucleotide diversity, calculated as Pi, was based on the number of nucleotide differences per site between two sequences within a group. This revealed Group 2 containing the reference genome Daliak was the most diverse, whereas Groups 3 and 4 consistently exhibited lower diversity (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>). This reflected the phylogeny and population structure analysis (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>). Furthermore, all the groups exhibited similar levels of reduced nucleotide diversity that aligned with a reduced number of detected variants located near the putative centromere (~27 Mb Chr1; <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>). This was consistent across pseudomolecules (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S8</bold>
</xref>). Some diversity hotspots were identified where there was a marked increase in Pi values, including positions at ~15 Mb Chr1 (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>), ~5 Mb and ~45 Mb Chr3 and ~48 Mb Chr 5 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S8</bold>
</xref>).</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Distribution of variants along the subterranean clover chromosome 1 based on sequence data from 39 individuals representing 36 subterranean clover cultivars. <bold>(A)</bold> Distribution of number of variants (all) and the number of nucleotide differences per site between two sequences in a 500 kb window within a population (Pi) for each group or subgroup as defined in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>. <bold>(B)</bold> Distribution of mean copy number variants (CNV) of each group or subgroup relative to Daliak in a 500 kb window.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1103857-g006.tif"/>
</fig>
<p>The resequenced cultivars were also compared for CNVs relative to the Daliak reference. A distribution of mean CNV for each of the four identified groups was plotted along each chromosome (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref>). In Chr1, there were marked reductions in CNV in the putative centromere (~27 Mb) as well as near a telomere (~48 Mb). An increase in CNV, particularly for Group 4 individuals, aligned with an increase in Pi for that group at ~15 Mb (<xref ref-type="fig" rid="f6">
<bold>Figures&#xa0;6A, B</bold>
</xref>). Groups 3 and 4, the two most different from the other resequenced cultivars, exhibited increased CNVs relative to the Daliak reference compared to the other groups as shown in Chr1 at positions ~10, ~15, ~18, ~32, ~35, and ~55 Mb (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref>). Similar patterns were observed in each of the other pseudomolecules (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S9A, B</bold>
</xref>).</p>
<p>To gain further insight into genotype diversity within cultivars, 10 individuals each of two and three accessions from cultivars Daliak and Woogenellup, respectively, were resequenced. The 4,141 SNP variants identified relative to the Daliak reference were classed as either reference (Ref), alternate (Alt) or heterozygous (Het) genotypes and were plotted for each individual (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S10</bold>
</xref>). There were very few heterozygous loci, reflecting the species&#x2019; autogamous reproduction. As expected, the majority (95-98%) of loci in the Daliak seed lots contained reference alleles and exhibited high levels of genotype consistency among individuals. By contrast, Woogenellup had fewer reference alleles, with either ~25 or ~75% being alternate alleles. Other than accession AK1231, there was less genotype consistency in the other Woogenellup seed samples which contained individuals with either ~25 or ~75% reference alleles (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S10</bold>
</xref>).</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<title>Discussion</title>
<p>We have generated a new scaffolded genome assembly for <italic>Trifolium subterraneum</italic> (Ts) by integrating multiple next-generation short and long genomic sequence sources combined with graphical genotypes, linkage analysis and Hi-C analysis. Revealing a genome similar in size to prior assemblies of cultivar Daliak (<xref ref-type="bibr" rid="B43">Kaur et&#xa0;al., 2017b</xref>), the new assembly TSUd_r3.0 markedly enhances pseudomolecule integrity, resolves large scale inversions, and increases the completeness of gene space. TSUd_r3.0 presents the highest quality Ts genome assembly for comparative studies with other legumes such as <italic>T. repens</italic> (Tr) and its progenitors (<xref ref-type="bibr" rid="B36">Griffiths et&#xa0;al., 2019</xref>), <italic>T. pratense</italic> (Tp) (<xref ref-type="bibr" rid="B21">De Vega et&#xa0;al., 2015</xref>; <xref ref-type="bibr" rid="B8">Bickhart et&#xa0;al., 2022</xref>), <italic>Medicago sativa</italic> (Ms) (<xref ref-type="bibr" rid="B58">Mudge and Farmer, 2021</xref>) and <italic>M. truncatula</italic> (Mt) (<xref ref-type="bibr" rid="B46">Krishnakumar et&#xa0;al., 2014</xref>) to date.</p>
<p>As described previously, earlier Ts reference genome assemblies showed a stepwise improvement in quality metrics. They were however derived from the same base assembly, TSUd_r1.1, reported by <xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al. (2016)</xref>. Alignment of that assembly with a Mt genome identified inversions suggesting possible genome duplications in TSUd_r1.1 assembly relative to Mt (Figure&#xa0;4A in <xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>). Alignment of our new assembly (TSUd_r3.0) with Mt did not show these inversion/duplications, yet they were detected when comparing our assembly with the original TSUd_r1.1. This strongly suggests these scaffold order issues are a feature of TRSUd_r1.1 that have been resolved in the TSUd_r3.0 reference genome. This has been supported further by graphical genotype analysis in the current study highlighting an improved scaffold order across all pseudomolecules. Comparative statistics reveal progression from earlier Ts assemblies, and that the current iteration is now of similar quality to other <italic>Trifolium</italic> species such as Tr (<xref ref-type="bibr" rid="B36">Griffiths et&#xa0;al., 2019</xref>) and Tp (<xref ref-type="bibr" rid="B21">De Vega et&#xa0;al., 2015</xref>; <xref ref-type="bibr" rid="B8">Bickhart et&#xa0;al., 2022</xref>). This improved Ts assembly offers a more robust platform for further genomic studies than previous assemblies. Additionally, gene prediction <italic>via</italic> multiple independent pathways indicated the new assembly captured the gene space more effectively than earlier iterations. This was reflected in the BUSCO score which improved from 77% (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>) to 87.6% (<xref ref-type="bibr" rid="B43">Kaur et&#xa0;al., 2017b</xref>) in previous iterations to the 94.4% in the current assembly. There was also a decrease in fragmented genes based on this analysis from 9.6% and 4.1% in previous versions to 1.7% in TSUd_r3.0. The transcriptome sequences alone accounted for approximately 43% of predicted genes, which contrasts with similar analysis in Tr where 86% of predicted genes were confirmed by transcriptional evidence using four diverse tissues from a mature plant (<xref ref-type="bibr" rid="B36">Griffiths et&#xa0;al., 2019</xref>). This difference in Ts may reflect sampling from a more restricted range of three tissues from seedlings, thereby yielding a transcriptome with reduced representation. However, it may reflect the gene prediction workflow which prioritized longer gene length when identical gene models were identified from the different pipelines (RNA-Seq or Iso-Seq transcriptome or peptide sequence or <italic>ab initio</italic>-derived) which in these cases may bias towards <italic>ab initio</italic> models.</p>
<p>Further evidence of the improved genome and annotation was the change in the number of clustered genes shared across a range of species. The current study identified 8% of 34,929 clusters as being Ts-specific compared with 49% of 30,048 clusters in the original assembly (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>). This striking difference in Ts-specific clusters suggests TSUd_r3.0 resolved mis-assemblies which in TSUd_r1.1 had generated an artificially large number of clusters not aligned with other species.</p>
<sec id="s4_1">
<title>Comparative genomics</title>
<p>Legume genomes are complex, arising from an historic whole genome duplication (<xref ref-type="bibr" rid="B13">Cannon et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B62">Ren et&#xa0;al., 2019</xref>). Subsequent genome evolution has resulted in x = 8 chromosomes being common within genera <italic>Medicago</italic> and <italic>Trifolium</italic>. Phylogenomic analysis based on 723 common genes in the current study indicates a common ancestor for Mt and <italic>Trifolium</italic> at 26 MYA, similar to the 23 MYA estimated using the Tp genome (<xref ref-type="bibr" rid="B21">De Vega et&#xa0;al., 2015</xref>), and greater than the 19 MYA calculated using an earlier iteration of the Ts genome (<xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>). This aligns with a putative origin of genus <italic>Trifolium</italic> in the early Miocene epoch, approximately 23 MYA (<xref ref-type="bibr" rid="B27">Ellison et&#xa0;al., 2006</xref>). Our new estimate of the divergence between <italic>Lotus</italic> (Robinoid) and <italic>Medicago</italic>/<italic>Trifolium</italic> spp. in the Papilionoideae inverted repeat-lacking clade (IRLC) is 59 MYA, an increase on the previous 48 MYA (Tp genome; <xref ref-type="bibr" rid="B21">De Vega et&#xa0;al., 2015</xref>) and 43 MYA (Ts genome; <xref ref-type="bibr" rid="B40">Hirakawa et&#xa0;al., 2016</xref>) estimates. Divergence of the common ancestor of Ts and the agronomically important white clover (Tr) dates back 11 MYA. Tr, however, is a recent (15-28 KYA) allopolyploid derived from the hybridization of two progenitor species that themselves diverged from a common ancestor approximately 500 KYA (<xref ref-type="bibr" rid="B36">Griffiths et&#xa0;al., 2019</xref>). This highlights that in the period since they shared common ancestors, Ts and Tr have followed quite different evolutionary paths yet retained significant macrosynteny, as described below.</p>
<p>More detailed analysis based on whole genome comparison is enhanced by quality assemblies. The first genome assembly of a forage legume was implemented in Mt (<xref ref-type="bibr" rid="B78">Young et&#xa0;al., 2011</xref>) and its iterative improvement as a model genome has provided a focal point for legume comparative genomic analysis. Combined with robust assemblies of Tr (<xref ref-type="bibr" rid="B36">Griffiths et&#xa0;al., 2019</xref>) and Tp (<xref ref-type="bibr" rid="B21">De Vega et&#xa0;al., 2015</xref>; <xref ref-type="bibr" rid="B8">Bickhart et&#xa0;al., 2022</xref>), our improved assembly of Ts, by correcting inversions and scaffolding errors, clarifies macrosynteny among these <italic>Trifolium</italic> species and Mt to a greater extent than previous Ts assemblies. The generally conserved synteny with Ts, Tr and Mt contrasts by comparison to Tp. While <xref ref-type="bibr" rid="B27">Ellison et&#xa0;al. (2006)</xref> have shown that among the economically important species of <italic>Trifolium</italic>, Tp is the closest relative of Ts, macrosynteny reveals a much more complex relationship. Ts and Tp belong to section Trichocephalum, whereas Tr is belongs to section Trifoliastrum within tribe Trifolieae (<xref ref-type="bibr" rid="B27">Ellison et&#xa0;al., 2006</xref>). Despite this distance in relatedness based on nuclear ribosomal DNA analysis, Ts and Tr, which are in different sections and diverged 11 MYA, have more highly conserved macrosynteny than Ts and Tp, which diverged 9 MYA. This Ts/Tr conservation includes signatures such as the inversion on the proximal arm of Chromosome 1 relative to Mt (<xref ref-type="bibr" rid="B35">Griffiths et&#xa0;al., 2013</xref>) and, as shown in the current study, approximately similar chromosome sizes. Furthermore, the Ts macrosyntenic conservation extends to large portions of the <italic>Medicago</italic> genome which has an earlier divergence from <italic>Trifolium</italic> of 26 MYA.</p>
<p>Presumably, the complexity of the macrosynteny between Ts and Tp was induced during the loss of a chromosome to create x = 7 in Tp. Genome transmission during chromosome loss in dysploidy is often convoluted (<xref ref-type="bibr" rid="B55">Mayrose and Lysak, 2021</xref>). This may have contributed to the Tp genome being reorganized relative to Ts, Tr and Mt, despite its more recent divergence time from Ts. The new genome assembly and comparison reveals rearrangement in <italic>Trifolium</italic> that is more complex than some other economic genera such as Gossypium and sister genera (<xref ref-type="bibr" rid="B73">Udall et&#xa0;al., 2019</xref>). With base chromosome numbers ranging from 5 to 9 (<xref ref-type="bibr" rid="B27">Ellison et&#xa0;al., 2006</xref>; <xref ref-type="bibr" rid="B11">Bouziane et&#xa0;al., 2019</xref>) and recent speciation events (<xref ref-type="bibr" rid="B36">Griffiths et&#xa0;al., 2019</xref>), the genus could be of interest for studies of genome evolution and reorganization.</p>
</sec>
<sec id="s4_2">
<title>Cultivar resequencing reveals diversity and variant analysis</title>
<p>The impact of sequence diversity, copy number variants (CNVs) and structural variants (SVs) on species diversification and breeding outcomes has come to the fore recently (<xref ref-type="bibr" rid="B75">Wellenreuther et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B20">Della Coletta et&#xa0;al., 2021</xref>). Characterization of SNP diversity and CNV/SVs is a motivation for widespread resequencing for pangenome analysis and is critical to accessing molecular diversity for breeding (<xref ref-type="bibr" rid="B56">McCouch et&#xa0;al., 2020</xref>). Such surveys in Ts are of particular interest due to the broad adaptation of the species, subspecific classification, and presence of chromosome-level SVs (<xref ref-type="bibr" rid="B11">Bouziane et&#xa0;al., 2019</xref>).</p>
<p>Here we report the first steps toward a wider pangenome analysis for Ts, using 36 cultivars from a core collection of subterranean clover (<xref ref-type="bibr" rid="B32">Ghamkhar et&#xa0;al., 2015</xref>). Our analysis has revealed SNPs and CNVs across the genome which have given insight into intra-specific and intra-subspecific relationships. There are currently three accepted subspecies in Ts: ssp. <italic>subterraneum</italic>, ssp. <italic>yanninicum</italic>, and ssp. <italic>brachycalycinum</italic> (<xref ref-type="bibr" rid="B60">Nichols et&#xa0;al., 2013</xref>). While there are geographic distribution overlaps among these subspecies (<xref ref-type="bibr" rid="B32">Ghamkhar et&#xa0;al., 2015</xref>), they exhibit specific traits and niche environmental preferences (<xref ref-type="bibr" rid="B60">Nichols et&#xa0;al., 2013</xref>; <xref ref-type="bibr" rid="B32">Ghamkhar et&#xa0;al., 2015</xref>). Recently, the Ts ssp. <italic>subterraneum</italic> cv. Daliak and ssp. <italic>yanninicum</italic> cv. Yarloop were resequenced and surveyed for SVs using an optical mapping strategy (<xref ref-type="bibr" rid="B79">Yuan et&#xa0;al., 2018</xref>). That investigation of two genotypes revealed 31 regions with multiple SVs. Our analysis has revealed SNPs and CNVs across the genome which have given insight into intra-specific and intra-subspecific relationships. The phylogenetic analysis of sequence from 36 cultivars in this study described groups that reflect the known taxonomical relationships with Groups 1a, 1b and 2, which are comprised of Ts ssp. <italic>subterraneum</italic>, and Groups 3 and 4 made up of Ts ssp. <italic>yanninicum</italic> and Ts ssp. <italic>brachycalycinum</italic>, respectively. Splitting the Ts ssp. <italic>subterraneum</italic> accessions into three groups suggests that the molecular assessment of this subspecies contradicts or adds previously cryptic insights to the taxonomic classification. The cultivars reflect their known pedigree and taxonomic categorization apart from three exceptions. Cultivar Larisa was placed with the Ts ssp. <italic>subterraneum</italic> cultivars in Group 1a based on genome data yet is taxonomically categorized as Ts ssp. <italic>yanninicum</italic>. Similarly, genomic data indicates Mt Barker should be classified as Ts ssp. <italic>yanninicum</italic> rather than its taxonomic Ts ssp. <italic>subterraneum</italic> characterization. Woogenellup is a naturalized ecotype and considered to be Ts ssp. <italic>subterraneum</italic>, but the sequence data indicates it is part of Group 4, Ts ssp. <italic>brachycalycinum.</italic> The phylogenetic relationship of Woogenellup suggests similarities with the other groups, hinting at a possible hybridization event between Ts subspecies, most likely Ts ssp. <italic>subterraneum</italic> and Ts ssp. <italic>brachycalycinum</italic>, based on the genomic outcomes. It may also be a result of an admixture event in seed increase, as discussed below. Nevertheless, these discrepancies between genomic and taxonomic data, particularly the partitioning of Ts ssp. <italic>subterraneum</italic> across three groups suggest that the subspecific classification should be reviewed.</p>
<p>Assessment of SNP variants across pseudomolecules showed differences among the identified population groups with Group 2 being the most variable and Groups 3 and 4 the least. Spatial trends for the SNPs within each pseudomolecule were consistent with decreases at the telomeres and in many cases towards the center, consistent with metacentric chromosome structure in genus <italic>Trifolium</italic> (<xref ref-type="bibr" rid="B76">Williams et&#xa0;al., 2012</xref>). CNVs showed similar patterns although some regions of increased variation were found in different sites to the SNP variants and in some cases were specific to different Ts Groups. In summary, the detailed genetic diversity at the genome level captured in this study has provided valuable insight into chromosome-wide variant analysis and its relationship among and within Ts groupings.</p>
<p>Given the autogamous reproduction of Ts, an unexpected outcome was the sequence variation between multiple accessions of cultivars Coolamon, Denmark and Leura shown in the phylogenetic tree. The differences among multiple accessions of the same cultivar were greater than that of two very closely related cultivars, Geraldton and its mutant-derivative Uniwager. While the differences among accessions of the same cultivar could represent trapped heterozygosity, it also suggests admixture of two or more populations into an accession. This can happen during seed increase at either genebank or breeding company level as cross-contamination of seed lots is one of many issues faced by genebanks (<xref ref-type="bibr" rid="B66">Singh et&#xa0;al., 2012</xref>), but can be detected and corrected using next generation sequencing (<xref ref-type="bibr" rid="B67">Singh et&#xa0;al., 2019</xref>) even when phenotypic differences are cryptic. This was exemplified by extending our genomic survey to 10 individual plants within two accessions of cv Daliak and three of cv Woogenellup, revealed a likely case of admixture in accession Woogenellup AK1342. Admixture may also be a contributing factor to some accessions in the phylogenetic tree aligning with one group when pedigree data and conventional subspecies designation suggests placement in another group. For instance, the Woogenellup accession AK1342 with apparent admixture is assigned to Group 4 based on genomic data, but pedigree suggests affinity with members of Group 1. Given the challenges of admixture detection in outcrossing forage species including Tr, Tp and Ms; Ts may offer useful insight into frequency of accession mixtures within forage genebanks.</p>
</sec>
<sec id="s4_3" sec-type="conclusions">
<title>Conclusions</title>
<p>This research has provided a new high-quality assembly of the Ts genome containing additional sequence data, <italic>de novo</italic> reassembly and pseudomolecule ordering. These data in conjunction with Hi-C scaffolding underpin significant improvements in assembly size and quality metrics. In addition to refining phylogenetic relationships with other economic forage legumes, this assembly has enabled new insight to relationships among and within Ts subspecies and cultivars. The discovery of likely admixture in accessions suggests molecular analysis of genebank accessions to provide definitive accession sources is warranted. Despite potential admixture, the distinct sub-grouping of cultivars into four groups, specifically Group 2 (Ts ssp. <italic>subterraneum</italic>) and its proximity to Groups 3 (Ts ssp. <italic>yanninicum</italic>) and 4 (Ts ssp. <italic>brachycalicinum</italic>) rather than Group 1 (Ts ssp. <italic>subterraneum</italic>), indicates a review of sub-specific classification in the species is justified.</p>
<p>Ts is of interest for multiple economically significant traits. It exhibits variation for suppression of methanogenesis in the rumen (<xref ref-type="bibr" rid="B42">Kaur et&#xa0;al., 2017a</xref>; <xref ref-type="bibr" rid="B33">Ghamkhar et&#xa0;al., 2018</xref>), which genome information can help further elucidate. Genome data may also offer insight into adaptive variation to edaphic stress in forage legumes. The new Ts assembly further enables comparative genetic analysis for key traits such as geocarpy with access to other geocarpic species such as <italic>Arachis hypogaea</italic> (<xref ref-type="bibr" rid="B7">Bertioli et&#xa0;al., 2019</xref>). Improving Ts for key traits including methane inhibition as a mitigating response to emissions from pastoral agriculture, and other traits of interest such as geocarpy is an important task for forage breeders. To enable acceleration of this work, assembling a Ts pangenome and identification of structural variants, including among known aneuploid sources, would provide an invaluable understanding of this species. This continued pangenome development should ideally include expression data for traits of interest and specifically climate adaptation traits among the Ts core collection. There is now an opportunity for application of this Ts genome and assembly in plant improvement in trait genetic analysis, genomic selection, improved cultivar and accession quality assurance and tracking, as well as utilization of genebank material.</p>
</sec>
</sec>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found in the text. The assembled genome and gene sequences are available at Plant GARDEN (<uri xlink:href="https://plantgarden.jp/ja/list/t3900/genome/t3900.G002">https://plantgarden.jp/ja/list/t3900/genome/t3900.G002</uri>). The genome sequence data have been submitted to the DDBJ/ENA/NCBI public sequence databases under the BioProject ID PRJDB7187 (<uri xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/PRJDB7187/">https://www.ncbi.nlm.nih.gov/bioproject/PRJDB7187/</uri>). Sequence data can also be found at the DNA Data Bank of Japan (DDBJ) Sequence Read Archive (<uri xlink:href="https://www.ddbj.nig.ac.jp/ddbj/updt-form-e.html">https://www.ddbj.nig.ac.jp/ddbj/updt-form-e.html</uri>) under the following submission codes: DRA007081 (hirakawa-0130); DRA007082 (hirakawa-0129); DRA007083 (hirakawa-0128); DRA007084 (hirakawa-0131); DRA007085 (hirakawa-0132); DRA007086 (hirakawa-0133); DRA007087 (hirakawa-0135); DRA007088 (hirakawa-0136). Additional repositories for sequence data are detailed in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>KS: Data curation; Formal analysis; Investigation; Visualization; Writing &#x2013; original draft. RM: Data curation; Formal analysis; Investigation; Visualization; Writing &#x2013; original draft. HH: Data curation; Formal analysis; Investigation; Visualization; Writing &#x2013; original draft. HN: Data curation; Formal analysis; Investigation; Visualization; Writing &#x2013; original draft. AG: Data curation; Formal analysis; Investigation; Visualization; Writing &#x2013; original draft. BB: Conceptualization; Writing &#x2013; original draft; Writing &#x2013; review &amp; editing. KG: Conceptualization; Project administration; Resources; Supervision; Funding acquisition; Writing &#x2013; original draft; Writing &#x2013; review &amp; editing. AGG: Conceptualization, Supervision; Visualization, Writing &#x2013; original draft; Writing &#x2013; review &amp; editing. SI: Conceptualization; Data curation; Formal analysis; Funding acquisition; Investigation; Project administration; Resources; Supervision; Visualization; Writing &#x2013; original draft; Writing &#x2013; review &amp; editing. All authors contributed to the article and approved the submitted version</p>
</sec>
</body>
<back>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>Kazusa DNA Research Institute provided support and funding for KS, AG, HH, HN and SI; Tea Break provided support for RM; AgResearch provided support and funding for KG, BB and AGG; New Zealand Ministry of Business, Innovation and Employment (contract C10X1701) provided funding for KG.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>The authors acknowledge Michelle Williamson from Margot Forde Germplasm Centre, AgResearch for advice and providing the seed material. Dr Jeanne Jacobs is acknowledged for input during project inception and Dr Tony Conner for encouraging this collaboration and providing support through AgResearch. We would also like to thank Akiko Watanabe, Yoshie Kishida, Shinobu Nakayama, Shigemi Sasamoto, Chiharu Minami, Hisano Tsuruoka, Akiko Komaki and Mitsuyo Kohara whose technical support at the Kazusa DNA Research Institute&#x2019;s Laboratory of Plant Genomics and Genetics is greatly appreciated. We also thank our colleague Dr Ali Saei and the reviewers whose constructive suggestions contributed to improvements in this manuscript.</p>
</ack>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>Author RM is employed by Tea Break Bioinformatics Limited.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial and financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2023.1103857/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpls.2023.1103857/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abdi</surname> <given-names>A. I.</given-names>
</name>
<name>
<surname>Nichols</surname> <given-names>P. G. H.</given-names>
</name>
<name>
<surname>Kaur</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Wintle</surname> <given-names>B. J.</given-names>
</name>
<name>
<surname>Erskine</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Morphological diversity within a core collection of subterranean clover (<italic>Trifolium subterraneum</italic> l.): Lessons in pasture adaptation from the wild</article-title>. <source>PloS One</source> <volume>15</volume> (<issue>1</issue>), <fpage>e0223699</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0223699</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alexander</surname> <given-names>D. H.</given-names>
</name>
<name>
<surname>Novembre</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lange</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Fast model-based estimation of ancestry in unrelated individuals</article-title>. <source>Genome Res.</source> <volume>19</volume> (<issue>9</issue>), <fpage>1655</fpage>&#x2013;<lpage>1664</lpage>. doi: <pub-id pub-id-type="doi">10.1101/gr.094052.109</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Andrews</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2010</year>) <source>FastQC: A quality control tool for high throughput sequence data</source>. Available at: <uri xlink:href="http://www.bioinformatics.babraham.ac.uk/projects/fastqc/">http://www.bioinformatics.babraham.ac.uk/projects/fastqc/</uri>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Banik</surname> <given-names>B. K.</given-names>
</name>
<name>
<surname>Durmic</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Erskine</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Nichols</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Ghamkhar</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Vercoe</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Variability of <italic>in vitro</italic> ruminal fermentation and methanogenic potential in the pasture legume biserrula (<italic>Biserrula pelecinus</italic> l.)</article-title>. <source>Crop Pasture Sci.</source> <volume>64</volume> (<issue>4</issue>), <fpage>409</fpage>&#x2013;<lpage>416</lpage>. doi: <pub-id pub-id-type="doi">10.1071/CP13073</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bankevich</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Pevzner</surname> <given-names>P. A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>TruSPAdes: barcode assembly of TruSeq synthetic long reads</article-title>. <source>Nat. Methods</source> <volume>13</volume> (<issue>3</issue>), <fpage>248</fpage>&#x2013;<lpage>250</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nmeth.3737</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bennett</surname> <given-names>M. D.</given-names>
</name>
<name>
<surname>Leitch</surname> <given-names>I. J.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Nuclear DNA amounts in angiosperms: Targets, trends and tomorrow</article-title>. <source>Ann. Bot.</source> <volume>107</volume>, <fpage>467</fpage>&#x2013;<lpage>590</lpage>. doi: <pub-id pub-id-type="doi">10.1093/aob/mcq258</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bertioli</surname> <given-names>D. J.</given-names>
</name>
<name>
<surname>Jenkins</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Clevenger</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Dudchenko</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Seijo</surname> <given-names>G.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>The genome sequence of segmental allotetraploid peanut arachis hypogaea</article-title>. <source>Nat. Genet.</source> <volume>51</volume> (<issue>5</issue>), <fpage>877</fpage>&#x2013;<lpage>884</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41588-019-0405-z</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bickhart</surname> <given-names>D. M.</given-names>
</name>
<name>
<surname>Koch</surname> <given-names>L. M.</given-names>
</name>
<name>
<surname>Smith</surname> <given-names>T. P. L.</given-names>
</name>
<name>
<surname>Riday</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Sullivan</surname> <given-names>M. L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Chromosome-scale assembly of the highly heterozygous genome of red clover (<italic>Trifolium pratense</italic> l.), an allogamous forage crop species</article-title>. <source>Gigabyte</source> <volume>2022</volume>, <fpage>1</fpage>&#x2013;<lpage>13</lpage>. doi: <pub-id pub-id-type="doi">10.46471/gigabyte.42</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bickhart</surname> <given-names>D. M.</given-names>
</name>
<name>
<surname>Rosen</surname> <given-names>B. D.</given-names>
</name>
<name>
<surname>Koren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sayre</surname> <given-names>B. L.</given-names>
</name>
<name>
<surname>Hastie</surname> <given-names>A. R.</given-names>
</name>
<name>
<surname>Chan</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). <article-title>Single-molecule sequencing and chromatin conformation capture enable <italic>de novo</italic> reference assembly of the domestic goat genome</article-title>. <source>Nat. Genet.</source> <volume>49</volume> (<issue>4</issue>), <fpage>643</fpage>&#x2013;<lpage>650</lpage>. doi: <pub-id pub-id-type="doi">10.1038/ng.3802</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boetzer</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Pirovano</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Toward almost closed genomes with GapFiller</article-title>. <source>Genome Biol.</source> <volume>13</volume> (<issue>6</issue>), <fpage>R56</fpage>. doi: <pub-id pub-id-type="doi">10.1186/gb-2012-13-6-r56</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bouziane</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Issolah</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Tahar</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Analysis of the chromosome variation within some natural populations of subterranean clover (<italic>Trifolium subterraneum</italic> l., fabaceae) in Algeria</article-title>. <source>Caryologia</source> <volume>72</volume> (<issue>4</issue>), <fpage>93</fpage>&#x2013;<lpage>104</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.13128/caryologia-164</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bradbury</surname> <given-names>P. J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Kroon</surname> <given-names>D. E.</given-names>
</name>
<name>
<surname>Casstevens</surname> <given-names>T. M.</given-names>
</name>
<name>
<surname>Ramdoss</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Buckler</surname> <given-names>E. S.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>TASSEL: Software for association mapping of complex traits in diverse samples</article-title>. <source>Bioinformatics</source> <volume>23</volume> (<issue>19</issue>), <fpage>2633</fpage>&#x2013;<lpage>2635</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btm308</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cannon</surname> <given-names>S. B.</given-names>
</name>
<name>
<surname>McKain</surname> <given-names>M. R.</given-names>
</name>
<name>
<surname>Harkess</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Nelson</surname> <given-names>M. N.</given-names>
</name>
<name>
<surname>Dash</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Deyholos</surname> <given-names>M. K.</given-names>
</name>
<etal/>
</person-group>. (<year>2014</year>). <article-title>Multiple polyploidy events in the early radiation of nodulating and nonnodulating legumes</article-title>. <source>Mol. Biol. Evol.</source> <volume>32</volume> (<issue>1</issue>), <fpage>193</fpage>&#x2013;<lpage>210</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/molbev/msu296</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Castresana</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Selection of conserved blocks from multiple alignments for their use in phylogenetic analysis</article-title>. <source>Mol. Biol. Evol.</source> <volume>17</volume> (<issue>4</issue>), <fpage>540</fpage>&#x2013;<lpage>552</lpage>. doi: <pub-id pub-id-type="doi">10.1093/oxfordjournals.molbev.a026334</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chapman</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Ho</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Sunkara</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Schroth</surname> <given-names>G. P.</given-names>
</name>
<name>
<surname>Rokhsar</surname> <given-names>D. S.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Meraculous: <italic>De novo</italic> genome assembly with short paired-end reads</article-title>. <source>PloS One</source> <volume>6</volume> (<issue>8</issue>), <fpage>e23501</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0023501</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chaulagain</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Frugoli</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>The regulation of nodule number in legumes is a balance of three signal transduction pathways</article-title>. <source>Int. J. Mol. Sci.</source> <volume>22</volume> (<issue>3</issue>), <fpage>1117</fpage>. doi: <pub-id pub-id-type="doi">10.3390/ijms22031117</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chikhi</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Medvedev</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Informed and automated k-mer size selection for genome assembly</article-title>. <source>Bioinformatics</source> <volume>30</volume> (<issue>1</issue>), <fpage>31</fpage>&#x2013;<lpage>37</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btt310</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Crusoe</surname> <given-names>M. R.</given-names>
</name>
<name>
<surname>Alameldin</surname> <given-names>H. F.</given-names>
</name>
<name>
<surname>Awad</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Boucher</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Caldwell</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Cartwright</surname> <given-names>R.</given-names>
</name>
<etal/>
</person-group>. (<year>2015</year>). <article-title>The khmer software package: Enabling efficient nucleotide sequence analysis</article-title>. <source>F1000Research</source> <volume>4</volume>, <fpage>900</fpage>&#x2013;<lpage>900</lpage>. doi: <pub-id pub-id-type="doi">10.12688/f1000research.6924.1</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Danecek</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Auton</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Abecasis</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Albers</surname> <given-names>C. A.</given-names>
</name>
<name>
<surname>Banks</surname> <given-names>E.</given-names>
</name>
<name>
<surname>DePristo</surname> <given-names>M. A.</given-names>
</name>
<etal/>
</person-group>. (<year>2011</year>). <article-title>The variant call format and VCFtools</article-title>. <source>Bioinformatics</source> <volume>27</volume> (<issue>15</issue>), <fpage>2156</fpage>&#x2013;<lpage>2158</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btr330</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Della Coletta</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Ou</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Hufford</surname> <given-names>M. B.</given-names>
</name>
<name>
<surname>Hirsch</surname> <given-names>C. N.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>How the pan-genome is changing crop genomics and improvement</article-title>. <source>Genome Biol.</source> <volume>22</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>19</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13059-020-02224-8</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Vega</surname> <given-names>J. J.</given-names>
</name>
<name>
<surname>Ayling</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Hegarty</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Kudrna</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Goicoechea</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Ergon</surname> <given-names>&#xc5;.</given-names>
</name>
<etal/>
</person-group>. (<year>2015</year>). <article-title>Red clover (<italic>Trifolium pratense</italic> l.) draft genome provides a platform for trait improvement</article-title>. <source>Sci. Rep.</source> <volume>5</volume> (<issue>1</issue>), <fpage>17394</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/srep17394</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dobin</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Davis</surname> <given-names>C. A.</given-names>
</name>
<name>
<surname>Schlesinger</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Drenkow</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zaleski</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Jha</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2012</year>). <article-title>STAR: ultrafast universal RNA-seq aligner</article-title>. <source>Bioinformatics</source> <volume>29</volume> (<issue>1</issue>), <fpage>15</fpage>&#x2013;<lpage>21</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/bts635</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dudchenko</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Pham</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Lui</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Batra</surname> <given-names>S. S.</given-names>
</name>
<name>
<surname>Hoeger</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Nyquist</surname> <given-names>S. K.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>Hi-C yields chromosome-length scaffolds for a legume genome, <italic>Trifolium subterraneum</italic>
</article-title>. <source>bioRxiv</source>, <fpage>473553</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1101/473553</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Durand</surname> <given-names>N. C.</given-names>
</name>
<name>
<surname>Robinson</surname> <given-names>J. T.</given-names>
</name>
<name>
<surname>Shamim</surname> <given-names>M. S.</given-names>
</name>
<name>
<surname>Machol</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Mesirov</surname> <given-names>J. P.</given-names>
</name>
<name>
<surname>Lander</surname> <given-names>E. S.</given-names>
</name>
<etal/>
</person-group>. (<year>2016</year>). <article-title>Juicebox provides a visualization system for Hi-c contact maps with unlimited zoom</article-title>. <source>Cell Syst.</source> <volume>3</volume> (<issue>1</issue>), <fpage>99</fpage>&#x2013;<lpage>101</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cels.2015.07.012</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Edgar</surname> <given-names>R. C.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>MUSCLE: A multiple sequence alignment method with reduced time and space complexity</article-title>. <source>BMC Bioinf.</source> <volume>5</volume> (<issue>1</issue>), <fpage>113</fpage>. doi: <pub-id pub-id-type="doi">10.1186/1471-2105-5-113</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Edgar</surname> <given-names>R. C.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Search and clustering orders of magnitude faster than BLAST</article-title>. <source>Bioinformatics</source> <volume>26</volume> (<issue>19</issue>), <fpage>2460</fpage>&#x2013;<lpage>2461</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btq461</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ellison</surname> <given-names>N. W.</given-names>
</name>
<name>
<surname>Liston</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Steiner</surname> <given-names>J. J.</given-names>
</name>
<name>
<surname>Williams</surname> <given-names>W. M.</given-names>
</name>
<name>
<surname>Taylor</surname> <given-names>N. L.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Molecular phylogenetics of the clover genus (Trifolium&#x2013;leguminosae)</article-title>. <source>Mol. Phylogenet. Evol.</source> <volume>39</volume> (<issue>3</issue>), <fpage>688</fpage>&#x2013;<lpage>705</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ympev.2006.01.004</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>English</surname> <given-names>A. C.</given-names>
</name>
<name>
<surname>Richards</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Han</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Vee</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Qu</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2012</year>). <article-title>Mind the gap: Upgrading genomes with pacific biosciences RS long-read sequencing technology</article-title>. <source>PloS One</source> <volume>7</volume> (<issue>11</issue>), <fpage>e47768</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0047768</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Niu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>CD-HIT: accelerated for clustering the next-generation sequencing data</article-title>. <source>Bioinformatics</source> <volume>28</volume> (<issue>23</issue>), <fpage>3150</fpage>&#x2013;<lpage>3152</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/bts565</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Bertrand</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Chia</surname> <given-names>B. K. H.</given-names>
</name>
<name>
<surname>Nagarajan</surname> <given-names>N.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>OPERA-LG: efficient and exact scaffolding of large, repeat-rich eukaryotic genomes with performance guarantees</article-title>. <source>Genome Biol.</source> <volume>17</volume> (<issue>1</issue>), <fpage>102</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13059-016-0951-y</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ghamkhar</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Isobe</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Nichols</surname> <given-names>P. G. H.</given-names>
</name>
<name>
<surname>Faithfull</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Ryan</surname> <given-names>M. H.</given-names>
</name>
<name>
<surname>Snowball</surname> <given-names>R.</given-names>
</name>
<etal/>
</person-group>. (<year>2012</year>). <article-title>The first genetic maps for subterranean clover (<italic>Trifolium subterraneum</italic> l.) and comparative genomics with <italic>T. pratense</italic> l. and <italic>Medicago truncatula</italic> gaertn. to identify new molecular markers for breeding</article-title>. <source>Mol. Breed.</source> <volume>30</volume> (<issue>1</issue>), <fpage>213</fpage>&#x2013;<lpage>226</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11032-011-9612-8</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ghamkhar</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Nichols</surname> <given-names>P. G. H.</given-names>
</name>
<name>
<surname>Erskine</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Snowball</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Murillo</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Appels</surname> <given-names>R.</given-names>
</name>
<etal/>
</person-group>. (<year>2015</year>). <article-title>Hotspots and gaps in the world collection of subterranean clover (<italic>Trifolium subterraneum</italic> l.)</article-title>. <source>J. Agric. Sci.</source> <volume>153</volume> (<issue>6</issue>), <fpage>1069</fpage>&#x2013;<lpage>1083</lpage>. doi: <pub-id pub-id-type="doi">10.1017/S0021859614000793</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ghamkhar</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Rochfort</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Banik</surname> <given-names>B. K.</given-names>
</name>
<name>
<surname>Revell</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Candidate metabolites for methane mitigation in the forage legume biserrula</article-title>. <source>Agron. Sustain. Dev.</source> <volume>38</volume> (<issue>3</issue>), <fpage>30</fpage>. doi: <pub-id pub-id-type="doi">10.1007/s13593-018-0510-x</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ghelfi</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Shirasawa</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Hirakawa</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Isobe</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Hayai-annotation plants: An ultra-fast and comprehensive functional gene annotation system in plants</article-title>. <source>Bioinformatics</source> <volume>35</volume> (<issue>21</issue>), <fpage>4427</fpage>&#x2013;<lpage>4429</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btz380</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Griffiths</surname> <given-names>A. G.</given-names>
</name>
<name>
<surname>Barrett</surname> <given-names>B. A.</given-names>
</name>
<name>
<surname>Simon</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>A. K.</given-names>
</name>
<name>
<surname>Bickerstaff</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Anderson</surname> <given-names>C. B.</given-names>
</name>
<etal/>
</person-group>. (<year>2013</year>). <article-title>An integrated genetic linkage map for white clover (<italic>Trifolium repens</italic> l.) with alignment to <italic>Medicago</italic>
</article-title>. <source>BMC Genomics</source> <volume>14</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>17</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/1471-2164-14-388</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Griffiths</surname> <given-names>A. G.</given-names>
</name>
<name>
<surname>Moraga</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Tausen</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Gupta</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Bilton</surname> <given-names>T. P.</given-names>
</name>
<name>
<surname>Campbell</surname> <given-names>M. A.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>Breaking free: The genomics of allopolyploidy-facilitated niche expansion in white clover</article-title>. <source>Plant Cell</source> <volume>31</volume> (<issue>7</issue>), <fpage>1466</fpage>&#x2013;<lpage>1487</lpage>. doi: <pub-id pub-id-type="doi">10.1105/tpc.18.00606</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haas</surname> <given-names>B. J.</given-names>
</name>
<name>
<surname>Papanicolaou</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Yassour</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Grabherr</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Blood</surname> <given-names>P. D.</given-names>
</name>
<name>
<surname>Bowden</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2013</year>). <article-title>
<italic>De novo</italic> transcript sequence reconstruction from RNA-seq using the trinity platform for reference generation and analysis</article-title>. <source>Nat. Protoc.</source> <volume>8</volume> (<issue>8</issue>), <fpage>1494</fpage>&#x2013;<lpage>1512</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nprot.2013.084</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hackl</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Hedrich</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Schultz</surname> <given-names>J.</given-names>
</name>
<name>
<surname>F&#xf6;rster</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Proovread: Large-scale high-accuracy PacBio correction through iterative short read consensus</article-title>. <source>Bioinformatics</source> <volume>30</volume> (<issue>21</issue>), <fpage>3004</fpage>&#x2013;<lpage>3011</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btu392</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="thesis">
<person-group person-group-type="author">
<name>
<surname>Harris</surname> <given-names>R. S.</given-names>
</name>
</person-group> (<year>2007</year>). <source>Improved pairwise alignment of genomic DNA. PhD dissertation, the Pennsylvania state university</source> (<publisher-name>Pennsylvania, USA</publisher-name>: <publisher-loc>Pennsylvania State University</publisher-loc>), <fpage>84</fpage>.</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hirakawa</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Kaur</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Shirasawa</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Nichols</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Nagano</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Appels</surname> <given-names>R.</given-names>
</name>
<etal/>
</person-group>. (<year>2016</year>). <article-title>Draft genome sequence of subterranean clover, a reference for genus <italic>Trifolium</italic>
</article-title>. <source>Sci. Rep.</source> <volume>6</volume>, <fpage>30358</fpage>. doi: <pub-id pub-id-type="doi">10.1038/srep30358</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Lei</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Ding</surname> <given-names>S.-W.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Skewer: A fast and accurate adapter trimmer for next-generation sequencing paired-end reads</article-title>. <source>BMC Bioinf.</source> <volume>15</volume> (<issue>1</issue>), <fpage>182</fpage>. doi: <pub-id pub-id-type="doi">10.1186/1471-2105-15-182</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaur</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Appels</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Bayer</surname> <given-names>P. E.</given-names>
</name>
<name>
<surname>Keeble-Gagnere</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Hirakawa</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>a). <article-title>Climate clever clovers: New paradigm to reduce the environmental footprint of ruminants by breeding low methanogenic forages utilizing haplotype variation</article-title>. <source>Front. Plant Sci.</source> <volume>8</volume>. doi: <pub-id pub-id-type="doi">10.3389/fpls.2017.01463</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaur</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Bayer</surname> <given-names>P. E.</given-names>
</name>
<name>
<surname>Milec</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Vr&#xe1;na</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Appels</surname> <given-names>R.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>b). <article-title>An advanced reference genome of <italic>Trifolium subterraneum</italic> l. reveals genes related to agronomic performance</article-title>. <source>Plant Biotechnol. J.</source> <volume>15</volume> (<issue>8</issue>), <fpage>1034</fpage>&#x2013;<lpage>1046</lpage>. doi: <pub-id pub-id-type="doi">10.1111/pbi.12697</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kie&#x142;basa</surname> <given-names>S. M.</given-names>
</name>
<name>
<surname>Wan</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Sato</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Horton</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Frith</surname> <given-names>M. C.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Adaptive seeds tame genomic sequence comparison</article-title>. <source>Genome Res.</source> <volume>21</volume> (<issue>3</issue>), <fpage>487</fpage>&#x2013;<lpage>493</lpage>. doi: <pub-id pub-id-type="doi">10.1101/gr.113985.110</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koboldt</surname> <given-names>D. C.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Larson</surname> <given-names>D. E.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>D.</given-names>
</name>
<name>
<surname>McLellan</surname> <given-names>M. D.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2012</year>). <article-title>VarScan 2: Somatic mutation and copy number alteration discovery in cancer by exome sequencing</article-title>. <source>Genome Res.</source> <volume>22</volume> (<issue>3</issue>), <fpage>568</fpage>&#x2013;<lpage>576</lpage>. doi: <pub-id pub-id-type="doi">10.1101/gr.129684.111</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krishnakumar</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Rosen</surname> <given-names>B. D.</given-names>
</name>
<name>
<surname>Karamycheva</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Bidwell</surname> <given-names>S. L.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2014</year>). <article-title>MTGD: The medicago truncatula genome database</article-title>. <source>Plant Cell Physiol.</source> <volume>56</volume> (<issue>1</issue>), <fpage>e1</fpage>&#x2013;<lpage>e1</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/pcp/pcu179</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Stecher</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Knyaz</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Tamura</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>MEGA X: Molecular evolutionary genetics analysis across computing platforms</article-title>. <source>Mol. Biol. Evol.</source> <volume>35</volume> (<issue>6</issue>), <fpage>1547</fpage>&#x2013;<lpage>1549</lpage>. doi: <pub-id pub-id-type="doi">10.1093/molbev/msy096</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kurtz</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Phillippy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Delcher</surname> <given-names>A. L.</given-names>
</name>
<name>
<surname>Smoot</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Shumway</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Antonescu</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2004</year>). <article-title>Versatile and open software for comparing large genomes</article-title>. <source>Genome Biol.</source> <volume>5</volume>, <fpage>R12</fpage>. doi: <pub-id pub-id-type="doi">10.1186/gb-2004-5-2-r12</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Langmead</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Salzberg</surname> <given-names>S. L.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Fast gapped-read alignment with bowtie 2</article-title>. <source>Nat. Methods</source> <volume>9</volume> (<issue>4</issue>), <fpage>357</fpage>&#x2013;<lpage>359</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nmeth.1923</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Durbin</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Fast and accurate long-read alignment with burrows&#x2013;wheeler transform</article-title>. <source>Bioinformatics</source> <volume>26</volume> (<issue>5</issue>), <fpage>589</fpage>&#x2013;<lpage>595</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btp698</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Handsaker</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Wysoker</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Fennell</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Ruan</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Homer</surname> <given-names>N.</given-names>
</name>
<etal/>
</person-group>. (<year>2009</year>). <article-title>The sequence Alignment/Map format and SAMtools</article-title>. <source>Bioinformatics</source> <volume>25</volume> (<issue>16</issue>), <fpage>2078</fpage>&#x2013;<lpage>2079</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btp352</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Stoeckert</surname> <given-names>C. J.</given-names> <suffix>Jr.</suffix>
</name>
<name>
<surname>Roos</surname> <given-names>D. S.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>OrthoMCL: Identification of ortholog groups for eukaryotic genomes</article-title>. <source>Genome Res.</source> <volume>13</volume> (<issue>9</issue>), <fpage>2178</fpage>&#x2013;<lpage>2189</lpage>. doi: <pub-id pub-id-type="doi">10.1101/gr.1224503</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lieberman-Aiden</surname> <given-names>E.</given-names>
</name>
<name>
<surname>van Berkum</surname> <given-names>N. L.</given-names>
</name>
<name>
<surname>Williams</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Imakaev</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Ragoczy</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Telling</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2009</year>). <article-title>Comprehensive mapping of long-range interactions reveals folding principles of the human genome</article-title>. <source>Science</source> <volume>326</volume> (<issue>5950</issue>), <fpage>289</fpage>&#x2013;<lpage>293</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/science.1181369</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mar&#xe7;ais</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Kingsford</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>A fast, lock-free approach for efficient parallel counting of occurrences of k-mers</article-title>. <source>Bioinformatics</source> <volume>27</volume> (<issue>6</issue>), <fpage>764</fpage>&#x2013;<lpage>770</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btr011</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mayrose</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Lysak</surname> <given-names>M. A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>The evolution of chromosome numbers: Mechanistic models and experimental approaches</article-title>. <source>Genome Biol. Evol.</source> <volume>13</volume> (<issue>2</issue>), <fpage>evaa220</fpage>. doi: <pub-id pub-id-type="doi">10.1093/gbe/evaa220</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McCouch</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Navabi</surname> <given-names>Z. K.</given-names>
</name>
<name>
<surname>Abberton</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Anglin</surname> <given-names>N. L.</given-names>
</name>
<name>
<surname>Barbieri</surname> <given-names>R. L.</given-names>
</name>
<name>
<surname>Baum</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Mobilizing crop biodiversity</article-title>. <source>Mol. Plant</source> <volume>13</volume> (<issue>10</issue>), <fpage>1341</fpage>&#x2013;<lpage>1344</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.molp.2020.08.011</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Morley</surname> <given-names>F. H. W.</given-names>
</name>
<name>
<surname>Katznelson</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>1965</year>). <source>Colonization in Australia by trifolium subterraneum l. the genetics of colonizing species</source>. Ed. <person-group person-group-type="editor">
<name>
<surname>Baker</surname> <given-names>H.</given-names>
</name>
</person-group> (<publisher-loc>New York</publisher-loc>: <publisher-name>Academic Press</publisher-name>), <fpage>269</fpage>&#x2013;<lpage>282</lpage>.</citation>
</ref>
<ref id="B58">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mudge</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Farmer</surname> <given-names>A. D.</given-names>
</name>
</person-group> (<year>2021</year>). <source>Sequencing, assembly, and annotation of the alfalfa genome. the alfalfa genome</source>. Eds. <person-group person-group-type="editor">
<name>
<surname>Yu</surname> <given-names>L.-X.</given-names>
</name>
<name>
<surname>Kole</surname> <given-names>C.</given-names>
</name>
</person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>87</fpage>&#x2013;<lpage>109</lpage>.</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Muir</surname> <given-names>S. K.</given-names>
</name>
<name>
<surname>Kennedy</surname> <given-names>A. J.</given-names>
</name>
<name>
<surname>Kearney</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Hutton</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Thompson</surname> <given-names>A. N.</given-names>
</name>
<name>
<surname>Vercoe</surname> <given-names>P.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Offering subterranean clover can reduce methane emissions compared with perennial ryegrass pastures during late spring and summer in sheep</article-title>. <source>Anim. Prod. Sci.</source> <volume>60</volume> (<issue>11</issue>), <fpage>1449</fpage>&#x2013;<lpage>1458</lpage>. doi: <pub-id pub-id-type="doi">10.1071/AN18624</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nichols</surname> <given-names>P. G. H.</given-names>
</name>
<name>
<surname>Foster</surname> <given-names>K. J.</given-names>
</name>
<name>
<surname>Piano</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Pecetti</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Kaur</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Ghamkhar</surname> <given-names>K.</given-names>
</name>
<etal/>
</person-group>. (<year>2013</year>). <article-title>Genetic improvement of subterranean clover (<italic>Trifolium subterraneum</italic> l.). 1. germplasm, traits and future prospects</article-title>. <source>Crop Pasture Sci.</source> <volume>64</volume> (<issue>4</issue>), <fpage>312</fpage>&#x2013;<lpage>346</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1071/CP13118</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rao</surname> <given-names>S. S. P.</given-names>
</name>
<name>
<surname>Huntley</surname> <given-names>M. H.</given-names>
</name>
<name>
<surname>Durand</surname> <given-names>N. C.</given-names>
</name>
<name>
<surname>Stamenova</surname> <given-names>E. K.</given-names>
</name>
<name>
<surname>Bochkov</surname> <given-names>I. D.</given-names>
</name>
<name>
<surname>Robinson</surname> <given-names>J. T.</given-names>
</name>
<etal/>
</person-group>. (<year>2014</year>). <article-title>A 3D map of the human genome at kilobase resolution reveals principles of chromatin looping</article-title>. <source>Cell</source> <volume>159</volume> (<issue>7</issue>), <fpage>1665</fpage>&#x2013;<lpage>1680</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cell.2014.11.021</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Cannon</surname> <given-names>S. B.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Reconstruction of ancestral genome reveals chromosome evolution history for selected legume species</article-title>. <source>New Phytol.</source> <volume>223</volume> (<issue>4</issue>), <fpage>2090</fpage>&#x2013;<lpage>2103</lpage>. doi: <pub-id pub-id-type="doi">10.1111/nph.15770</pub-id>
</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sato</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Nakamura</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Kaneko</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Asamizu</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Kato</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Nakao</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2008</year>). <article-title>Genome structure of the legume, <italic>Lotus japonicus</italic>
</article-title>. <source>DNA Res.</source> <volume>15</volume> (<issue>4</issue>), <fpage>227</fpage>&#x2013;<lpage>239</lpage>. doi: <pub-id pub-id-type="doi">10.1093/dnares/dsn008</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sedlazeck</surname> <given-names>F. J.</given-names>
</name>
<name>
<surname>Rescheneder</surname> <given-names>P.</given-names>
</name>
<name>
<surname>von Haeseler</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>NextGenMap: Fast and accurate read mapping in highly polymorphic genomes</article-title>. <source>Bioinformatics</source> <volume>29</volume> (<issue>21</issue>), <fpage>2790</fpage>&#x2013;<lpage>2791</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btt468</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sim&#xe3;o</surname> <given-names>F. A.</given-names>
</name>
<name>
<surname>Waterhouse</surname> <given-names>R. M.</given-names>
</name>
<name>
<surname>Ioannidis</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Kriventseva</surname> <given-names>E. V.</given-names>
</name>
<name>
<surname>Zdobnov</surname> <given-names>E. M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>BUSCO: Assessing genome assembly and annotation completeness with single-copy orthologs</article-title>. <source>Bioinformatics</source> <volume>31</volume>, <fpage>3210</fpage>&#x2013;<lpage>3212</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btv351</pub-id>
</citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singh</surname> <given-names>A. K.</given-names>
</name>
<name>
<surname>Varaprasad</surname> <given-names>K. S.</given-names>
</name>
<name>
<surname>Venkateswaran</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Conservation costs of plant genetic resources for food and agriculture: Seed genebanks</article-title>. <source>Agric. Res.</source> <volume>1</volume> (<issue>3</issue>), <fpage>223</fpage>&#x2013;<lpage>239</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s40003-012-0029-3</pub-id>
</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singh</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Raupp</surname> <given-names>W. J.</given-names>
</name>
<name>
<surname>Sehgal</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Arora</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Tiwari</surname> <given-names>V.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>Efficient curation of genebanks using next generation sequencing reveals substantial duplication of germplasm accessions</article-title>. <source>Sci. Rep.</source> <volume>9</volume> (<issue>1</issue>), <fpage>650</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-018-37269-0</pub-id>
</citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Slater</surname> <given-names>G. S. C.</given-names>
</name>
<name>
<surname>Birney</surname> <given-names>E.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Automated generation of heuristics for biological sequence comparison</article-title>. <source>BMC Bioinf.</source> <volume>6</volume> (<issue>1</issue>), <fpage>31</fpage>. doi: <pub-id pub-id-type="doi">10.1186/1471-2105-6-31</pub-id>
</citation>
</ref>
<ref id="B69">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Smit</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Hubley</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2008-2015</year>) <source>RepeatModeler open-1.0</source>. Available at: <uri xlink:href="http://www.repeatmasker.org">http://www.repeatmasker.org</uri>.</citation>
</ref>
<ref id="B70">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Smit</surname> <given-names>A. F. A.</given-names>
</name>
<name>
<surname>Hubley</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Green</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2013-2015</year>) <source>RepeatMasker open-4.0</source>. Available at: <uri xlink:href="http://www.repeatmasker.org">http://www.repeatmasker.org</uri>.</citation>
</ref>
<ref id="B71">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Florea</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Rcorrector: Efficient and accurate error correction for illumina RNA-seq reads</article-title>. <source>GigaScience</source> <volume>4</volume> (<issue>1</issue>):<fpage>48</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13742-015-0089-y</pub-id>
</citation>
</ref>
<ref id="B72">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stanke</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Morgenstern</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>AUGUSTUS: A web server for gene prediction in eukaryotes that allows user-defined constraints</article-title>. <source>Nucleic Acids Res.</source> <volume>33</volume> (<supplement>suppl_2</supplement>), <fpage>W465</fpage>&#x2013;<lpage>W467</lpage>. doi: <pub-id pub-id-type="doi">10.1093/nar/gki458</pub-id>
</citation>
</ref>
<ref id="B73">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Udall</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Long</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Ramaraj</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Conover</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Grover</surname> <given-names>C. E.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>The genome sequence of <italic>Gossypioides kirkii</italic> illustrates a descending dysploidy in plants</article-title>. <source>Front. Plant Sci.</source> <volume>10</volume> (<issue>1541</issue>). doi: <pub-id pub-id-type="doi">10.3389/fpls.2019.01541</pub-id>
</citation>
</ref>
<ref id="B74">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vi&#x17e;intin</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Javornik</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Bohanec</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Genetic characterization of selected <italic>Trifolium</italic> species as revealed by nuclear DNA content and ITS rDNA region analysis</article-title>. <source>Plant Sci.</source> <volume>170</volume> (<issue>4</issue>), <fpage>859</fpage>&#x2013;<lpage>866</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.plantsci.2005.12.007</pub-id>
</citation>
</ref>
<ref id="B75">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wellenreuther</surname> <given-names>M.</given-names>
</name>
<name>
<surname>M&#xe9;rot</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Berdan</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Bernatchez</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Going beyond SNPs: The role of structural genomic variants in adaptive evolution and species diversification</article-title>. <source>Mol. Ecol.</source> <volume>28</volume> (<issue>6</issue>), <fpage>1203</fpage>&#x2013;<lpage>1209</lpage>. doi: <pub-id pub-id-type="doi">10.1111/mec.15066</pub-id>
</citation>
</ref>
<ref id="B76">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Williams</surname> <given-names>W. M.</given-names>
</name>
<name>
<surname>Ellison</surname> <given-names>N. W.</given-names>
</name>
<name>
<surname>Ansari</surname> <given-names>H. A.</given-names>
</name>
<name>
<surname>Verry</surname> <given-names>I. M.</given-names>
</name>
<name>
<surname>Hussain</surname> <given-names>S. W.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Experimental evidence for the ancestry of allotetraploid trifolium repens and creation of synthetic forms with value for plant breeding</article-title>. <source>BMC Plant Biol.</source> <volume>12</volume>, <fpage>55</fpage>&#x2013;<lpage>55</lpage>. doi: <pub-id pub-id-type="doi">10.1186/1471-2229-12-55</pub-id>
</citation>
</ref>
<ref id="B77">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Tammi</surname> <given-names>M. T.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>CNV-seq, a new method to detect copy number variation using high-throughput sequencing</article-title>. <source>BMC Bioinf.</source> <volume>10</volume> (<issue>1</issue>), <fpage>80</fpage>. doi: <pub-id pub-id-type="doi">10.1186/1471-2105-10-80</pub-id>
</citation>
</ref>
<ref id="B78">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Young</surname> <given-names>N. D.</given-names>
</name>
<name>
<surname>Debell&#xe9;</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Oldroyd</surname> <given-names>G. E. D.</given-names>
</name>
<name>
<surname>Geurts</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Cannon</surname> <given-names>S. B.</given-names>
</name>
<name>
<surname>Udvardi</surname> <given-names>M. K.</given-names>
</name>
<etal/>
</person-group>. (<year>2011</year>). <article-title>The medicago genome provides insight into the evolution of rhizobial symbioses</article-title>. <source>Nature</source> <volume>480</volume>, <fpage>520</fpage>&#x2013;<lpage>524</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nature10625</pub-id>
</citation>
</ref>
<ref id="B79">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Milec</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Bayer</surname> <given-names>P. E.</given-names>
</name>
<name>
<surname>Vr&#xe1;na</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Dole&#x17e;el</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Edwards</surname> <given-names>D.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>Large-Scale structural variation detection in subterranean clover subtypes using optical mapping</article-title>. <source>Front. Plant Sci.</source> <volume>9</volume>. doi: <pub-id pub-id-type="doi">10.3389/fpls.2018.00971</pub-id>
</citation>
</ref>
<ref id="B80">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zerbino</surname> <given-names>D. R.</given-names>
</name>
<name>
<surname>Birney</surname> <given-names>E.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Velvet: Algorithms for <italic>de novo</italic> short read assembly using de bruijn graphs</article-title>. <source>Genome Res.</source> <volume>18</volume> (<issue>5</issue>), <fpage>821</fpage>&#x2013;<lpage>829</lpage>. doi: <pub-id pub-id-type="doi">10.1101/gr.074492.107</pub-id>
</citation>
</ref>
<ref id="B81">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zohary</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Heller</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>1984</year>). <article-title>The genus trifolium. Jerusalem, Israel academy of sciences and humanities</article-title>.</citation>
</ref>
</ref-list>
</back>
</article>