<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">755564</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2021.755564</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Data Report</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Chromosome-Level Genome Assembly of the Rare and Endangered Tropical Plant <italic>Speranskia yunnanensis</italic> (Euphorbiaceae)</article-title>
<alt-title alt-title-type="left-running-head">Yuan et&#x20;al.</alt-title>
<alt-title alt-title-type="right-running-head">Genome Assembly of <italic>Speranskia yunnanensis</italic>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Yuan</surname>
<given-names>Guofang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="FN1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1439024/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Tan</surname>
<given-names>Shufang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="FN1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1434294/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Dandan</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Yang</surname>
<given-names>Yongzhi</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1021890/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Tian</surname>
<given-names>Bin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Key Laboratory for Forest Resources Conservation and Utilization in the Southwest Mountains of China, Ministry of Education, Southwest Forestry University</institution>, <addr-line>Kunming</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>State Key Laboratory of Grassland Agro- Ecosystems, Institute of Innovation Ecology and School of Life Sciences, Lanzhou University</institution>, <addr-line>Lanzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>CAS Key Laboratory for Plant Diversity and Biogeography, Kunming Institute of Botany, Chinese Academy of Sciences</institution>, <addr-line>Kunming</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/57034/overview">Mahendar Thudi</ext-link>, Dr. Rajendra Prasad Central Agricultural University, India</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/336267/overview">Jarkko Saloj&#xe4;rvi</ext-link>, Nanyang Technological University, Singapore</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/337150/overview">Li Guo</ext-link>, Peking University Institute of Advanced Agricultural Sciences, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Yongzhi Yang, <email>yangyongzhi2008@gmail.com</email>; Bin Tian, <email>tianbin@swfu.edu.cn</email>
</corresp>
<fn fn-type="equal" id="FN1">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this&#x20;work</p>
</fn>
<fn fn-type="other">
<p>This article was submitted to Plant Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>24</day>
<month>01</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>12</volume>
<elocation-id>755564</elocation-id>
<history>
<date date-type="received">
<day>11</day>
<month>08</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>13</day>
<month>12</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Yuan, Tan, Wang, Yang and Tian.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Yuan, Tan, Wang, Yang and Tian</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these&#x20;terms.</p>
</license>
</permissions>
<kwd-group>
<kwd>genome assembly</kwd>
<kwd>chromosome-level genome</kwd>
<kwd>phylogenetic relationships</kwd>
<kwd>evolution</kwd>
<kwd>
<italic>Speranskia yunnanensis</italic>
</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<p>
<italic>Speranskia yunnanensis</italic> S. M. Hwang is an endangered shrub narrowly distributed in tropical regions, and its populations are gradually shrinking. We assembled and annotated the genome of <italic>S. yunnanensis</italic> at the chromosome level by combining Nanopore sequencing, Illumina HiSeq sequencing and Hi-C technology. The final genome assembly was ~417.65 Mb, with a contig N50 value of 12.52 Mb, and 408.62 Mb (97.84%) of which could be grouped into seven pseudochromosomes. Approximately 69.11% of the assembly was identified as repetitive elements, and 25,467 protein-coding genes were annotated. Based on the 1,517 single-copy orthologous genes, and 751 expanded and 1,645 contracted gene families among the 16,389 gene families in <italic>S. yunnanensis</italic>, a phylogenetic tree was further built. The high-quality, annotated, and chromosome-level genome of <italic>S. yunnanensis</italic> will present an important source of data for future research on the evolution of Euphorbiaceae genomes, and provide genomic resources toward studies on speciation, local adaptation, as well as conservation genomics of the ecologically important genus <italic>Speranskia</italic>.</p>
<sec id="s1">
<title>Introduction</title>
<p>
<italic>Speranskia</italic> Baill., a small genus within the tribe Ricineae subfamily Acalyphoideae of Euphorbiaceae, is endemic in China (<xref ref-type="bibr" rid="B18">Hwang, 1989</xref>; <xref ref-type="bibr" rid="B35">Webster, 2014</xref>). Three members are recognized in <italic>Speranskia</italic>, namely <italic>S. cantonensis</italic> (Hance) Pax et Hoffm, <italic>S. tuberculata</italic> (Bunge) Baill., and <italic>S. yunnanensis</italic> Hwang. Of them, <italic>S. cantonensis</italic> and <italic>S. tuberculata</italic> are widely distributed from southwestern to northern China, whereas <italic>S. yunnanensis</italic> maintains a narrow distribution in tropical Yunnan and is found only in three small and fragmented natural populations (<xref ref-type="bibr" rid="B18">Hwang, 1989</xref>). As a member of the castor tribe (Ricineae) and a sister taxon to the castor bean (<italic>Ricinus</italic>) (<xref ref-type="bibr" rid="B35">Webster, 2014</xref>), <italic>Speranskia</italic> is important for us to infer the origin and evolution of castor bean as well as the evolution of ricin proteins. However, its genome composition and phylogenetic position within Ricineae are largely unknown. Here, we conducted a series of genomic analyses on <italic>S. yunnanensis</italic> (2n &#x3d; 14), including chromosome-level assembly, annotations, phylogenetic reconstructions, gene family expansion and contraction analyses, divergence time estimation, and <italic>Ks</italic> analysis. The genome assembly and resources produced in this study will provide important insights as well as resource for future study in <italic>S. yunnanensis</italic> to facilitate conservation and in the genus <italic>Speranskia</italic> in general.</p>
</sec>
<sec sec-type="results|discussion" id="s2">
<title>Results and Discussion</title>
<sec id="s2-1">
<title>Library Preparation and Whole-Genome Sequencing</title>
<p>We sequenced the genome of <italic>S. yunnanensis</italic> S. M. Hwang using a combination of Illumina short-read sequencing, Oxford Nanopore Technologies, and Hi-C sequencing technologies. The short-insert library of 400&#xa0;bp was constructed, and a total of 20.57&#xa0;Gb of raw data was generated using HiSeq X ten of the Illumina platform. Low-quality reads were removed, and duplication reads were filtered to obtain 18.73&#xa0;Gb of clean data (91.05%). Additionally, we obtained 31.35&#xa0;Gb of pass reads from one flow cell of the PromethION sequencer for genome assembly. Furthermore, the Hi-C library was constructed and sequenced with paired-end (PE) 150&#xa0;bp reads for chromosome-level scaffolding. Finally, approximately 65.16&#xa0;Gb of Hi-C data was generated.</p>
</sec>
<sec id="s2-2">
<title>Estimating <italic>S. yunnanensis</italic> Genome Size</title>
<p>Before genome assembly, a genome survey was performed to assess the genome size based on 18.73&#xa0;Gb of Illumina clean data. Using a 21-mer analysis method, the major peak was located around a k-mer depth of 25, and the other clear peak was located at half of the expected depth inferred to be a heterozygous peak. The final predicted genome size of <italic>S. yunnanensis</italic> was &#x223c;452.54&#xa0;Mb with a heterozygosity rate of &#x223c;0.62% and a repeat ratio of 53.4% (<xref ref-type="sec" rid="s9">Supplementary Figure&#x20;S1</xref>).</p>
</sec>
<sec id="s2-3">
<title>De Novo Genome Assembly and Pseudochromosome Construction</title>
<p>Following self-error correction, Nanopore long reads were initially assembled into contigs with NextDenovo, which produced a preliminary assembly length of 414.61&#xa0;Mb and a contig N50 size of 12.42&#xa0;Mb (<xref ref-type="sec" rid="s9">Supplementary Table S1</xref>). To further optimize the assembly, the preliminary assembly was polished using NextPolish. The final assembly of 417.65&#xa0;Mb (92.29% of the predicted genome size) was obtained with a contig N50 size of 12.52&#xa0;Mb (<xref ref-type="sec" rid="s9">Supplementary Table S1</xref>). Based on 65.16&#xa0;Gb of Hi-C data (&#x223c;163 &#xd7; coverage), we obtained seven pseudochromosomes using JUICER and 3D-DNA programs. In total, 97.85% (408.62&#xa0;Mb) of the assembly was anchored and oriented on pseudochromosomes with a scaffold N50 of 61.04&#xa0;Mb, ranging from 45.31 to 68.49&#xa0;Mb in length (<xref ref-type="sec" rid="s9">Supplementary Figure S2</xref> and <xref ref-type="sec" rid="s9">Supplementary Table&#x20;S2</xref>).</p>
<p>To evaluate our assembly, we first used the Benchmarking Universal Single-Copy Orthologs (BUSCO) to assess assembly completeness, and 98.3% Embryophyta-conserved genes could be completely predicted in our assembly (<xref ref-type="sec" rid="s9">Supplementary Table S3</xref>). We then estimated the base accuracy of the assembly by mapping Illumina reads. In total, 94.24% of Illumina data could be mapped to the genome and 94.01 and 91.33% of the assembled genome sequence could be covered by at least 4- and 10-fold, respectively (<xref ref-type="sec" rid="s9">Supplementary Table S4</xref>). Furthermore, GC content Poisson distributions presented a complete and high-quality genome assembly (<xref ref-type="sec" rid="s9">Supplementary Figure&#x20;S3</xref>).</p>
</sec>
<sec id="s2-4">
<title>Repeat Annotation of the Genome Assembly</title>
<p>A total of &#x223c;288.65&#xa0;Mb repetitive elements (accounting for approximately 69.11% of the genome) were identified via two methods on the basis of <italic>de novo</italic> and homology-based predictions. Among these repetitive sequences, tandem and interspersed repeats were approximately 49.43&#xa0;Mb (11.84% of the genome) and 212.22&#xa0;Mb (50.81% of the genome), respectively. Additionally, retrotransposons and DNA transposons were primary components of the interspersed repeats accounting for 43.80 and 7.02% of the genome, respectively. The dominant type of retrotransposons was Long Terminal Retrotransposons (LTRs), which accounted for approximately 42.19% of the genome (35.00% LTR/Gypsy and 6.70% LTR/Copia retrotransposons) (<xref ref-type="sec" rid="s9">Supplementary Table&#x20;S5</xref>).</p>
</sec>
<sec id="s2-5">
<title>Gene Prediction and Noncoding RNA Annotation</title>
<p>We applied multiple gene model prediction methods to accurately predict gene sets in the <italic>S. yunnanensis</italic> genome, including <italic>de novo</italic>, homology-based, and transcriptome-based methods. Analyses showed that a total of 25,467&#x20;protein-coding genes were predicted with an average gene length of 2,849.31&#xa0;bp, an average coding DNA sequence (CDS) size of 1,194.81&#xa0;bp, and average exons per gene of 5.24 (<xref ref-type="table" rid="T1">Table&#x20;1</xref> and <xref ref-type="sec" rid="s9">Supplementary Table S6</xref>). Moreover, 1,375 (96.1%) BUSCO genes were completely matched to our predicted <italic>S. yunnanensis</italic> gene sets, suggesting high completeness and accuracy of protein-coding genes (<xref ref-type="sec" rid="s9">Supplementary Table S7</xref>). Overall, a total of 23,078 (90.62%) genes could be assigned to functional annotation within the public protein databases: TrEMBL (90.28%), Swiss-Prot (71.37%), NR (90.52%), EggNOG (42.55%), Gene Ontology (GO) (64.37%), Kyoto Encyclopedia of Genes and Genomes (KEGG) (25.24%), and InterPro (55.14%) (<xref ref-type="sec" rid="s9">Supplementary Table S8</xref>). Additionally, noncoding RNAs were also identified, including 469 transfer RNAs (tRNAs), 617 ribosomal RNAs (rRNAs), 117 microRNAs (miRNAs), and 718 small nuclear RNAs (snRNAs) (<xref ref-type="sec" rid="s9">Supplementary Table&#x20;S9</xref>).</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Assembly and annotation summary of <italic>Speranskia yunnanensis</italic> genome.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Assembly feature</th>
<th align="left"/>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Genome size (bp)</td>
<td align="center">417,645,011</td>
</tr>
<tr>
<td align="left">Longest contig (bp)</td>
<td align="center">42,011,156</td>
</tr>
<tr>
<td align="left">N50 of contig (bp)</td>
<td align="center">12,521,520</td>
</tr>
<tr>
<td align="left">GC ratio (%)</td>
<td align="center">32.95</td>
</tr>
<tr>
<td align="left">BUSCO score of assembly (%)</td>
<td align="center">98.3</td>
</tr>
<tr>
<td align="left">Number of genes</td>
<td align="center">25,467</td>
</tr>
<tr>
<td align="left">Percentage of repetitive sequence</td>
<td align="center">69.11</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-6">
<title>Gene Families and Phylogenetic Analysis</title>
<p>For phylogenetic analysis and discerning quantities of potential orthologous gene families, genes from 11 species, including <italic>Arabidopsis thaliana, Hevea brasiliensis, Jatropha curcas, Manihot esculenta, Medicago truncatula, Oryza sativa, Prunus persica, Populus trichocarpa, Ricinus communis,</italic> and <italic>Vitis vinifera</italic>, were clustered into gene families. In total, 20,851&#x20;<italic>S. yunnanensis</italic> genes (81.87%) were clustered into 16,389 gene families, of which 524 were unique to <italic>S. yunnanensis</italic> (<xref ref-type="sec" rid="s9">Supplementary Table S10</xref>). Furthermore, <italic>S. yunnanensis</italic> shared 12,300 gene families with four other species (<italic>H. brasiliensis, M. esculenta</italic>, J. <italic>curcas</italic> and <italic>R. communis</italic>) and contained 635 unique gene families (<xref ref-type="sec" rid="s9">Supplementary Figure S4</xref>). We identified and selected a total of 1,517&#x20;single-copy orthologous gene families for phylogenetic analyses and divergence time estimation. It was showed that <italic>S. yunnanensis</italic> and <italic>R. communis</italic> were closely related species in the Euphorbiaceae that diverged approximately 26.43 million years ago (Mya) (<xref ref-type="fig" rid="F1">Figure&#x20;1</xref> and <xref ref-type="sec" rid="s9">Supplementary Figure&#x20;S5</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Phylogenetic tree of <italic>S. yunnanensis</italic> and 10 other species. Gene family expansions (&#x2b;) and contractions (&#x2212;) are indicated by red and blue, respectively. Black numbers represent divergence time between species. The numbers of gene families, clustered genes, and all predicted genes are indicated next to each species. Calibration point is marked by red star.</p>
</caption>
<graphic xlink:href="fgene-12-755564-g001.tif"/>
</fig>
<p>We further discovered 751 expanded and 1,646 contracted gene families in <italic>S. yunnanensis</italic>. Among them, 85 gene families exhibited remarkable expansion and 30 gene families showed dramatic contraction (<xref ref-type="sec" rid="s9">Supplementary Table S11</xref>). The enrichment results showed that the expanded genes were mainly related to GO terms in ion binding, such as zinc and iron (GO:0008270, GO:0046872, GO:0043169, GO:0005506, and GO:0043167), nuclease activity (GO:0004521, GO:0016891, and GO:0004540), the terpenoid synthesis process (GO:0046246 and GO:0016114) (<xref ref-type="sec" rid="s9">Supplementary Table S12</xref>), and KEGG maps of metabolic synthesis of terpenoids (K09109, K00909, K00902, and K00904) (<xref ref-type="sec" rid="s9">Supplementary Table S13</xref>). However, the contracted genes were mainly related to the ribosomal composition process (GO:0003735, GO:0022625, GO:0044391, GO:0015934, GO:0022626, GO:0005840, GO:0042788, and GO:0005844).</p>
</sec>
<sec id="s2-7">
<title>Whole-Genome Duplication Analysis</title>
<p>To investigate polyploidization events within <italic>S. yunnanensis</italic>, we performed a comparative analysis of the genome sequences from <italic>S. yunnanensis</italic> and the other three species (<italic>J.&#x20;curcas, R. communis</italic> and <italic>V. vinifera</italic>). By measuring the <italic>Ks</italic> rate of orthologous gene pairs, we found <italic>S. yunnanensis</italic> shared similar <italic>Ks</italic> distributions (with a peak <italic>Ks</italic> value of 1.52) with <italic>J.&#x20;curcas</italic> and <italic>R. communis</italic>, indicating a shared, ancient polyploidy event (&#x3b3;) of <italic>S. yunnanensis</italic> with <italic>V. vinifera</italic> whereas no additional, recent species-specific WGD event in the species (<xref ref-type="sec" rid="s9">Supplementary Figure&#x20;S6</xref>).</p>
</sec>
</sec>
<sec sec-type="materials|methods" id="s3">
<title>Materials and Methods</title>
<sec id="s3-1">
<title>Sample Processing and Whole-Genome Sequencing</title>
<p>The natural plants of <italic>S. yunnanensis</italic> were collected from Zhenkang County, Yunnan Province, China. Genomic DNA was extracted from fresh young leaves using the QIAGEN Genomic reagent kit, according to the manufacturer&#x2019;s protocols. On the basis of adequate quality sample, a paired-end library with short-insert sizes of 400&#xa0;bp was prepared for sequencing on the HiSeq X Ten PE150 platform using standard Illumina instructions. Following the Nanopore library construction protocol, a Nanopore library was constructed and the long-read data was generated using the PromethION sequencer (Oxford Nanopore Technologies, United&#x20;Kingdom). Sequencing adapters were removed, and reads of low quality and short length were filtered out. Freshly harvested leaves were used to construct the Hi-C libraries. First, the intranuclear chromatin was fixed with formaldehyde to facilitate covalent bond formation. Subsequently, MboI, a restriction endonuclease, was used to digest cross-linked DNA, after which sticky DNA ends were repaired with biotin-marked nucleotides and the resulting blunt ends were ligated together with DNA ligase. Proteins were then removed with proteases. To release DNA molecules from cross-links, purified DNA was randomly interrupted to fragments with an average size of 300&#x20;bp using ultrasound and attached to adapters. Biotin-labeled DNA fragments were ultimately captured and enriched using streptavidin beads to construct paired-end sequencing libraries, which were then sequenced on the Illumina HiSeq platform to obtain 2&#x20;&#xd7; 150&#xa0;bp Hi-C raw reads. Finally, to aid gene annotation, we performed RNA sequencing for fresh tissues of the leaf, root, seed and stem from the same plant using an Illumina HiSeq 2500 platform.</p>
</sec>
<sec id="s3-2">
<title>Estimating of <italic>S. yunnanensis</italic> Genome Size</title>
<p>We first filtered Illumina reads using FASTP v0.20.0 (<xref ref-type="bibr" rid="B7">Chen et&#x20;al., 2018</xref>) with default parameters. Clean reads were analyzed by KMC v3.1.1 (<xref ref-type="bibr" rid="B22">Marek et&#x20;al., 2017</xref>) to generate the k-mer depth distribution with a k-mer size of 21&#xa0;bp, and the final result was plotted in GenomeScope v2.0 (<ext-link ext-link-type="uri" xlink:href="http://qb.cshl.edu/genomescope/genomescope2.0/">http://qb.cshl.edu/genomescope/genomescope2.0/</ext-link>) (<xref ref-type="bibr" rid="B30">Ranallo-Benavidez et&#x20;al., 2020</xref>).</p>
</sec>
<sec id="s3-3">
<title>Genome Assembly and Pseudochromosome Construction</title>
<p>High-quality controlled Oxford Nanopore Technologies long reads were capitalized on assembly using NextDenovo v2.4.0 (<ext-link ext-link-type="uri" xlink:href="https://github.com/Nextomics/NextDenovo">https://github.com/Nextomics/NextDenovo</ext-link>) with a read cutoff length of 4&#xa0;kb. A preliminary assembly was achieved using NextGraph, a subprogram of NextDenovo. Then, iterative polishing was performed repetitively using NextPolish v1.3.1 (<xref ref-type="bibr" rid="B16">Hu et&#x20;al., 2019</xref>). At this stage, Oxford Nanopore Technologies long reads and Illumina short-clean reads were used repeatedly for 10 rounds of genome correction after finalizing the preassembly. The completeness of genome assembly was assessed by BUSCO v3.0.2 (<xref ref-type="bibr" rid="B31">Sim&#xe3;o et&#x20;al., 2016</xref>) with the embryophyta_odb10 databases.</p>
<p>For Hi-C sequence data, we also initially filtered out low-quality reads using FASTP with default parameters. Clean Hi-C data were then analyzed using JUICER v1.6.2 (<xref ref-type="bibr" rid="B10">Durand et&#x20;al., 2016</xref>) with default parameters, including softlinking and indexing the genome sequence in the references/folder, creating restriction enzyme proposed cutting sites, and initializing Hi-C reads. Subsequently, the resulting file was utilized as input into the 3D-DNA program (<xref ref-type="bibr" rid="B8">Dudchenko et&#x20;al., 2017</xref>) with the parameters -r 2&#x20;-q 1 for further analysis. We used 3D-DNA to elaborately optimize the ordering and orientation of each clustered group, and scaffold the genome to produce a chromosomal level assembly. Finally, an interaction heatmap was corrected and plotted using JuiceBox v1.11.9 (<xref ref-type="bibr" rid="B9">Durand et&#x20;al., 2016</xref>).</p>
</sec>
<sec id="s3-4">
<title>Repeat Annotation</title>
<p>Repetitive elements, including tandem repeats and interspersed repeats, were predicted in the <italic>S. yunnanensis</italic> genome. Tandem Repeats Finder (TRF) v4.09 (<xref ref-type="bibr" rid="B4">Benson, 1999</xref>) was first applied to annotate tandem repeats. For interspersed repeats, both homology-based and <italic>de novo</italic> approaches were mainly used. RepeatMasker v4.0.7 (<xref ref-type="bibr" rid="B34">Tarailo-Graovac and Chen, 2009</xref>) and RepeatProteinMasker were used to identify interspersed repeats on the basis of homology alignment of the input <italic>S. yunnanensis</italic> genome sequence against Repbase v16.10 (<xref ref-type="bibr" rid="B3">Bao et&#x20;al., 2015</xref>). RepeatModeler v5.8.8 (<xref ref-type="bibr" rid="B32">Smit and Hubley, 2008</xref>) was utilized to construct the repeat library, which comprised a repeat consensus database with classification information. Finally, RepeatMasker was employed to generate the <italic>de novo</italic> predictions.</p>
</sec>
<sec id="s3-5">
<title>Gene Structure Annotation</title>
<p>Protein-coding annotations were predicted using a combination of <italic>de novo</italic>, homology-based, and transcriptome-based approaches based on the repeat masked genome. For <italic>de novo</italic> prediction, AUGUSTUS v3.2.3 (<xref ref-type="bibr" rid="B33">Stanke et&#x20;al., 2008</xref>), GENSCAN (<xref ref-type="bibr" rid="B6">Burge and Karlin, 1998</xref>), and GlimmerHMM v3.0.4 (<xref ref-type="bibr" rid="B25">Majoros et&#x20;al., 2004</xref>) were applied on the basis of the models trained with CDS data from <italic>A. thaliana</italic>. GeMoMa v1.7.1 (<xref ref-type="bibr" rid="B21">Keilwagen et&#x20;al., 2016</xref>) was used for homology prediction, with protein sequences from <italic>A. thaliana, H. brasiliensis</italic>, <italic>J.&#x20;curcas</italic>, <italic>M. esculenta</italic>, <italic>P. persica</italic>, <italic>P. trichocarpa, R. communis</italic>, and <italic>V. vinifera</italic>. For transcriptome-based prediction, after trimming low-quality bases and adapter sequences with Trimmomatic v0.39 (<xref ref-type="bibr" rid="B5">Bolger et&#x20;al., 2014</xref>), nonredundant full-length transcriptomes from the <italic>de novo</italic> assembly using TRINITY v2.9.1 (<xref ref-type="bibr" rid="B13">Haas et&#x20;al., 2013</xref>) were subsequently aligned to the genome to resolve gene structure using PASA v2.3.3 (<xref ref-type="bibr" rid="B12">Haas et&#x20;al., 2003</xref>) with parameters -c RunPASA.config--TRANSDECODER -C -r -R --ALIGNERS blat, gmap. In addition, EvidenceModeler (EVM) v1.1.1 (<xref ref-type="bibr" rid="B14">Haas et&#x20;al., 2008</xref>) was used to generate the final consensus set of the gene model using the previous three approaches. Ultimately, the completeness of the genome assembly was further assessed using BUSCO based on the embryophyta_odb10 databases.</p>
</sec>
<sec id="s3-6">
<title>Gene Functional Annotation</title>
<p>Functional annotations of protein-coding genes were carried out using BLASTP (e-value 1e&#x2212;5) v2.2.26 (<xref ref-type="bibr" rid="B1">Altschul et&#x20;al., 1997</xref>) against publicly available databases including the Swiss-Prot (<xref ref-type="bibr" rid="B2">Bairoch and Apweiler, 2000</xref>), TrEMBL (<xref ref-type="bibr" rid="B2">Bairoch and Apweiler, 2000</xref>), NR (<xref ref-type="bibr" rid="B28">Pruitt et&#x20;al., 2007</xref>) and eggNOG (<xref ref-type="bibr" rid="B17">Huerta-Cepas et&#x20;al., 2015</xref>). Protein motifs and domains were annotated using InterProScan v5.30&#x2212;69.0 (<xref ref-type="bibr" rid="B19">Jones et&#x20;al., 2014</xref>) by searching ProDom, SMART, SUPERFAMILY and PRINTS. Potential pathways of each gene were found in the KEGG Automatic Annotation Server (<ext-link ext-link-type="uri" xlink:href="https://www.genome.jp/kegg/kaas/">https://www.genome.jp/kegg/kaas/</ext-link>) using the KEGG database.</p>
</sec>
<sec id="s3-7">
<title>Noncoding RNA Annotation</title>
<p>Noncoding RNA genes, including tRNA, rRNA, miRNA, and snRNA, were predicted in the assembled genome. tRNA genes were predicted using tRNAscan-SE v1.3.1 (<xref ref-type="bibr" rid="B24">Lowe and Eddy, 1997</xref>) with eukaryote parameters, and rRNA with high conservation were predicted by aligning reads to the <italic>Arabidopsis</italic> template rRNA sequences using BLASTN (<xref ref-type="bibr" rid="B1">Altschul et&#x20;al., 1997</xref>), with an e value of 1e&#x2212;5. Additionally, INFERNAL (<xref ref-type="bibr" rid="B26">Nawrocki and Eddy, 2013</xref>) was used to predict miRNA and snRNA genes on the basis of the Rfam database (<xref ref-type="bibr" rid="B11">Griffiths-Jones et&#x20;al., 2005</xref>).</p>
</sec>
<sec id="s3-8">
<title>Gene Families and Phylogenetic Analysis</title>
<p>To reveal <italic>S. yunnanensis</italic> genome evolution, protein-coding gene sequences from 11 species were selected for phylogenetic analysis: <italic>A. thaliana, H. brasiliensis, J.&#x20;curcas, M. esculenta, M. truncatula, O. sativa, P. persica, P. trichocarpa, R. communis,</italic> and <italic>V. vinifera.</italic> First, an all-vs-all BLASTP (e-value 1e&#x2212;5) was applied to calculate gene similarities. And the paralogs and orthologs were respectively clustered using OrthoMCL v2.0.9 (<xref ref-type="bibr" rid="B23">Li et&#x20;al., 2003</xref>). Second, single-copy orthologous genes were extracted from the OrthoMCL clustering results and aligned using MAFFT v7.453 (<xref ref-type="bibr" rid="B20">Katoh and Standley, 2013</xref>) with default parameters. The protein sequence alignments were converted into CDS alignments, which were then concatenated into a supergene for phylogenetic analysis. IQ-TREE v1.6.12 (<xref ref-type="bibr" rid="B27">Lam-Tung et&#x20;al., 2015</xref>) was used to construct a maximum likelihood (ML) tree with the MFP model and a bootstrap of 1000. Subsequently, the ML tree was input into BaSeml v4.0 and MCMCTree v4.0 (<xref ref-type="bibr" rid="B36">Yang, 2007</xref>) to estimate the nucleotide substitution rates and the divergence times, respectively. The approximate divergence times between <italic>A. thaliana</italic> and <italic>O. sativa</italic> (115&#x2013;308&#xa0;Mya), as well as between <italic>A. thaliana</italic> and <italic>V. vinifera</italic> (107&#x2013;135&#xa0;Mya; <ext-link ext-link-type="uri" xlink:href="http://http:/%20/www.timetree.org/">http://www.timetree.org/</ext-link>) were used as calibrators in time estimation. The MCMCTree parameters were set as follows: model &#x3d; 7, BDparas &#x3d; 110, kappa_gamma &#x3d; 62, alpha_gamma &#x3d; 11, burnin &#x3d; 500,000, sampfreq &#x3d; 5,000, nsample &#x3d; 20,000. Based on the divergence times estimated and the gene families identified via OrthoMCL, the expansion and contraction of gene families were predicted using CAF&#xc9; v3.1 (<xref ref-type="bibr" rid="B15">Han et&#x20;al., 2013</xref>) under a random birth-and-death&#x20;model.</p>
</sec>
<sec id="s3-9">
<title>Whole-Genome Duplication Analysis</title>
<p>Protein-coding sequences within one genome or between two different genomes were aligned using BLASTP with an e-value cutoff of 1e&#x2212;5. Syntenic blocks and synonymous nucleotide substitutions (<italic>Ks</italic>) were determined from protein sequence alignments based on the detected homologous gene pairs using WGDI (<ext-link ext-link-type="uri" xlink:href="https://github.com/SunPengChuan/wgdi">https://github.com/SunPengChuan/wgdi</ext-link>). We further filtered the tandem duplicated gene pairs. WGD and speciation events were inferred from paralogous and orthologous pairs of <italic>Ks</italic> distribution peaks, respectively.</p>
</sec>
</sec>
</body>
<back>
<sec id="s4">
<title>Data Availability Statement</title>
<p>Whole-genome sequence reads (including the Nanopore long reads, NGS short reads, and Hi-C reads) used in this study have been deposited in the NCBI database under the BioProject accession number PRJNA744706. The genome assembly file and genome annotation files (repeat annotation and gene structure annotation) are available at Figshare (<ext-link ext-link-type="uri" xlink:href="https://figshare.com/articles/dataset/Speranskia_yunnanensis/15057930">https://figshare.com/articles/dataset/Speranskia_yunnanensis/15057930</ext-link>).</p>
</sec>
<sec id="s5">
<title>Author Contributions</title>
<p>BT and YY conceived and managed the project. BT collected samplings and completed species identification, as well as prepared the DNA samples and performed the sequencing. ST and DW assembled the genome, performed the gene annotation, gene family, and evolutionary analyses. GY and BT wrote the manuscript with the assistance from YY. All authors have read and approved the final version of the manuscript.</p>
</sec>
<sec id="s6">
<title>Funding</title>
<p>This study was financially supported by the National Natural Science Foundation of China (41861008) and the Ten-thousand Talents Program of Yunnan Province (YNWR-QNBJ-2020). We received support for computational work from the Big Data Computing Platform for Western Ecological Environment and Regional Development and Supercomputing Center of Lanzhou University.</p>
</sec>
<sec sec-type="COI-statement" id="s7">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s8">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>We are thankful to Mou Yin, Wenjie Mou, Zhenyue Wang, Xin Du, Hongyin Hu, Ying Li, Shangzhe Zhang and Zeyu Zheng for assisting on data analyses in this study. We also give our thanks to Haiyang Wu and Pei Zhang for their help during the field survey.</p>
</ack>
<sec id="s9">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2021.755564/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2021.755564/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altschul</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Madden</surname>
<given-names>T. L.</given-names>
</name>
<name>
<surname>Schffer</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Webb</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>1997</year>). <article-title>Gapped Blast and Psi-Blast: a New Generation of Protein Database Search Programs</article-title>. <source>Nucleic Acids Res.</source> <volume>25</volume> (<issue>17</issue>), <fpage>3389</fpage>&#x2013;<lpage>3402</lpage>. <pub-id pub-id-type="doi">10.1093/nar/25.17.3389</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bairoch</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Apweiler</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>The SWISS-PROT Protein Sequence Database and its Supplement TrEMBL in 2000</article-title>. <source>Nucleic Acids Res.</source> <volume>28</volume> (<issue>1</issue>), <fpage>45</fpage>&#x2013;<lpage>48</lpage>. <pub-id pub-id-type="doi">10.1093/nar/24.1.21</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Kojima</surname>
<given-names>K. K.</given-names>
</name>
<name>
<surname>Kohany</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Repbase Update, a Database of Repetitive Elements in Eukaryotic Genomes</article-title>. <source>Mobile DNA</source> <volume>6</volume>, <fpage>11</fpage>. <pub-id pub-id-type="doi">10.1186/s13100-015-0041-9</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Benson</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Tandem Repeats Finder: a Program to Analyze DNA Sequences</article-title>. <source>Nucleic Acids Res.</source> <volume>27</volume> (<issue>2</issue>), <fpage>573</fpage>&#x2013;<lpage>580</lpage>. <pub-id pub-id-type="doi">10.1093/nar/27.2.573</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bolger</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Lohse</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Usadel</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Trimmomatic: a Flexible Trimmer for Illumina Sequence Data</article-title>. <source>Bioinformatics</source> <volume>30</volume>, <fpage>2114</fpage>&#x2013;<lpage>2120</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btu170</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Burge</surname>
<given-names>C. B.</given-names>
</name>
<name>
<surname>Karlin</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Finding the Genes in Genomic DNA</article-title>. <source>Curr. Opin. Struct. Biol.</source> <volume>8</volume>, <fpage>346</fpage>&#x2013;<lpage>354</lpage>. <pub-id pub-id-type="doi">10.1016/S0959-440X(98)80069-910</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Fastp: an Ultra-fast All-In-One FASTQ Preprocessor</article-title>. <source>Bioinformatics</source> <volume>34</volume> (<issue>17</issue>), <fpage>i884</fpage>&#x2013;<lpage>i890</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty560</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dudchenko</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Batra</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Omer</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Nyquist</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Hoeger</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Durand</surname>
<given-names>N. C.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>De Novo assembly of the <italic>Aedes aegypti</italic> Genome Using Hi-C Yields Chromosome-Length Scaffolds</article-title>. <source>Science</source> <volume>356</volume> (<issue>6333</issue>), <fpage>92</fpage>&#x2013;<lpage>95</lpage>. <pub-id pub-id-type="doi">10.1126/science.aal3327</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Durand</surname>
<given-names>N. C.</given-names>
</name>
<name>
<surname>Robinson</surname>
<given-names>J.&#x20;T.</given-names>
</name>
<name>
<surname>Shamim</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Machol</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Mesirov</surname>
<given-names>J.&#x20;P.</given-names>
</name>
<name>
<surname>Lander</surname>
<given-names>E. S.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Juicebox Provides a Visualization System for Hi-C Contact Maps with Unlimited Zoom</article-title>. <source>Cel Syst.</source> <volume>3</volume>, <fpage>99</fpage>&#x2013;<lpage>101</lpage>. <pub-id pub-id-type="doi">10.1016/j.cels.2015.07.012</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Durand</surname>
<given-names>N. C.</given-names>
</name>
<name>
<surname>Shamim</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Machol</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Rao</surname>
<given-names>S. S. P.</given-names>
</name>
<name>
<surname>Huntley</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Lander</surname>
<given-names>E. S.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Juicer Provides a One-Click System for Analyzing Loop-Resolution Hi-C Experiments</article-title>. <source>Cel Syst.</source> <volume>3</volume> (<issue>1</issue>), <fpage>95</fpage>&#x2013;<lpage>98</lpage>. <pub-id pub-id-type="doi">10.1016/j.cels.2016.07.002</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Griffiths-Jones</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Moxon</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Marshall</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Khanna</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Eddy</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Bateman</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Rfam: Annotating Non-coding RNAs in Complete Genomes</article-title>. <source>Nucleic Acids Res.</source> <volume>33</volume> (<issue>Suppl. l_1</issue>), <fpage>D121</fpage>&#x2013;<lpage>D124</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gki081</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haas</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Delcher</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Mount</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Wortman</surname>
<given-names>J.&#x20;R.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>R. K.</given-names>
<suffix>Jr</suffix>
</name>
<name>
<surname>Hannick</surname>
<given-names>L. I.</given-names>
</name>
<etal/>
</person-group> (<year>2003</year>). <article-title>Improving the Arabidopsis Genome Annotation Using Maximal Transcript Alignment Assemblies</article-title>. <source>Nucleic Acids Res.</source> <volume>31</volume> (<issue>19</issue>), <fpage>5654</fpage>&#x2013;<lpage>5666</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkg770</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haas</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Papanicolaou</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Yassour</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Grabherr</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Blood</surname>
<given-names>P. D.</given-names>
</name>
<name>
<surname>Bowden</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>
<italic>De Novo</italic> transcript Sequence Reconstruction from RNA-Seq Using the Trinity Platform for Reference Generation and Analysis</article-title>. <source>Nat. Protoc.</source> <volume>8</volume>, <fpage>1494</fpage>&#x2013;<lpage>1512</lpage>. <pub-id pub-id-type="doi">10.1038/nprot.2013.084</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haas</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Salzberg</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Pertea</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Allen</surname>
<given-names>J.&#x20;E.</given-names>
</name>
<name>
<surname>Orvis</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Automated Eukaryotic Gene Structure Annotation Using EVidenceModeler and the Program to Assemble Spliced Alignments</article-title>. <source>Genome Biol.</source> <volume>9</volume> (<issue>1</issue>), <fpage>R7</fpage>. <pub-id pub-id-type="doi">10.1186/gb-2008-9-1-r7</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname>
<given-names>M. V.</given-names>
</name>
<name>
<surname>Thomas</surname>
<given-names>G. W. C.</given-names>
</name>
<name>
<surname>Lugo-Martinez</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hahn</surname>
<given-names>M. W.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Estimating Gene Gain and Loss Rates in the Presence of Error in Genome Assembly and Annotation Using CAFE 3</article-title>. <source>Mol. Biol. Evol.</source> <volume>30</volume>, <fpage>1987</fpage>&#x2013;<lpage>1997</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/mst100</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Nextpolish: a Fast and Efficient Genome Polishing Tool for Long-Read Assembly</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>2253</fpage>&#x2013;<lpage>2255</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz891</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huerta-Cepas</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Szklarczyk</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Forslund</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Cook</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Heller</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Walter</surname>
<given-names>M. C.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>eggNOG 4.5: a Hierarchical Orthology Framework with Improved Functional Annotations for Eukaryotic, Prokaryotic and Viral Sequences</article-title>. <source>Nucleic Acids Res.</source> <volume>44</volume> (<issue>D1</issue>), <fpage>D286</fpage>&#x2013;<lpage>D293</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkv1248</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hwang</surname>
<given-names>S. M.</given-names>
</name>
</person-group> (<year>1989</year>). <article-title>A Notes on Genera Speranskia in China (Euphorbiaceae)</article-title>. <source>Bull. Bot. Res.</source> <volume>9</volume>, <fpage>37</fpage>&#x2013;<lpage>40</lpage>. </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jones</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Binns</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>H.-Y.</given-names>
</name>
<name>
<surname>Fraser</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>McAnulla</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>InterProScan 5: Genome-Scale Protein Function Classification</article-title>. <source>Bioinformatics</source> <volume>30</volume>, <fpage>1236</fpage>&#x2013;<lpage>1240</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btu031</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Katoh</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Standley</surname>
<given-names>D. M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>MAFFT Multiple Sequence Alignment Software Version 7: Improvements in Performance and Usability</article-title>. <source>Mol. Biol. Evol.</source> <volume>30</volume> (<issue>4</issue>), <fpage>772</fpage>&#x2013;<lpage>780</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/mst010</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Keilwagen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wenk</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Erickson</surname>
<given-names>J.&#x20;L.</given-names>
</name>
<name>
<surname>Schattat</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Grau</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hartung</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Using Intron Position Conservation for Homology-Based Gene Prediction</article-title>. <source>Nucleic Acids Res.</source> <volume>44</volume>, <fpage>e89</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw092</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kokot</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>D&#x142;ugosz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Deorowicz</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Kmc 3: Counting and Manipulating K-Mer Statistics</article-title>. <source>Bioinformatics</source> <volume>33</volume>, <fpage>2759</fpage>&#x2013;<lpage>2761</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx304</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Stoeckert</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Roos</surname>
<given-names>D. S.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>OrthoMCL: Identification of Ortholog Groups for Eukaryotic Genomes</article-title>. <source>Genome Res.</source> <volume>13</volume>, <fpage>2178</fpage>&#x2013;<lpage>2189</lpage>. <pub-id pub-id-type="doi">10.1101/gr.1224503</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lowe</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Eddy</surname>
<given-names>S. R.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>tRNAscan-SE: a Program for Improved Detection of Transfer RNA Genes in Genomic Sequence</article-title>. <source>Nucleic Acids Res.</source> <volume>25</volume> (<issue>5</issue>), <fpage>955</fpage>&#x2013;<lpage>964</lpage>. <pub-id pub-id-type="doi">10.1093/nar/25.5.955</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Majoros</surname>
<given-names>W. H.</given-names>
</name>
<name>
<surname>Pertea</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Salzberg</surname>
<given-names>S. L.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>TigrScan and GlimmerHMM: Two Open Source Ab Initio Eukaryotic Gene-Finders</article-title>. <source>Bioinformatics</source> <volume>20</volume> (<issue>16</issue>), <fpage>2878</fpage>&#x2013;<lpage>2879</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bth315</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nawrocki</surname>
<given-names>E. P.</given-names>
</name>
<name>
<surname>Eddy</surname>
<given-names>S. R.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Infernal 1.1: 100-fold Faster RNA Homology Searches</article-title>. <source>Bioinformatics</source> <volume>29</volume> (<issue>22</issue>), <fpage>2933</fpage>&#x2013;<lpage>2935</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btt509</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname>
<given-names>L.-T.</given-names>
</name>
<name>
<surname>Schmidt</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>von Haeseler</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Minh</surname>
<given-names>B. Q.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>IQ-TREE: A Fast and Effective Stochastic Algorithm for Estimating Maximum-Likelihood Phylogenies</article-title>. <source>Mol. Biol. Evol.</source> <volume>32</volume> (<issue>1</issue>), <fpage>268</fpage>&#x2013;<lpage>274</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msu300</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pruitt</surname>
<given-names>K. D.</given-names>
</name>
<name>
<surname>Tatusova</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Maglott</surname>
<given-names>D. R.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>NCBI Reference Sequences (RefSeq): a Curated Non-redundant Sequence Database of Genomes, Transcripts and Proteins</article-title>. <source>Nucleic Acids Res.</source> <volume>35</volume> (<issue>Suppl. l_1</issue>), <fpage>D61</fpage>&#x2013;<lpage>D65</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkl842</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ranallo-Benavidez</surname>
<given-names>T. R.</given-names>
</name>
<name>
<surname>Jaron</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Schatz</surname>
<given-names>M. C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Genomescope 2.0 and Smudgeplot for Reference-free Profiling of Polyploid Genomes</article-title>. <source>Nat. Commun.</source> <volume>11</volume> (<issue>1</issue>), <fpage>1432</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-020-14998-3</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sim&#xe3;o</surname>
<given-names>F. A.</given-names>
</name>
<name>
<surname>Waterhouse</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Ioannidis</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kriventseva</surname>
<given-names>E. V.</given-names>
</name>
<name>
<surname>Zdobnov</surname>
<given-names>E. M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>BUSCO: Assessing Genome Assembly and Annotation Completeness with Single-Copy Orthologs</article-title>. <source>Bioinformatics</source> <volume>31</volume>, <fpage>3210</fpage>&#x2013;<lpage>3212</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv351</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Smit</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hubley</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>RepeatModeler Open-1.0</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="http://www.repeatmasker.org">http://www.repeatmasker.org</ext-link> (accessed August 10, 2020)</comment>. </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stanke</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Diekhans</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Baertsch</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Haussler</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Using Native and Syntenically Mapped cDNA Alignments to Improve De Novo Gene Finding</article-title>. <source>Bioinformatics</source> <volume>24</volume>, <fpage>637</fpage>&#x2013;<lpage>644</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btn013</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tarailo&#x2010;Graovac</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Using RepeatMasker to Identify Repetitive Elements in Genomic Sequences</article-title>. <source>Curr. Protoc. Bioinformatics</source> <volume>25</volume> (<issue>1</issue>), <fpage>4</fpage>&#x2013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1002/0471250953.bi0410s25</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Webster</surname>
<given-names>G. L.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Euphorbiaceae</article-title>,&#x201d; in <source>The Families and Genera of Vascular Plants, Volume XI. Flowering Plants. Eudicots. Malpighiales</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Kubitzki</surname>
<given-names>K.</given-names>
</name>
</person-group> (<publisher-loc>Berlin</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>51</fpage>&#x2013;<lpage>216</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-642-39417-1_10</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>PAML 4: Phylogenetic Analysis by Maximum Likelihood</article-title>. <source>Mol. Biol. Evol.</source> <volume>24</volume>, <fpage>1586</fpage>&#x2013;<lpage>1591</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msm088</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>