<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2022.853861</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Genomic Analysis Based on Chromosome-Level Genome Assembly Reveals an Expansion of Terpene Biosynthesis of <italic>Azadirachta indica</italic></article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Du</surname> <given-names>Yuhui</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1633314/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Song</surname> <given-names>Wei</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Yin</surname> <given-names>Zhiqiu</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1006395/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wu</surname> <given-names>Shengbo</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Liu</surname> <given-names>Jiaheng</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Ning</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1697584/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Jin</surname> <given-names>Hua</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Qiao</surname> <given-names>Jianjun</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/234552/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Huo</surname> <given-names>Yi-Xin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="corresp" rid="c003"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/767283/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Key Laboratory of Molecular Medicine and Biotherapy, School of Life Sciences, Beijing Institute of Technology</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>National Engineering Laboratory for Efficient Utilization of Soil and Fertilizer Resources, College of Resources and Environment, Shandong Agricultural University</institution>, <addr-line>Tai&#x2019;an</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Key Laboratory of Systems Bioengineering (Ministry of Education), School of Chemical Engineering and Technology, Tianjin University</institution>, <addr-line>Tianjin</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>SynBio Research Platform, Collaborative Innovation Centre of Chemical Science and Engineering (Tianjin), Tianjin University</institution>, <addr-line>Tianjin</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>Tobacco Research Institute, Chinese Academy of Agricultural Sciences</institution>, <addr-line>Qingdao</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Jeremy Coate, Reed College, United States</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Sunil Kumar Sahu, Beijing Genomics Institute (BGI), China; Liangsheng Zhang, Zhejiang University, China</p></fn>
<corresp id="c001">&#x002A;Correspondence: Hua Jin, <email>huajin@bit.edu.cn</email></corresp>
<corresp id="c002">Jianjun Qiao, <email>jianjunq@tju.edu.cn</email></corresp>
<corresp id="c003">Yi-Xin Huo, <email>huoyixin@bit.edu.cn</email></corresp>
<fn fn-type="equal" id="fn002"><p><sup>&#x2020;</sup>These authors have contributed equally to this work</p></fn>
<fn fn-type="other" id="fn004"><p>This article was submitted to Plant Systematics and Evolution, a section of the journal Frontiers in Plant Science</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>18</day>
<month>04</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>853861</elocation-id>
<history>
<date date-type="received">
<day>13</day>
<month>01</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>03</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2022 Du, Song, Yin, Wu, Liu, Wang, Jin, Qiao and Huo.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Du, Song, Yin, Wu, Liu, Wang, Jin, Qiao and Huo</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p><italic>Azadirachta indica</italic> (neem), an evergreen tree of the Meliaceae family, is a source of the potent biopesticide azadirachtin. The lack of a chromosome-level assembly impedes an in-depth understanding of its genome architecture and the comparative genomic analysis of <italic>A. indica</italic>. Here, a high-quality genome assembly of <italic>A. indica</italic> was constructed using a combination of data from Illumina, PacBio, and Hi-C technology, which is the first chromosome-scale genome assembly of <italic>A. indica</italic>. Based on the length of our assembly, the genome size of <italic>A. indica</italic> is estimated to be 281 Mb anchored to 14 chromosomes (contig N50 = 6 Mb and scaffold N50 = 19 Mb). The genome assembly contained 115 Mb repetitive elements and 25,767 protein-coding genes. Evolutional analysis revealed that <italic>A. indica</italic> didn&#x2019;t experience any whole-genome duplication (WGD) event after the core eudicot &#x03B3; event, but some genes and genome segment might likely experienced recent duplications. The secondary metabolite clusters, TPS genes, and CYP genes were also identified. Comparative genomic analysis revealed that most of the <italic>A. indica</italic>-specific TPS genes and CYP genes were located on the terpene-related clusters on chromosome 13. It is suggested that chromosome 13 may play an important role in the specific terpene biosynthesis of <italic>A. indica</italic>. The gene duplication events may be responsible for the terpene biosynthesis expansion in <italic>A. indica</italic>. The genomic dataset and genomic analysis created for <italic>A. indica</italic> will shed light on terpene biosynthesis in <italic>A. indica</italic> and facilitate comparative genomic research of the family Meliaceae.</p>
</abstract>
<kwd-group>
<kwd><italic>Azadirachta indica</italic></kwd>
<kwd>chromosome-level assembly</kwd>
<kwd>comparative genomics</kwd>
<kwd>terpene biosynthesis</kwd>
<kwd>genome evolution</kwd>
</kwd-group>
<counts>
<fig-count count="6"/>
<table-count count="2"/>
<equation-count count="0"/>
<ref-count count="72"/>
<page-count count="14"/>
<word-count count="9163"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="intro">
<title>Introduction</title>
<p><italic>Azadirachta indica</italic> (neem) is a member of Meliaceae family, which is extensively studied for its bioactive products (<xref ref-type="bibr" rid="B49">Schmutterer, 1995</xref>). It grows natively on the Indian subcontinent and also in other countries such as Egypt and the Kingdom of Saudi Arabia. <italic>A. indica</italic> is a source of abundant limonoids and simple terpenoids which are responsible for its biological activity (<xref ref-type="bibr" rid="B8">Dai et al., 2001</xref>). Azadirachtin, the most important active compound in the neem tree, has been intensively studied because of its wide range of insecticidal properties and low toxicity in mammals (<xref ref-type="bibr" rid="B32">Ley, 1994</xref>). Additionally, the neem tree extracts also exhibit many pharmaceutical functions, such as anti-inflammatory, anticancer, antimicrobial, and antidiabetic activities (<xref ref-type="bibr" rid="B52">Soares et al., 2014</xref>; <xref ref-type="bibr" rid="B1">Abdelhady et al., 2015</xref>). A lot of studies have focused on the synthesis of azadirachtin, including chemosynthesis, hairy root culture, cell line culture, and callus culture (<xref ref-type="bibr" rid="B61">Veitch et al., 2007</xref>; <xref ref-type="bibr" rid="B53">Srivastava and Srivastava, 2013</xref>; <xref ref-type="bibr" rid="B36">Mithilesh and Rakhi, 2014</xref>; <xref ref-type="bibr" rid="B48">Rodrigues et al., 2014</xref>). However, these methods are either of low extraction efficiency or not environmentally friendly. Therefore, the reconstruction of biosynthetic pathway of azadirachtin for heterologous production is an alternative method.</p>
<p>An omics strategy is an effective method to study the biosynthesis of secondary metabolites. Transcriptomes of <italic>A. indica</italic> tissues (stem, leaf, flower, root, and fruit) have been sequenced, which paved the way to the potential synthetic pathway of azadirachtin and gene expression profiles in various organs. Draft genomes have also been sequenced, which led to a basic understanding of the genetic characteristics of <italic>A. indica</italic> (<xref ref-type="bibr" rid="B26">Krishnan et al., 2012</xref>, <xref ref-type="bibr" rid="B25">2016</xref>; <xref ref-type="bibr" rid="B30">Kuravadi et al., 2015</xref>). However, the lack of a chromosome-level genome sequence has hindered a full understanding of the secondary metabolite biosynthesis and the evolution of <italic>A. indica</italic>. In addition, Meliaceae are known to produce around 1,500 structurally diverse limonoids, which have agricultural and medical values (<xref ref-type="bibr" rid="B16">Hodgson et al., 2019</xref>). A chromosome-level genome is essential for genome-wide studies of the Meliaceae family.</p>
<p>In this study, the first chromosome-level genome of <italic>A. indica</italic> was assembled through a combination of Illumina, PacBio, and Hi-C technology. Based on the assembled genome sequence and annotation, we characterized the history of gene and whole-genome duplication (WGD) events, as well as the evolution of secondary metabolite clusters and resistance genes. These results improved the understanding of the genomic architecture of <italic>A. indica</italic>. This chromosome-level genome assembly can be used as a new reference genome for <italic>A. indica</italic>, laying a substantial foundation for further genomic studies.</p>
</sec>
<sec id="S2" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="S2.SS1">
<title>Plant Material, DNA Preparation, and Genome Sequencing</title>
<p>Fresh tissues of <italic>A. indica</italic> were randomly collected from a locally grown tree in the Liufang Yuan park of Hainan University (100.61438 E, 36.28672 N), Hainan Province, China. Fresh leaves were collected to isolate genomic DNA of <italic>A. indica</italic> for <italic>de novo</italic> sequencing and assembly. Genomic DNA was extracted from leaves of <italic>A. indica</italic> using the DNAsecure Plant Kit (TIANGEN, Biotech Co., Ltd., Beijing, China). For Illumina sequencing, a paired-end library with an insert size of 270 bp was generated and sequenced on the Illumina HiSeq X Ten platform. For PacBio sequencing, a 20 kb insert library was generated and sequenced on the PacBio RSII platform.</p>
</sec>
<sec id="S2.SS2">
<title>Genome Assembly</title>
<p>First, Canu v2.0 (<xref ref-type="bibr" rid="B23">Koren et al., 2017</xref>) software was used to correct and assemble raw PacBio sequencing reads, and 886 contigs with N50 &#x223C; 6 M were assembled by Canu. In addition, we performed a round of polishing on the assembled contigs using the RACON (<xref ref-type="bibr" rid="B60">Vaser et al., 2017</xref>) with the PacBio long reads, and the polished contigs were further corrected two rounds on the genome-wide base-level by Pilon v1.21 (<xref ref-type="bibr" rid="B62">Walker et al., 2014</xref>) with the Illumina short reads. 870 contigs were left after error correction with RACON and Pilon software. Genome size of <italic>A. indica</italic> was estimated by flow cytometry (<xref ref-type="bibr" rid="B44">Pellicer and Leitch, 2020</xref>).</p>
</sec>
<sec id="S2.SS3">
<title>Chromosome Assembly Using Hi-C</title>
<p>For Hi-C sequencing, a 150 bp paired-end library was generated and sequenced on the Illumina HiSeq X Ten platform. Bowtie2 (<xref ref-type="bibr" rid="B31">Langmead and Salzberg, 2012</xref>) with the default parameters was used to map the clean reads to the <italic>A. indica</italic>. HiC-Pro v2.11.1 (<xref ref-type="bibr" rid="B50">Servant et al., 2015</xref>) was used to map the Hi-C sequencing reads to the assembled draft genome and detect the valid contacts. Then we used ALLHIC v0.9.12 (<xref ref-type="bibr" rid="B70">Zhang et al., 2019</xref>) to cluster contigs into chromosome-scale scaffolds based on the relationships among valid contacts.</p>
</sec>
<sec id="S2.SS4">
<title>Assessment of Genomic Integrity</title>
<p>The draft genome sequence of <italic>A. indica</italic> (GCA_000439995.3) was downloaded from NCBI as a reference. The accuracy and integrity of the genome assembly was evaluated using BUSCO v3.0.2, based on the OrthoDB<sup><xref ref-type="fn" rid="footnote1">1</xref></sup> database. LTR Assembly Index (LAI) scores were calculated by LTR_Retriever (v2.8) with the default parameters (<xref ref-type="bibr" rid="B43">Ou and Jiang, 2018</xref>; <xref ref-type="bibr" rid="B42">Ou et al., 2018</xref>). The transcriptomic NGS short reads from 5 tissues of <italic>A.indica</italic> (SRR12709585, SRR12709584, SRR12709583, SRR12709582, and SRR12709581) (<xref ref-type="bibr" rid="B64">Wang et al., 2020</xref>) were mapped against the assemblies using Hisat2 (<xref ref-type="bibr" rid="B22">Kim et al., 2015</xref>) with default parameters. The genomic NGS short reads were also mapped to the assemblies using Bowtie2. Finally, the collinearity analysis between our assembly and GCA_000439995.3 was performed with Minimap2<sup><xref ref-type="fn" rid="footnote2">2</xref></sup> and dotPlotly.<sup><xref ref-type="fn" rid="footnote3">3</xref></sup></p>
</sec>
<sec id="S2.SS5">
<title>Repetitive Elements</title>
<p>We identified repetitive elements through both RepeatModeler v1.0.8 (<xref ref-type="bibr" rid="B46">Price et al., 2005</xref>) and RepeatMasker v4.0.7 (<xref ref-type="bibr" rid="B57">Tarailo-Graovac and Chen, 2009</xref>). The LTRs of <italic>A. indica</italic> were identified by using LTRharvest v1.6.1 (<xref ref-type="bibr" rid="B11">Ellinghaus et al., 2008</xref>) and LTR_Finder v1.05 (<xref ref-type="bibr" rid="B66">Xu and Wang, 2007</xref>). LTR_retriever v2.8.7 (<xref ref-type="bibr" rid="B43">Ou and Jiang, 2018</xref>) was used to integrate the results of LTRharvest and LTR_Finder. RepeatModeler employed RECON v1.08 and RepeatScout v1.0.5 to predict interspersed repeats and then combined the repeat sequences from LTR-retriever with the repeat sequences from RepeatModeler to be the local repeat library. To recover the repeats in the <italic>A. indica</italic> genome, a homology-based repeat search was conducted by using RepeatMasker with the <italic>ab initio</italic> repeat database and Repbase.<sup><xref ref-type="fn" rid="footnote4">4</xref></sup></p>
</sec>
<sec id="S2.SS6">
<title>Non-coding RNAs</title>
<p>Non-coding RNAs were detected through searching against various RNA libraries. Reliable tRNA positions were searched <italic>via</italic> tRNAscan-SE v1.3.1 (<xref ref-type="bibr" rid="B35">Lowe and Eddy, 1997</xref>). Small nuclear RNAs (snRNAs) and microRNAs (miRNAs) were searched by using INFERNAL v1.1 (<xref ref-type="bibr" rid="B38">Nawrocki and Eddy, 2013</xref>) against the Rfam (<xref ref-type="bibr" rid="B13">Griffiths-Jones et al., 2005</xref>) database.</p>
</sec>
<sec id="S2.SS7">
<title>Gene Prediction</title>
<p>Homology annotation was performed using genomes of three representative species, including <italic>Citrus sinensis</italic> (<xref ref-type="bibr" rid="B65">Xu et al., 2013</xref>), <italic>Theobroma cacao</italic> (<xref ref-type="bibr" rid="B3">Argout et al., 2011</xref>), and <italic>Acer yangbiense</italic> (<xref ref-type="bibr" rid="B67">Yang et al., 2019</xref>). The TBLASTN software (<xref ref-type="bibr" rid="B5">Camacho et al., 2009</xref>) was used to align the protein sequences of these species to <italic>A. indica</italic> genome sequence, with an <italic>E</italic>-value &#x2264; 1e-5. The exact gene structures were predicted using GeneWise 2.2.0 (<xref ref-type="bibr" rid="B4">Birney et al., 2004</xref>) according to the TBLASTN results. We used Cufflinks v2.2.1 (<xref ref-type="bibr" rid="B58">Trapnell et al., 2012</xref>) to preliminarily identify gene structures based on the RNA-seq data. <italic>ab initio</italic> annotation was performed using Augustus v3.2.2 (<xref ref-type="bibr" rid="B54">Stanke et al., 2004</xref>) and SNAP (<xref ref-type="bibr" rid="B24">Korf, 2004</xref>) with the repeat-masked genome sequences. All genes predicted from the three annotation procedures were integrated with MAKER (<xref ref-type="bibr" rid="B17">Holt and Yandell, 2011</xref>) software.</p>
</sec>
<sec id="S2.SS8">
<title>Functional Annotation</title>
<p>The protein sequences of the consensus gene set were aligned to four protein databases, including NR,<sup><xref ref-type="fn" rid="footnote5">5</xref></sup> InterPro,<sup><xref ref-type="fn" rid="footnote6">6</xref></sup> Swiss-Prot,<sup><xref ref-type="fn" rid="footnote7">7</xref></sup> and EggNOG (<xref ref-type="bibr" rid="B45">Powell et al., 2012</xref>), for predicted gene annotation. The physically clustered specialized metabolic pathway genes were identified by the PlantiSMASH analytical pipeline (<xref ref-type="bibr" rid="B21">Kautsar et al., 2017</xref>). Plant disease resistance (R) genes were predicted by the Disease Resistance Analysis and Gene Ontology (DRAGO) pipeline (<xref ref-type="bibr" rid="B41">Osuna-Cruz et al., 2018</xref>).</p>
</sec>
<sec id="S2.SS9">
<title>Phylogenetic Analysis and Expansion/Contraction of Gene Families</title>
<p>The genome of <italic>A. indica</italic> and 13 other plants were selected for phylogenetic analysis. All-vs.-all BLASTP (<xref ref-type="bibr" rid="B2">Altschul et al., 1997</xref>) search results with an <italic>E</italic>-value &#x2264; 1e-5 were grouped into orthologous and paralogous clusters using OrthoFinder v2.3.7 (<xref ref-type="bibr" rid="B12">Emms and Kelly, 2019</xref>). Multiple sequence alignments of all single-copy orthologous gene families were performed by using MUSCLE (<xref ref-type="bibr" rid="B10">Edgar, 2004</xref>). The set of single nucleotide polymorphisms (SNPs) presented in each single-copy orthologous gene family was extracted and then integrated according to the arrangement of the genes on the <italic>A. indica</italic> genome. A maximum likelihood (ML) tree was constructed using the integrated SNPs by PhyML v3.1 (<xref ref-type="bibr" rid="B14">Guindon et al., 2009</xref>). Divergence time between species was estimated using MCMCtree, which was incorporated in the PAML v4.8 package (<xref ref-type="bibr" rid="B68">Yang, 1997</xref>). CAF&#x00C9; v3.1 (<xref ref-type="bibr" rid="B9">De Bie et al., 2006</xref>) was used to measure the expansion/contraction of orthologous gene families.</p>
</sec>
<sec id="S2.SS10">
<title>Genome Duplication Analysis</title>
<p>MCScan v0.8 (<xref ref-type="bibr" rid="B56">Tang et al., 2008</xref>) package with default parameters was used for the detection of syntenic blocks, defined as regions with more than 5 collinear genes. We aligned the amino acid sequences of syntenic block gene pairs and reciprocal best hits (RBH) gene pairs using MAFFT and further aligned their nucleotide sequences using ParaAT (<xref ref-type="bibr" rid="B71">Zhang et al., 2012</xref>). The synonymous substitution rate (<italic>Ks</italic>) values of these gene pairs were calculated using YN model in KaKs_Calculator v2.0 (<xref ref-type="bibr" rid="B63">Wang et al., 2010</xref>). The value of <italic>Ks</italic> peak was determinated by the abscissa value of the highest point of the <italic>A. indica Ks</italic> plot. The WGD events of each species were estimated based on the <italic>Ks</italic> distributions. The gene pairs with the median <italic>Ks</italic> &#x003C; 0.05 were defined as the retained genes from the recent segmental duplication. According to the formula <italic>T</italic> = <italic>Ks</italic>/2<italic>r</italic>, the <italic>Ks</italic> values were converted to divergence times, where <italic>T</italic> is divergence time and <italic>r</italic> is the neutral substitution rate (<italic>r</italic> = 3.39 &#x00D7; 10<sup>&#x2013;9</sup>). The paralog analysis in <italic>A. indica</italic> genome were performed using RBH from all-vs.-all BLASTp searches using <italic>A. indica</italic> protein sequences. RBHs are defined as reciprocal best BLASTp matches with <italic>e</italic>-value threshold of 1e-5, c-score threshold of 0.3 (<xref ref-type="bibr" rid="B15">Guo et al., 2018</xref>).</p>
</sec>
<sec id="S2.SS11">
<title>Identification and Phylogenetic Analysis of Terpene Synthase and Cytochrome P450 Family Members</title>
<p>Genomes were aligned using HMMER 3.0 search with an <italic>E</italic>-value 1e-5 against the Pfam-A database (02-May-2020) locally. PF01397 (Terpene synthase, N-terminal domain) and PF03936 (Terpene synthase family, metal binding domain) domains were used to identify the members of the TPS gene family. The collection used for phylogenetic analysis consisted of 403 putative TPSs from <italic>A. indica</italic> and other 13 plants and six reported TPSs belonged to TPS- a (AAX16121.1), b (AAQ16588.1), c (AAD04292.1), e (Q39548.1), f (Q93YV0.1), and g (ADD81294.1) subfamilies (<xref ref-type="bibr" rid="B29">Kumar et al., 2018b</xref>; <xref ref-type="bibr" rid="B72">Zhou et al., 2020</xref>). PF00067 (Cytochrome P450) was used to identify the members of the CYP gene family. Putative CYPs were screened by amino acid length (450 &#x003C; length &#x003C; 600) to perform phylogenetic analysis. Protein sequences were aligned using ClustalX in MEGAX using default sets (<xref ref-type="bibr" rid="B27">Kumar et al., 2018a</xref>). The ML trees were constructed based on the alignment of TPS and CYP protein sequences using MEGAX software with 100 bootstrap replicates, respectively. The identification of <italic>A. indica</italic>-specific TPS and CYP genes was based on the phylogenetic analysis using other 13 plant genome as the outgroup and a cutoff of 55% identity, which indicated separate subfamily assignment (<xref ref-type="bibr" rid="B33">Liu et al., 2018</xref>; <xref ref-type="bibr" rid="B59">Tu et al., 2020</xref>).</p>
</sec>
</sec>
<sec id="S3" sec-type="results">
<title>Results</title>
<sec id="S3.SS1">
<title>Genome Sequencing and Assembly</title>
<p>To obtain a chromosome-level assembly of <italic>A. indica</italic>, the genome was sequenced using a combination of Illumina, PacBio, and Hi-C methods, and assembled by a hierarchical approach. A total of 110 Gb (providing 188 &#x00D7; genome coverage) Illumina paired-end short reads were produced and the heterozygosity ratio was estimated to be 0.896%. Based on the 21-mer depth distribution of the Illumina short reads, the genome size was estimated to be 165 Mb (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 1</xref>).</p>
<p>We also generated 126 Gb of raw PacBio sequencing reads from the single-molecule real-time (SMRT) sequencing platform, reaching 256 &#x00D7; coverage of the <italic>A. indica</italic> genome (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 2</xref> and <xref ref-type="supplementary-material" rid="TS1">Supplementary Table 1</xref>). The total size of the reads assembled from the post-correction genome was 281,629,231 bp with a GC content of 32.2%, consisting of 870 contigs. The contig N50 was 6,039,544 bp, and the longest contig was 15,111,501 bp. Our genome assembly constitutes &#x223C;73.2% of the 385 Mb genome estimated by flow cytometry (<xref ref-type="bibr" rid="B44">Pellicer and Leitch, 2020</xref>).</p>
<p>We further conducted the Hi-C sequencing to scaffold the preliminary assemblies and enhance the assembled contiguity at the chromosome level. In total, the Hi-C sequencing generated approximately 40.48 Gb clean reads. 94.5% reads from Hi-C sequencing were mapped to the assembled contigs, of which 26.1% were unique mapped read pairs (<xref ref-type="supplementary-material" rid="TS1">Supplementary Table 1</xref>). The verified read pairs were selected after considering the map position and orientation of the unique mapped read pairs. Then, according to the contiguity information between Hi-C read pairs, ALLHIC software was used to cluster, order, and orient the previous assemblies for chromosome-level scaffolding (<xref ref-type="fig" rid="F1">Figure 1A</xref>). A total of 70 scaffolds were obtained after Hi-C sequencing reads assist chromosome assemble, of which 14 scaffolds formed chromosomes (<xref ref-type="fig" rid="F1">Figure 1B</xref> and <xref ref-type="supplementary-material" rid="TS1">Supplementary Table 1</xref>). The final size of the <italic>A. indica</italic> genome assembly was 281 Mb, and the scaffold N50 was 19 Mb (<xref ref-type="table" rid="T1">Table 1</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>Genome features of genome assembly of <italic>A. indica.</italic> <bold>(A)</bold> Hi-C contact data mapped on the <italic>A. indica</italic> genome showing genome-wide all- by-all interactions. <bold>(B)</bold> The landscape of genome assembly and annotation of <italic>A. indica</italic>. (A) Circular representation of the pseudomolecules. (B) The distribution of gene density with densities calculated in 500 kb windows. (C&#x2013;G) Expression of <italic>A. indica</italic> genes (from outside to inside tracks: stem, root, leaf, fruit and flower). (H,I) The distribution of repeat density and GC density with densities calculated in 500 kb windows. (J) Locations of genes mapped to secondary metabolism. (K) Syntenic blocks.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-853861-g001.tif"/>
</fig>
<table-wrap position="float" id="T1">
<label>TABLE 1</label>
<caption><p>Statistics of the <italic>A. indica</italic> genome assembly.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Feature</td>
<td valign="top" align="center">Value</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Genome size (Mb)</td>
<td valign="top" align="center">281</td>
</tr>
<tr>
<td valign="top" align="left">Genome GC%</td>
<td valign="top" align="center">32.2</td>
</tr>
<tr>
<td valign="top" align="left">N50 (Mb)</td>
<td valign="top" align="center">19</td>
</tr>
<tr>
<td valign="top" align="left">Gene number</td>
<td valign="top" align="center">25,767</td>
</tr>
<tr>
<td valign="top" align="left">Average gene length (bp)</td>
<td valign="top" align="center">2,837</td>
</tr>
<tr>
<td valign="top" align="left">Exon no. per gene</td>
<td valign="top" align="center">5.4</td>
</tr>
<tr>
<td valign="top" align="left">Exon number</td>
<td valign="top" align="center">138,941</td>
</tr>
<tr>
<td valign="top" align="left">Average exon length (bp)</td>
<td valign="top" align="center">231</td>
</tr>
<tr>
<td valign="top" align="left">Total exon length (bp)</td>
<td valign="top" align="center">32,191,037</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="S3.SS2">
<title>Evaluation of the Genome Assembly</title>
<p>The quality of the assembly was assessed and compared with the reference genome sequence of <italic>A. indica</italic> from NCBI (GCA_000439995.3) (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 3</xref> and <xref ref-type="table" rid="T2">Table 2</xref>). The Benchmarking Universal Single-Copy Orthologs (BUSCO) (<xref ref-type="bibr" rid="B51">Simao et al., 2015</xref>) analysis was used to evaluate the integrity of the genome. The BUSCO assessment showed that the completeness of the assembled genome of <italic>A. indica</italic> was 91.7%, which was much higher than that of the reference genome (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 3A</xref>, <xref ref-type="supplementary-material" rid="DS1">Table 2</xref>, and <xref ref-type="supplementary-material" rid="TS2">Supplementary Table 2</xref>). The average LAI score of <italic>A. indica</italic> genome was 4.82, which was lower than the &#x201C;reference&#x201D; quality (10 &#x003C; LAI &#x003C; 20) based on the LAI classification (<xref ref-type="bibr" rid="B42">Ou et al., 2018</xref>). The Illumina short reads were also used to assess the integrity of the genome. The transcriptomic Illumina sequencing short reads were mapped to the two assemblies by Hisat2 (<xref ref-type="bibr" rid="B22">Kim et al., 2015</xref>), and approximately 92.76 and 87.49% of the reads were mapped to our assembly and GCA_000439995.3, respectively. By using Bowtie2 (<xref ref-type="bibr" rid="B31">Langmead and Salzberg, 2012</xref>) software, the genomic Illumina sequencing short reads were also mapped to the assemblies. About 99.29 and 97.09% of the Illumina short reads could map to our assembly and GCA_000439995.3, respectively (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 3B</xref>). Finally, collinearity analysis revealed good collinearity between our assembly and GCA_000439995.3 (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 3C</xref>).</p>
<table-wrap position="float" id="T2">
<label>TABLE 2</label>
<caption><p>Comparison of the <italic>A. indica</italic> genome assembly versions.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Feature</td>
<td valign="top" align="center">This study</td>
<td valign="top" align="center"><xref ref-type="bibr" rid="B26">Krishnan et al., 2012</xref></td>
<td valign="top" align="center">GCA_000439995.3</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Sequence technology</td>
<td valign="top" align="center">Illumina + PacBio<break/>+ Hi-C</td>
<td valign="top" align="center">Illumina + PacBio</td>
<td valign="top" align="center">Illumina</td>
</tr>
<tr>
<td valign="top" align="left">Assembly level</td>
<td valign="top" align="center">Chromosome</td>
<td valign="top" align="center">Scaffold</td>
<td valign="top" align="center">Contig</td>
</tr>
<tr>
<td valign="top" align="left">Genome size (Mb)</td>
<td valign="top" align="center">281</td>
<td valign="top" align="center">216</td>
<td valign="top" align="center">264</td>
</tr>
<tr>
<td valign="top" align="left">Genome GC%</td>
<td valign="top" align="center">32.2</td>
<td valign="top" align="center">31.9</td>
<td valign="top" align="center">32.0</td>
</tr>
<tr>
<td valign="top" align="left">Number of scaffolds</td>
<td valign="top" align="center">70</td>
<td valign="top" align="center">25,560</td>
<td valign="top" align="center">126,142</td>
</tr>
<tr>
<td valign="top" align="left">Scaffold N50 (bp)</td>
<td valign="top" align="center">19,542,739</td>
<td valign="top" align="center">2,629,187</td>
<td valign="top" align="center">3,491</td>
</tr>
<tr>
<td valign="top" align="left">Number of contigs</td>
<td valign="top" align="center">870</td>
<td valign="top" align="center">48,555</td>
<td valign="top" align="center">142,701</td>
</tr>
<tr>
<td valign="top" align="left">Contig N50 (bp)</td>
<td valign="top" align="center">6,039,544</td>
<td valign="top" align="center">25,406</td>
<td valign="top" align="center">3,310</td>
</tr>
<tr>
<td valign="top" align="left">BUSCO</td>
<td valign="top" align="center">91.7%</td>
<td valign="top" align="center">91.4%</td>
<td valign="top" align="center">79.9%</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="S3.SS3">
<title>Gene Prediction and Genome Annotation</title>
<p>Gene models were generated by a combination of reference plant protein homology support, transcriptome data, and <italic>ab initio</italic> gene prediction. All gene models were merged with MAKER (<xref ref-type="bibr" rid="B17">Holt and Yandell, 2011</xref>), resulting in a total of 25,767 protein-coding genes with an average sequence length of 2,837 bp. On average, each predicted gene contained 5.4 exons with a mean sequence length of 231 bp (<xref ref-type="table" rid="T1">Table 1</xref>). In addition, 3,856 non-coding RNAs, including 1,381 rRNAs, 1,204 tRNAs, 173 microRNAs (miRNAs), and 1,098 small nuclear RNAs (snRNAs) were identified (<xref ref-type="supplementary-material" rid="TS3">Supplementary Table 3</xref>). We also identified 40.99% of the assembled sequences as repetitive sequences, which was higher than that of the reported genomes (<xref ref-type="bibr" rid="B30">Kuravadi et al., 2015</xref>; <xref ref-type="bibr" rid="B25">Krishnan et al., 2016</xref>). The majority of the repeats were long terminal repeats (LTRs), constituting 16.88% of the genome. Unclassified elements, DNA elements, and long interspersed nuclear elements (LINEs), accounted for 14.28, 6.54, and 1.08% of the genome, respectively (<xref ref-type="supplementary-material" rid="TS4">Supplementary Table 4</xref>).</p>
<p>To further evaluate the functional validity of the predicted genes, Diamond, BLASTP, InterProScan and EggNOG-mapper were utilized by searching the Nr, SwissProt, InterPro, and EggNOG databases (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 3D</xref>). Overall, 24,801 genes (96.2%) were functionally assigned. 95.4 and 81.6% of these genes found homologies and annotated proteins in the Nr and SwissProt databases, respectively. 84.3% of the genes were detected with conserved protein domains using InterProScan. In addition, 47.4% of the genes were categorized by Kyoto Encyclopedia of Genes and Genomes (KEGG) pathway (<xref ref-type="bibr" rid="B37">Moriya et al., 2007</xref>; <xref ref-type="supplementary-material" rid="TS5">Supplementary Table 5</xref>).</p>
</sec>
<sec id="S3.SS4">
<title>Phylogenetic Analysis</title>
<p>To investigate the genetic diversity and evolutionary history of <italic>A. indica</italic> genome, a gene family clustering analysis with the <italic>A. indica</italic> genome and 13 other representative plant species was performed. These selected species included two plants in the <italic>Sapindales</italic> order (<italic>Acer yangbiense</italic> and <italic>Citrus sinensis</italic>), eight plants in the eudicot clade (<italic>Arabidopsis thaliana</italic>, <italic>Theobroma cacao</italic>, <italic>Gossypium raimondii</italic>, <italic>Carica papaya</italic>, <italic>Vitis vinifera</italic>, <italic>Cucumis sativus</italic>, <italic>Fragaria vesca</italic>, <italic>Prunus persica</italic>, and <italic>Solanum lycopersicum</italic>), and two outgroup species (<italic>Brachypodium distachyon</italic> and <italic>Amborella trichopoda</italic>).</p>
<p>OrthoFinder (<xref ref-type="bibr" rid="B12">Emms and Kelly, 2019</xref>) was used to construct a phylogenetic tree with 1,338 single-copy orthologous genes among 14 species, which showed that <italic>A. indica</italic> is most closely related to <italic>C. sinensis</italic> (<xref ref-type="fig" rid="F2">Figure 2</xref>). Further analysis showed that 36 gene families were specific to <italic>A. indica</italic> (<xref ref-type="supplementary-material" rid="TS6">Supplementary Table 6</xref>). Enrichment analysis showed that these specific genes were mostly involved in &#x201C;binding,&#x201D; &#x201C;catalytic activity,&#x201D; &#x201C;metabolic process,&#x201D; &#x201C;cellular process,&#x201D; and &#x201C;membrane&#x201D; (<xref ref-type="supplementary-material" rid="TS7">Supplementary Table 7</xref>). With the divergence time between <italic>P. persica</italic> and <italic>F. vesca</italic> as a calibration point (with the corrected time obtained from TimeTree <xref ref-type="bibr" rid="B28">Kumar et al. (2017)</xref>), the divergence time among these species were also estimated. <italic>A. indica</italic> and <italic>C. sinensis</italic> diverged from a common ancestor &#x223C;57 Mya (<xref ref-type="fig" rid="F2">Figure 2</xref>). To better understand the genetic basis of <italic>A. indica</italic>, the expansion and contraction of gene families were investigated. 997 gene families were expanded in <italic>A. indica</italic>, while 293 gene families were contracted from the <italic>A. indica</italic> genome. Compared with <italic>C. sinensis</italic>, which has 369 expanded gene families and 682 contracted gene families, <italic>A. indica</italic> has expanded more gene families. GO and KEGG analysis of the expanded and contracted gene families were also performed (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 4</xref> and <xref ref-type="supplementary-material" rid="TS8">Supplementary Table 8</xref>). The <italic>A. indica</italic>-specific expanded and contracted gene families might be related to the adaptation to <italic>A. indica</italic>-specific tropical niches. Further researches are required to verify the function of these genes.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Maximum Likelihood based phylogenetic analysis of <italic>A. indica</italic> and 13 other plant species using 1,338 single-copy orthologous genes with PhyML software (<xref ref-type="bibr" rid="B14">Guindon et al., 2009</xref>). The estimation of the divergence times (Mya; red; star) of <italic>A. indica</italic> with its closely related species are indicated. The number of expanded gene families (+; blue) and the number of contracted gene families (&#x2013;; red) are shown in each branch.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-853861-g002.tif"/>
</fig>
</sec>
<sec id="S3.SS5">
<title>Genome Duplication Analysis</title>
<p>To investigate genome wide duplications in <italic>A. indica</italic> genome, self-comparison of the <italic>A. indica</italic> genome was performed using MCScan (<xref ref-type="bibr" rid="B56">Tang et al., 2008</xref>; <xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 5</xref>). 242 homologous blocks were identified in the intragenomic gene synteny of <italic>A. indica</italic>, containing 2,281 gene pairs. These homologous blocks were distributed across the 14 chromosomes, covering 17.66% of protein-coding genes (4,139/25,767). The synonymous nucleotide substitutions (<italic>Ks</italic>) of the gene pairs peaked at approximately 0.01 and 1.12 (<xref ref-type="fig" rid="F3">Figure 3A</xref>). The first peak at approximately 1.12 indicated the core eudicot &#x03B3; triplication event (&#x223C;165 Mya). The second peak at approximately 0.01 indicated a relatively recent duplication event or events. To distinguish whether this peak represents a whole genome duplication event or background duplications, we performed synteny analysis on <italic>A. indica</italic>, <italic>V. vinifera</italic>, <italic>C. sinensis</italic>, and <italic>A. yangbiense</italic> genomes (<xref ref-type="fig" rid="F3">Figure 3B</xref> and <xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 6</xref>). Intergenomic collinearity analysis showed 611 homologous blocks containing 14,674 gene pairs and a 3:3 syntenic relationship between <italic>A. indica</italic> and <italic>V. vinifera</italic> (<xref ref-type="fig" rid="F3">Figure 3B</xref> and <xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 7A</xref>). Although there were 2:1 syntenic relationship between <italic>A. indica</italic> vs. <italic>C. sinensis</italic> and <italic>A. indica</italic> vs. <italic>A. yangbiense</italic> (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figures 7B,C</xref>), only 13 and 14% of the <italic>A. indica</italic> gene models in syntenic blocks, respectively, were present as two copies. Meanwhile, we did not identify large <italic>C. sinensis</italic> and <italic>A. yangbiense</italic> segments that have two syntenic copies in <italic>A. indica</italic> by the synteny dot plot of <italic>A. indica</italic> vs. <italic>C. sinensis</italic> and <italic>A. indica</italic> vs. <italic>A. yangbiense</italic> (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 6</xref>). Our analysis indicated that <italic>A. indica</italic> didn&#x2019;t experience additional WGD after the &#x03B3; event, but a recent small-scale segmental duplication (<xref ref-type="bibr" rid="B65">Xu et al., 2013</xref>; <xref ref-type="bibr" rid="B67">Yang et al., 2019</xref>). The calculation of <italic>Ks</italic> for <italic>A. indica</italic> vs. <italic>C. sinensis</italic> indicated that this recent segmental duplication event occurred approximately 1.5 Mya. Furthermore, we also performed paralog analysis in <italic>A. indica</italic> genome using reciprocal best hits (RBH) from primary protein sequences by all-vs.-all BLASTp matches. We detected 6,298 RBH paralogous gene pairs in the <italic>A. indica</italic> genome, and the RBH paralog <italic>Ks</italic> distribution shows a <italic>Ks</italic> peak at around 0.01 (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 8</xref>). That this RBH <italic>Ks</italic> peak is close to the syntelog <italic>Ks</italic> peak also indicates <italic>A. indica</italic> has a recent segmental duplication mixed with gene duplication.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>Genome evolution of <italic>A. indica</italic>. <bold>(A)</bold> <italic>Ks</italic> distribution (The gene pairs located in syntenic blocks) between syntenic genes within the <italic>A. indica</italic> genome or between genomes. <bold>(B)</bold> Inter-genomic syntenic analysis between <italic>A. indica</italic> genome and <italic>V. vinifera</italic> genome. <bold>(C)</bold> GO enrichment of the <italic>A. indica</italic> genome (green) and genes retained after recent gene duplication (orange) (<italic>P</italic> &#x003C; 0.05). &#x002A;Pearson chi-square test <italic>P</italic>-value &#x003C; 0.05; &#x002A;&#x002A;Pearson chi-square test <italic>P</italic>-value &#x003C; 0.01.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-853861-g003.tif"/>
</fig>
<p>Generally, gene duplication events vary the genomic architecture, including genome size, genome density, gene content, and gene expression. In this study, we defined the RBH paralogous gene pairs with the median <italic>Ks</italic> &#x003C; 0.05 as the retained genes from recent gene duplication. A total of 768 gene pairs were retained after recent gene duplication. GO analysis revealed that these gene pairs were significantly involved in binding, catalytic activity, metabolic process, cellular process, and reproductive process (<xref ref-type="fig" rid="F3">Figure 3C</xref>). Recent gene duplication may also affect the percentage of genes in many function categories with different contributions. In the <italic>A. indica</italic> genome, the percentage of retained genes from recent gene duplication in &#x201C;catalytic activity GO:0003824,&#x201D; &#x201C;recognition of pollen GO:0048544,&#x201D; &#x201C;pollen-pistil interaction GO:0009875,&#x201D; &#x201C;pollination GO:0009856,&#x201D; and &#x201C;multi-multicellular organism process GO:0044706&#x201D; was greater than that of the average genome content (<xref ref-type="fig" rid="F3">Figure 3C</xref> and <xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 9</xref>). We further calculated the omega values (<italic>Ka</italic>/<italic>Ks</italic>) for most of the homologous gene pairs. Most of the omega values for the homologous gene pairs were smaller than 1, which indicated that purifying selection may be the predominant action within the retained genes from recent gene duplication (<xref ref-type="bibr" rid="B55">Stix, 1992</xref>). However, 120 gene pairs were identified that have experienced potential positive selection. GO analysis showed that these genes were mainly enriched in &#x201C;catalytic activity, acting on a protein GO:0140096,&#x201D; &#x201C;protein binding GO:0005515,&#x201D; &#x201C;protein-containing complex GO:0032991,&#x201D; and &#x201C;membrane-bounded organelle GO:0043227&#x201D; (<xref ref-type="supplementary-material" rid="TS9">Supplementary Table 9</xref> and <xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 10</xref>).</p>
</sec>
<sec id="S3.SS6">
<title>Secondary Metabolite Analysis</title>
<p>Genes encoding some specialized metabolic pathways are found physically clustered in plant genomes (<xref ref-type="bibr" rid="B40">Nutzmann et al., 2016</xref>; <xref ref-type="bibr" rid="B34">Liu et al., 2020</xref>). We utilized the PlantiSMASH analytical pipeline (<xref ref-type="bibr" rid="B18">Hu et al., 2019</xref>) to identify physically clustered specialized metabolic pathway genes. According to the analysis, 50 clusters including 692 genes were identified in the <italic>A. indica</italic> genome (<xref ref-type="supplementary-material" rid="TS10">Supplementary Table 10</xref>). The sizes of the identified clusters range from 27.2 to 1634.4 kb. 105 (out of 692) clustered genes were contained in the 997 <italic>A. indica</italic>-specific expansion gene families (<xref ref-type="supplementary-material" rid="TS10">Supplementary Table 10</xref>). Furthermore, 41 (<italic>C. sinensis</italic>), 51 (<italic>A. yangbiense</italic>), 48 (<italic>T. cocoa</italic>), 47 (<italic>G. raimondii</italic>), 45 (<italic>A. thaliana</italic>), 35 (<italic>F. vesca</italic>), 33 (<italic>P. persica</italic>), 30 (<italic>C. sativus</italic>), 46 (<italic>V. vinifera</italic>), 47 (<italic>S. lycopersicum</italic>), and 29 (<italic>B. distachyon</italic>) clusters were detected in other 11 species (<xref ref-type="fig" rid="F4">Figure 4A</xref>). As expected, more terpene-related clusters were identified in the <italic>A. indica</italic> genome than that of other species.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p>Secondary metabolite analysis. <bold>(A)</bold> Secondary metabolite analysis of <italic>A. indica</italic> and other 11 species. <bold>(B)</bold> The organization and architecture of terpene-related gene clusters on chromosome 13. <bold>(C)</bold> Dot plot of the KEGG pathway enrichment analysis of genes on chromosome 13. <bold>(D)</bold> The GO enrichment analysis of genes on chromosome 13 (<italic>P</italic> &#x003C; 0.05).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-853861-g004.tif"/>
</fig>
<p>Azadirachtin is a triterpenoid compound of neem tree, which has effective insecticidal activities against a wide range of insect species, but has very low toxicity to mammals. Terpene synthase (TPS), cytochrome P450 (CYP450), alcohol dehydrogenase (ADH), acyltransferase (ACT), and esterase (EST) were proposed to be involved in biosynthesis of azadirachtin (<xref ref-type="bibr" rid="B64">Wang et al., 2020</xref>). In this study, a large number of genes encoding CYP 450s (78), TPSs (58), and ACTs (34) were identified in secondary metabolite biosynthesis gene clusters. Genes encoding ADHs, and ESTs may reside dispersed in the genome. The terpene-related clusters mainly distributed on chromosome 1, 2, 3, 5, 6, 7, 10, 11, 12, and 13. Four terpene-related clusters (cluster 18&#x2013;21) covering &#x223C;1.4 Mb were distributed on chromosome 13 (<xref ref-type="fig" rid="F4">Figure 4B</xref>). Among the 83 clustered terpene-related genes on chromosome 13, 12 genes were contained in the <italic>A. indica</italic>-specific expanded gene families. These genes are proposed to be potential genes participated in the terpene biosynthesis specific to <italic>A. indica</italic>.</p>
<p>KEGG enrichment analysis was performed to investigate the function of genes on chromosome 13. The result showed that genes on chromosome 13 were mainly involved in &#x201C;Protein processing in endoplasmic reticulum,&#x201D; &#x201C;Sesquiterpenoid and triterpenoid biosynthesis,&#x201D; and &#x201C;Ovarian steroidogenesis&#x201D; (<xref ref-type="fig" rid="F4">Figure 4C</xref>). Furthermore, when we performed GO enrichment analysis using the genes on chromosome 13, the genes associated with &#x201C;terpene synthase activity&#x201D; (GO:0010333) exhibited a low <italic>P</italic>-value, indicating that &#x201C;terpene synthase activity&#x201D; was the most enriched functional category of chromosome 13 (<xref ref-type="fig" rid="F4">Figure 4D</xref>).</p>
</sec>
<sec id="S3.SS7">
<title>Terpene Synthase Gene Family</title>
<p>TPS gene family is characterized by two large domains: PF01397 (Terpene synthase, N-terminal domain) and PF03936 (Terpene synthase family, metal binding domain). To investigate the characteristics and evolution of the TPS gene families, we identified a total of 512 putative TPS genes in <italic>A. indica</italic> and other 13 plant genome. 70 putative TPS genes were identified in <italic>A. indica</italic>; These consisted of 44 AziTPS genes containing both PF01397 and PF03936 domains, nine AziTPS genes containing PF01397 domain, and 17 AziTPS genes containing PF03936 domain. <italic>A. indica</italic> (<italic>N</italic> = 70) contained the most copies of TPSs compared with other plants, followed by <italic>C. sinensis</italic> (<italic>N</italic> = 49) and <italic>A. yangbiense</italic> (<italic>N</italic> = 57) (<xref ref-type="fig" rid="F5">Figure 5A</xref> and <xref ref-type="supplementary-material" rid="TS11">Supplementary Table 11</xref>). In addition, eight AziTPS genes experienced recent gene duplication (<xref ref-type="supplementary-material" rid="TS11">Supplementary Table 11</xref>).</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption><p>Analysis of TPS gene family in <italic>A. indica</italic>. <bold>(A)</bold> Statistics of predicted TPS genes in <italic>A. indica</italic> and other 13 plants. <bold>(B)</bold> Phylogenetic analysis of TPS family members based on the protein sequences. A Maximum-Likelihood (ML) tree was generated from an alignment of 409 TPS protein sequences, comprising 403 putative TPSs from <italic>A. indica</italic> and other 13 plants (the remaining TPS genes were too short for meaningful alignment) and other six reported TPS genes belonged to TPS- a (AAX16121.1), b (AAQ16588.1), c (AAD04292.1), e (Q39548.1), f (Q93YV0.1), and g (ADD81294.1) subfamilies. The <italic>A. indica</italic> TPS genes are highlighted by red stars. <bold>(C)</bold> <italic>A. indica</italic> TPS genes substantially expressed in five tested organs (root, flower, fruit, leaf, and stem). The color of TPS genes depends on the subfamily information.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-853861-g005.tif"/>
</fig>
<p>Phylogenetic analysis was performed using 403 TPSs (the remaining TPS genes were too short for meaningful alignment) from <italic>A. indica</italic> and other 13 plants, including six reported TPS genes belonged to TPS- a, b, c, e, f, and g subfamilies, respectively (<xref ref-type="supplementary-material" rid="TS11">Supplementary Table 11</xref>). As shown in <xref ref-type="fig" rid="F5">Figure 5B</xref>, the topology of six subfamilies is similar to that of the previous papers (<xref ref-type="bibr" rid="B29">Kumar et al., 2018b</xref>; <xref ref-type="bibr" rid="B72">Zhou et al., 2020</xref>; <xref ref-type="bibr" rid="B19">Ji et al., 2021</xref>). Among the 40 AziTPS used in phylogenetic analysis, 15, 13, 4, 3, 1, and 4 AziTPS genes fell in TPS- a, b, c, e, f, and g subfamilies, respectively. TPS-a and -b subfamilies were the main subfamilies in <italic>A. indica</italic>, approximately 37.5 and 32.5% of the total AziTPS genes in phylogenetic analysis. This is in accordance with other plant species, including tea, grape, and Chinese mahogany (<xref ref-type="bibr" rid="B6">Chen et al., 2011</xref>; <xref ref-type="bibr" rid="B72">Zhou et al., 2020</xref>; <xref ref-type="bibr" rid="B19">Ji et al., 2021</xref>). Furthermore, we identified putative <italic>A. indica</italic>-specific TPSs using phylogenetic analysis and a cutoff of 55% identity, which indicates separate subfamily assignment (<xref ref-type="bibr" rid="B33">Liu et al., 2018</xref>; <xref ref-type="bibr" rid="B59">Tu et al., 2020</xref>). A total of nine <italic>A. indica</italic>-specific TPS genes were identified (<xref ref-type="supplementary-material" rid="TS11">Supplementary Table 11</xref>). Interestingly, seven of these specific AziTPSs (Indica_007028, Indica_007047, Indica_007053, Indica_007068, Indica_007070, Indica_007072, and Indica_007143) were located in the terpene-related clusters (cluster 18, 19, and 20) of chromosome 13 (<xref ref-type="supplementary-material" rid="TS11">Supplementary Table 11</xref>).</p>
<p>We further investigated the expression pattern of TPS genes in <italic>A. indica</italic>. Transcriptome datasets from five tissues of <italic>A. indica</italic> were obtained from our previous study (<xref ref-type="bibr" rid="B64">Wang et al., 2020</xref>) and remapped to the chromosome-level genome assembly in this study. More than 88% of the RNAseq reads were mapped uniquely to the genome assembly across all samples (<xref ref-type="supplementary-material" rid="TS12">Supplementary Table 12</xref>). Transcripts of 27 TPS genes were detected in the tested tissues. Most of the detected transcripts exhibited a spatial-specific expression pattern (<xref ref-type="fig" rid="F5">Figure 5C</xref>). Nine, one, four, three, and four genes were exclusively expressed in flower, fruit, root, leaf, and stem, respectively. Seven genes (AziTPS30, &#x2212;48, &#x2212;5, &#x2212;57, &#x2212;26, &#x2212;63, and &#x2212;50) were primarily expressed in one or two tissues.</p>
</sec>
<sec id="S3.SS8">
<title>Cytochrome P450 Gene Family</title>
<p>The characteristics and evolution of the cytochrome P450 (CYP) gene families were also investigated. In total, 3,657 CYP genes were identified from all 14 plant genomes (<xref ref-type="fig" rid="F6">Figure 6A</xref> and <xref ref-type="supplementary-material" rid="TS13">Supplementary Table 13</xref>). A total of 355 CYP genes were in the <italic>A. indica</italic> genome, of which 36 CYP genes were involved in recent gene duplication (<xref ref-type="supplementary-material" rid="TS13">Supplementary Table 13</xref>). Moreover, 157 full length CYP (450 &#x003C; length &#x003C; 600) protein sequences of <italic>A. indica</italic> were aligned to construct a phylogenetic tree. As shown in <xref ref-type="fig" rid="F6">Figure 6B</xref>, the phylogenetic tree was divided into two major clades: A type (49%; 77/157) and non-A type (51%; 80/157); and further clustered into nine clans. The Clan 71 is the largest clan and comprises of 49% (77/157) members; 18, 4, 28, and 25 members are classified into Clan72, Clan74, Clan85, and Clan86; remaining Clan51, Clan710, Clan711, and Clan727 are single family clans.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption><p>Analysis of cytochrome P450 gene family in <italic>A. indica</italic>. <bold>(A)</bold> Statistics of predicted cytochrome P450 genes in <italic>A. indica</italic> and other 13 plants. <bold>(B)</bold> Phylogenetic analysis cytochrome P450 family members in <italic>A. indica</italic> based on the protein sequences. A ML tree was generated from an alignment of 156 <italic>A. indica</italic> cytochrome P450 protein sequences (450 &#x003C; length &#x003C; 600). The entire family of cytochrome P450 genes is shown for each clan next to the tree with different color. The <italic>A. indica</italic> P450 genes are highlighted by red stars. <bold>(C)</bold> <italic>A. indica</italic> CYP genes substantially expressed in five tested organs (fruit, flower, root, stem, and leaf).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-853861-g006.tif"/>
</fig>
<p>In order to identify putative <italic>A. indica</italic>-specific CYP genes, we constructed a phylogenetic tree using amino acid sequence alignment of 2,807 (450 &#x003C; length &#x003C; 600) CYP genes in <italic>A. indica</italic> and other 13 plants genome with a cutoff of 55% identity (<xref ref-type="bibr" rid="B33">Liu et al., 2018</xref>; <xref ref-type="bibr" rid="B59">Tu et al., 2020</xref>). Six <italic>A. indica-</italic>specific CYP genes were identified (<xref ref-type="supplementary-material" rid="TS13">Supplementary Table 13</xref>). Similar to TPS genes, five of these CYP genes (Indica_007272, Indica_007273, Indica_007276, Indica_007277, and Indica_007278) were located in the terpene-related cluster 21 of chromosome 13 (<xref ref-type="supplementary-material" rid="TS13">Supplementary Table 13</xref>). These specific-TPSs and CYPs in the terpene-related secondary metabolite biosynthesis gene clusters of chromosome 13 might be involved in the specific biosynthesis of azadirachtin.</p>
<p>We also investigated the expression pattern of <italic>A. indica</italic> CYP genes in different tissues (fruit, flower, root, stem, and leaf). Transcripts of 221 CYP genes were detected with different patterns (<xref ref-type="fig" rid="F6">Figure 6C</xref>). There were more high-expressed CYPs in fruit, stem and leaf than flower and root. The high-expressed CYPs in fruit, stem and leaf were 83, 88, and 97, respectively. CYPs with a high-expression in the tissues (fruit and leave) with high azadirachtin. A content, are more likely to be involved in azadirachtin biosynthesis. Furthermore, <italic>A. indica</italic>-specific AziCYP256 (Indica_007272) and AziCYP8 (Indica_007273) were highly expressed in fruit and flower.</p>
</sec>
<sec id="S3.SS9">
<title>Resistance Genes</title>
<p>Plants have developed a wide range of defense mechanisms to protect themselves against the attack of pathogens in their constant struggle for survival. In general, proteins encoded by resistance (R) genes display modular domain structures. In this study, putative R genes in the <italic>A. indica</italic> genome (1,488) and other 13 species were identified (<xref ref-type="supplementary-material" rid="TS14">Supplementary Table 14</xref>). In the <italic>A. indica</italic> genome, 238 R genes may exert their disease resistance function as cytoplasmic protein through canonical resistance domains, such as the nucleotide-binding sites (NBSs), the leucine-rich repeat (LRR), and terminal inverted repeat (TIR) domains (<xref ref-type="supplementary-material" rid="TS14">Supplementary Table 14</xref>). 167 NBS genes were identified in the <italic>A. indica</italic> genome, which could be divided into five classes according to the conserved domains: N, CN, CNL, NL, and TNL. The majority were N type which contained only the NB-ARC domain. In comparison with other genomes in malvids, most of the NBS genes in the <italic>A. indica</italic> genome were underrepresented relative to other Sapindales genomes (<italic>C. sinensis</italic> and <italic>A. yangbiense</italic>) and Malvales genomes (<italic>T. cacao</italic> and <italic>G. raimondii</italic>), but overrepresented relative to other Brassicales genome (<italic>A. thaliana</italic> and <italic>C. papaya</italic>). In addition, 447 genes were classified as transmembrane receptors, including 221 receptor-like kinases (RLK), and 226 receptor-like proteins (RLP). 721 putative kinases were also identified in the <italic>A. indica</italic> genome.</p>
</sec>
</sec>
<sec id="S4" sec-type="discussion">
<title>Discussion</title>
<p><italic>A. indica</italic> is a valuable plant species given its economic and pharmaceutical significance (<xref ref-type="bibr" rid="B55">Stix, 1992</xref>). A high-quality reference genome is essential for the genetic and genomic studies of <italic>A. indica</italic>. However, molecular-level studies on this species are limited. Here, we assembled the first chromosome-scale genome of <italic>A. indica</italic> by a combination of Illumina, PacBio, and Hi-C technology. The size of the genome assembly is approximately 281 Mb, with a scaffold N50 value of 19 Mb. The N50 of our assembled genome is much higher than that of the previous published draft genomes (<xref ref-type="bibr" rid="B26">Krishnan et al., 2012</xref>, <xref ref-type="bibr" rid="B25">2016</xref>; <xref ref-type="bibr" rid="B30">Kuravadi et al., 2015</xref>). Our assembled genome size covered &#x223C;73.2% of the estimated genome size (385 Mb) by flow cytometry. However, previously assembled 12 contig-level <italic>A. indica</italic> genomes were generally less than 300 Mb (<xref ref-type="bibr" rid="B25">Krishnan et al., 2016</xref>). The <italic>A. indica</italic> genome shows a high level of heterozygosity (0.896%) and repeat content (40.99%), rendering substantial challenges for its assembly (<xref ref-type="bibr" rid="B39">Nowak et al., 2015</xref>). Hi-C technology has been broadly available for many complex species (<xref ref-type="bibr" rid="B7">Chen et al., 2020</xref>). In this study, Hi-C technology facilitated the completeness and accuracy of a chromosome-level genome assembly for <italic>A. indica</italic>. The improvement of BUSCO evaluation shows that our assembly represents a better template for gene annotation than the reference sequence. Considering that the genome is highly heterozygous and repetitive, the present version represents a high-quality genome assembly. The obtained genome is also the second chromosome-level genome of the Meliaceae family, which will pave the way for further genetic and genomic studies of this family.</p>
<p>Gene duplication is an important evolutionary force that provides abundant raw materials for genetic novelty, morphological diversity and speciation (<xref ref-type="bibr" rid="B47">Qiao et al., 2018</xref>). In this study, we find no evidence that <italic>A. indica</italic> experienced WGD after the ancient &#x03B3; event shared by all eudicots. However, recent gene duplication events mixed with small-scale segmental duplication likely affected multiple genes in <italic>A. indica</italic>. This may also explain the fact that <italic>A. indica</italic> had more expanded gene families than <italic>C. sinensis</italic>. Our result is in agreement with the research of Chinese mahogany, which indicated that a recent WGD occurred in <italic>Toona sinensis</italic> (<xref ref-type="bibr" rid="B19">Ji et al., 2021</xref>). The occurrence of recent WGD mixed with gene duplications has been reported in <italic>Papaver somniferum</italic> L. genome (<xref ref-type="bibr" rid="B15">Guo et al., 2018</xref>). Furthermore, recent WGD was also observed in <italic>Panax notoginseng</italic> genome (<xref ref-type="bibr" rid="B20">Jiang et al., 2021</xref>). All these results are highly benefit for in-depth investigation of the survival and diversification history the of Meliaceae family.</p>
<p>Limonoids are natural triterpenoid products made by plants of the Meliaceae family. They are known for their insecticidal activity and potential pharmaceutical properties. <italic>A. indica</italic> is known as the reservoir of azadirachtin, the most famous limonoid insecticide. Secondary metabolite analysis revealed that <italic>A. indica</italic> contained more terpene-related clusters than that of the other 11 species. Eighty three (out of 247) clustered terpene-related genes were located on chromosome 13. The KEGG pathway enrichment analysis revealed that 33 genes were correlated with the &#x201C;Sesquiterpenoid and triterpenoid biosynthesis&#x201D; pathway. These results indicated that chromosome 13 may have played a central role in the evolution of terpenoid biosynthetic machinery in <italic>A. indica</italic>.</p>
<p>The TPS and CYP gene families are responsible for the biosynthesis of terpenoids in plants. 70 TPS genes were identified in <italic>A. indica</italic>, which is much more than that of the other 13 species. This is consistent with the result of Chinese mahogany (<italic>T. sinensis</italic>), the first chromosome-level genome assembly of the Meliaceae family (<xref ref-type="bibr" rid="B19">Ji et al., 2021</xref>). Furthermore, TPS genes have also been reported to be abundant in other angiosperms that are rich in terpenoids. For example, the <italic>Nymphaea colorata</italic> genome harbored 92 putative TPS genes, mainly consisting of copies from subfamily TPS-b, with no TPS-a copies (<xref ref-type="bibr" rid="B69">Zhang et al., 2020</xref>). In contrast, more than a dozen TPS-a genes were identified in the <italic>A. indica</italic> genome. These TPS-a genes might be responsible for sesquiterpene biosynthesis in <italic>A. indica</italic>. In addition, 355 CYP genes were identified in <italic>A. indica</italic>, six of which were <italic>A. indica</italic> specific CYPs. The expansion of terpene-related gene clusters, TPSs and CYPs, may promote the formation of terpenoids in <italic>A. indica</italic>. A total of eight TPS genes and 36 CYP genes were involved in recent gene duplication, suggesting that recent gene duplication event may have been responsible for terpenoid biosynthesis-related gene expansion in <italic>A. indica</italic>, after its split from <italic>C. sinensis</italic>. Notably, most of the identified <italic>A. indica</italic> -specific TPSs and CYPs were located in the terpene-related clusters on chromosome 13, indicating that these regions were likely to be involved in azadirachtin biosynthesis. This study provided the first chromosome-level genome of <italic>A. indica</italic>, and a genomic perspective for the synthesis and evolution of azadirachtin.</p>
</sec>
<sec id="S5" sec-type="data-availability">
<title>Data Availability Statement</title>
<p>Raw data from this study were deposited in the NCBI SRA (Sequence Read Archive) database under the Bioproject ID: <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="PRJNA645650">PRJNA645650</ext-link>. The genome sequence data (Illumina, PacBio, and Hi-C data) are available under accession numbers <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="SRR12315383">SRR12315383</ext-link>, <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="SRR12321691">SRR12321691</ext-link>, and <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="SRR12321285">SRR12321285</ext-link>. The assembled genome was submitted to DDBJ/ENA/GenBank with accession number <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="JAGQDM000000000">JAGQDM000000000</ext-link>.</p>
</sec>
<sec id="S6">
<title>Author Contributions</title>
<p>YD and WS designed the project and wrote the draft manuscript. WS participated in the genome assembly and annotation. YD, SW, JL, and ZY contributed to the genome evolution analysis, gene family analysis, and resistance gene identification. NW, HJ, JQ, and Y-XH revised the manuscript. All authors read and approved the final manuscript.</p>
</sec>
<sec id="conf1" sec-type="COI-statement">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="pudiscl1" sec-type="disclaimer">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<sec id="S7" sec-type="funding-information">
<title>Funding</title>
<p>This study was funded by the National Key R&#x0026;D Program of China (2017YFD0201400), the Fundamental Research Funds for the Central Universities, and the General Program of National Natural Science Foundation of China (31970622).</p>
</sec>
<sec id="S8" sec-type="supplementary-material">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2022.853861/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpls.2022.853861/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.DOCX" id="DS1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_1.xlsx" id="TS1" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_2.docx" id="TS2" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_3.docx" id="TS3" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_4.docx" id="TS4" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_5.docx" id="TS5" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_6.docx" id="TS6" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_7.docx" id="TS7" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_8.xlsx" id="TS8" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_9.xlsx" id="TS9" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_10.xlsx" id="TS10" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_11.xlsx" id="TS11" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_12.xlsx" id="TS12" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_13.xlsx" id="TS13" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_14.xlsx" id="TS14" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abdelhady</surname> <given-names>M. I. S.</given-names></name> <name><surname>Bader</surname> <given-names>A.</given-names></name> <name><surname>Shaheen</surname> <given-names>U.</given-names></name> <name><surname>El-Malah</surname> <given-names>Y.</given-names></name> <name><surname>Barghash</surname> <given-names>M. F.</given-names></name></person-group> (<year>2015</year>). <article-title>Azadirachta indica as a source for antioxidant and cytotoxic polyphenolic compounds.</article-title> <source><italic>Biosci. Biotechnol. Res. Asia</italic></source> <volume>12</volume> <fpage>1209</fpage>&#x2013;<lpage>1222</lpage>. <pub-id pub-id-type="doi">10.13005/bbra/1774</pub-id></citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Altschul</surname> <given-names>S. F.</given-names></name> <name><surname>Madden</surname> <given-names>T. L.</given-names></name> <name><surname>Schaffer</surname> <given-names>A. A.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Miller</surname> <given-names>W.</given-names></name><etal/></person-group> (<year>1997</year>). <article-title>Gapped BLAST and PSI-BLAST: a new generation of protein database search programs.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>25</volume> <fpage>3389</fpage>&#x2013;<lpage>3402</lpage>. <pub-id pub-id-type="doi">10.1093/nar/25.17.3389</pub-id> <pub-id pub-id-type="pmid">9254694</pub-id></citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Argout</surname> <given-names>X.</given-names></name> <name><surname>Salse</surname> <given-names>J.</given-names></name> <name><surname>Aury</surname> <given-names>J. M.</given-names></name> <name><surname>Guiltinan</surname> <given-names>M. J.</given-names></name> <name><surname>Droc</surname> <given-names>G.</given-names></name> <name><surname>Gouzy</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2011</year>). <article-title>The genome of <italic>Theobroma cacao</italic>.</article-title> <source><italic>Nat. Gene.</italic></source> <volume>43</volume> <fpage>101</fpage>&#x2013;<lpage>108</lpage>. <pub-id pub-id-type="doi">10.1038/ng.736</pub-id> <pub-id pub-id-type="pmid">21186351</pub-id></citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Birney</surname> <given-names>E.</given-names></name> <name><surname>Clamp</surname> <given-names>M.</given-names></name> <name><surname>Durbin</surname> <given-names>R.</given-names></name></person-group> (<year>2004</year>). <article-title>Genewise and genomewise.</article-title> <source><italic>Gen. Res.</italic></source> <volume>14</volume> <fpage>988</fpage>&#x2013;<lpage>995</lpage>. <pub-id pub-id-type="doi">10.1101/gr.1865504</pub-id> <pub-id pub-id-type="pmid">15123596</pub-id></citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Camacho</surname> <given-names>C.</given-names></name> <name><surname>Coulouris</surname> <given-names>G.</given-names></name> <name><surname>Avagyan</surname> <given-names>V.</given-names></name> <name><surname>Ma</surname> <given-names>N.</given-names></name> <name><surname>Papadopoulos</surname> <given-names>J.</given-names></name> <name><surname>Bealer</surname> <given-names>K.</given-names></name><etal/></person-group> (<year>2009</year>). <article-title>BLAST+: architecture and applications.</article-title> <source><italic>BMC Bioinform.</italic></source> <volume>10</volume>:<fpage>421</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-10-421</pub-id> <pub-id pub-id-type="pmid">20003500</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>F.</given-names></name> <name><surname>Tholl</surname> <given-names>D.</given-names></name> <name><surname>Bohlmann</surname> <given-names>J.</given-names></name> <name><surname>Pichersky</surname> <given-names>E.</given-names></name></person-group> (<year>2011</year>). <article-title>The family of terpene synthases in plants: a mid-size family of genes for specialized metabolism that is highly diversified throughout the kingdom.</article-title> <source><italic>Plant J.</italic></source> <volume>66</volume> <fpage>212</fpage>&#x2013;<lpage>229</lpage>. <pub-id pub-id-type="doi">10.1111/j.1365-313X.2011.04520.x</pub-id> <pub-id pub-id-type="pmid">21443633</pub-id></citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>J. D.</given-names></name> <name><surname>Zheng</surname> <given-names>C.</given-names></name> <name><surname>Ma</surname> <given-names>J. Q.</given-names></name> <name><surname>Jiang</surname> <given-names>C. K.</given-names></name> <name><surname>Ercisli</surname> <given-names>S.</given-names></name> <name><surname>Yao</surname> <given-names>M. Z.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>The chromosome-scale genome reveals the evolution and diversification after the recent tetraploidization event in tea plant.</article-title> <source><italic>Hortic. Res.</italic></source> <volume>7</volume>:<fpage>11</fpage>. <pub-id pub-id-type="doi">10.1038/s41438-020-0288-2</pub-id> <pub-id pub-id-type="pmid">32377354</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dai</surname> <given-names>J. M.</given-names></name> <name><surname>Yaylayan</surname> <given-names>V. A.</given-names></name> <name><surname>Raghavan</surname> <given-names>G. S. V.</given-names></name> <name><surname>Pare</surname> <given-names>J. R.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name></person-group> (<year>2001</year>). <article-title>Multivariate calibration for the determination of total azadirachtin-related limonoids and simple terpenoids in neem extracts using vanillin assay.</article-title> <source><italic>J. Agric. Food Chem.</italic></source> <volume>49</volume> <fpage>1169</fpage>&#x2013;<lpage>1174</lpage>. <pub-id pub-id-type="doi">10.1021/jf001141n</pub-id> <pub-id pub-id-type="pmid">11312830</pub-id></citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>De Bie</surname> <given-names>T.</given-names></name> <name><surname>Cristianini</surname> <given-names>N.</given-names></name> <name><surname>Demuth</surname> <given-names>J. P.</given-names></name> <name><surname>Hahn</surname> <given-names>M. W.</given-names></name></person-group> (<year>2006</year>). <article-title>CAFE: a computational tool for the study of gene family evolution.</article-title> <source><italic>Bioinformatics</italic></source> <volume>22</volume> <fpage>1269</fpage>&#x2013;<lpage>1271</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btl097</pub-id> <pub-id pub-id-type="pmid">16543274</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Edgar</surname> <given-names>R. C.</given-names></name></person-group> (<year>2004</year>). <article-title>MUSCLE: multiple sequence alignment with high accuracy and high throughput.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>32</volume> <fpage>1792</fpage>&#x2013;<lpage>1797</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkh340</pub-id> <pub-id pub-id-type="pmid">15034147</pub-id></citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ellinghaus</surname> <given-names>D.</given-names></name> <name><surname>Kurtz</surname> <given-names>S.</given-names></name> <name><surname>Willhoeft</surname> <given-names>U.</given-names></name></person-group> (<year>2008</year>). <article-title>LTRharvest, an efficient and flexible software for de novo detection of LTR retrotransposons.</article-title> <source><italic>BMC Bioinform.</italic></source> <volume>9</volume>:<fpage>14</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-9-18</pub-id> <pub-id pub-id-type="pmid">18194517</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Emms</surname> <given-names>D. M.</given-names></name> <name><surname>Kelly</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>OrthoFinder: phylogenetic orthology inference for comparative genomics.</article-title> <source><italic>Genome Biol.</italic></source> <volume>20</volume>:<fpage>238</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-019-1832-y</pub-id> <pub-id pub-id-type="pmid">31727128</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Griffiths-Jones</surname> <given-names>S.</given-names></name> <name><surname>Moxon</surname> <given-names>S.</given-names></name> <name><surname>Marshall</surname> <given-names>M.</given-names></name> <name><surname>Khanna</surname> <given-names>A.</given-names></name> <name><surname>Eddy</surname> <given-names>S. R.</given-names></name> <name><surname>Bateman</surname> <given-names>A.</given-names></name></person-group> (<year>2005</year>). <article-title>Rfam: annotating non-coding RNAs in complete genomes.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>33</volume> <fpage>D121</fpage>&#x2013;<lpage>D124</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gki081</pub-id> <pub-id pub-id-type="pmid">15608160</pub-id></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guindon</surname> <given-names>S.</given-names></name> <name><surname>Delsuc</surname> <given-names>F.</given-names></name> <name><surname>Dufayard</surname> <given-names>J. F.</given-names></name> <name><surname>Gascuel</surname> <given-names>O.</given-names></name></person-group> (<year>2009</year>). <article-title>Estimating maximum likelihood phylogenies with PhyML.</article-title> <source><italic>Methods Mol. Biol.</italic></source> <volume>537</volume> <fpage>113</fpage>&#x2013;<lpage>137</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-59745-251-9_6</pub-id></citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>L.</given-names></name> <name><surname>Winzer</surname> <given-names>T.</given-names></name> <name><surname>Yang</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Ning</surname> <given-names>Z.</given-names></name> <name><surname>He</surname> <given-names>Z.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>The opium poppy genome and morphinan production.</article-title> <source><italic>Science</italic></source> <volume>362</volume> <fpage>343</fpage>&#x2013;<lpage>347</lpage>. <pub-id pub-id-type="doi">10.1126/science.aat4096</pub-id> <pub-id pub-id-type="pmid">30166436</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hodgson</surname> <given-names>H.</given-names></name> <name><surname>De La Pena</surname> <given-names>R.</given-names></name> <name><surname>Stephenson</surname> <given-names>M. J.</given-names></name> <name><surname>Thimmappa</surname> <given-names>R.</given-names></name> <name><surname>Vincent</surname> <given-names>J. L.</given-names></name> <name><surname>Sattely</surname> <given-names>E. S.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Identification of key enzymes responsible for protolimonoid biosynthesis in plants: opening the door to azadirachtin production.</article-title> <source><italic>Proc. Natl. Acad. Sci. U.S.A</italic></source> <volume>116</volume> <fpage>17096</fpage>&#x2013;<lpage>17104</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1906083116</pub-id> <pub-id pub-id-type="pmid">31371503</pub-id></citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Holt</surname> <given-names>C.</given-names></name> <name><surname>Yandell</surname> <given-names>M.</given-names></name></person-group> (<year>2011</year>). <article-title>MAKER2: an annotation pipeline and genome-database management tool for second-generation genome projects.</article-title> <source><italic>BMC Bioinform.</italic></source> <volume>12</volume>:<fpage>14</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-12-491</pub-id> <pub-id pub-id-type="pmid">22192575</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>L.</given-names></name> <name><surname>Xu</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>M.</given-names></name> <name><surname>Fan</surname> <given-names>R.</given-names></name> <name><surname>Yuan</surname> <given-names>D.</given-names></name> <name><surname>Wu</surname> <given-names>B.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>The chromosome-scale reference genome of black pepper provides insight into piperine biosynthesis.</article-title> <source><italic>Nat. Commun.</italic></source> <volume>10</volume>:<fpage>4702</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-019-12607-6</pub-id> <pub-id pub-id-type="pmid">31619678</pub-id></citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ji</surname> <given-names>Y. T.</given-names></name> <name><surname>Xiu</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>C. H.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>J. X.</given-names></name> <name><surname>Sui</surname> <given-names>J. J.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Long read sequencing of <italic>Toona sinensis</italic> (a. juss) roem: a chromosome-level reference genome for the family meliaceae.</article-title> <source><italic>Mol. Ecol. Resour</italic>.</source> <volume>21</volume> <fpage>1243</fpage>&#x2013;<lpage>1255</lpage>. <pub-id pub-id-type="doi">10.1111/1755-0998.13318</pub-id> <pub-id pub-id-type="pmid">33421343</pub-id></citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>Z.</given-names></name> <name><surname>Tu</surname> <given-names>L.</given-names></name> <name><surname>Yang</surname> <given-names>W.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Hu</surname> <given-names>T.</given-names></name> <name><surname>Ma</surname> <given-names>B.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>The chromosome-level reference genome assembly for <italic>Panax notoginseng</italic> and insights into ginsenoside biosynthesis.</article-title> <source><italic>Plant Commun.</italic></source> <volume>2</volume>:<fpage>100113</fpage>. <pub-id pub-id-type="doi">10.1016/j.xplc.2020.100113</pub-id> <pub-id pub-id-type="pmid">33511345</pub-id></citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kautsar</surname> <given-names>S. A.</given-names></name> <name><surname>Suarez Duran</surname> <given-names>H. G.</given-names></name> <name><surname>Blin</surname> <given-names>K.</given-names></name> <name><surname>Osbourn</surname> <given-names>A.</given-names></name> <name><surname>Medema</surname> <given-names>M. H.</given-names></name></person-group> (<year>2017</year>). <article-title>plantiSMASH: automated identification, annotation and expression analysis of plant biosynthetic gene clusters.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>45</volume> <fpage>W55</fpage>&#x2013;<lpage>W63</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkx305</pub-id> <pub-id pub-id-type="pmid">28453650</pub-id></citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>D.</given-names></name> <name><surname>Langmead</surname> <given-names>B.</given-names></name> <name><surname>Salzberg</surname> <given-names>S. L.</given-names></name></person-group> (<year>2015</year>). <article-title>HISAT: a fast spliced aligner with low memory requirements.</article-title> <source><italic>Nat. Methods</italic></source> <volume>12</volume> <fpage>357</fpage>&#x2013;<lpage>360</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.3317</pub-id> <pub-id pub-id-type="pmid">25751142</pub-id></citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Koren</surname> <given-names>S.</given-names></name> <name><surname>Walenz</surname> <given-names>B. P.</given-names></name> <name><surname>Berlin</surname> <given-names>K.</given-names></name> <name><surname>Miller</surname> <given-names>J. R.</given-names></name> <name><surname>Bergman</surname> <given-names>N. H.</given-names></name> <name><surname>Phillippy</surname> <given-names>A. M.</given-names></name></person-group> (<year>2017</year>). <article-title>Canu: scalable and accurate long-read assembly <italic>via</italic> adaptive k-mer weighting and repeat separation.</article-title> <source><italic>Genome Res.</italic></source> <volume>27</volume> <fpage>722</fpage>&#x2013;<lpage>736</lpage>. <pub-id pub-id-type="doi">10.1101/gr.215087.116</pub-id> <pub-id pub-id-type="pmid">28298431</pub-id></citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Korf</surname> <given-names>I.</given-names></name></person-group> (<year>2004</year>). <article-title>Gene finding in novel genomes.</article-title> <source><italic>BMC Bioinform.</italic></source> <volume>5</volume>:<fpage>59</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-5-59</pub-id> <pub-id pub-id-type="pmid">15144565</pub-id></citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krishnan</surname> <given-names>N. M.</given-names></name> <name><surname>Jain</surname> <given-names>P.</given-names></name> <name><surname>Gupta</surname> <given-names>S.</given-names></name> <name><surname>Hariharan</surname> <given-names>A. K.</given-names></name> <name><surname>Panda</surname> <given-names>B.</given-names></name></person-group> (<year>2016</year>). <article-title>An improved genome assembly of <italic>Azadirachta indica</italic> a. juss.</article-title> <source><italic>G3 (Bethesda)</italic></source> <volume>6</volume> <fpage>1835</fpage>&#x2013;<lpage>1840</lpage>. <pub-id pub-id-type="doi">10.1534/g3.116.030056</pub-id> <pub-id pub-id-type="pmid">27172223</pub-id></citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krishnan</surname> <given-names>N. M.</given-names></name> <name><surname>Pattnaik</surname> <given-names>S.</given-names></name> <name><surname>Jain</surname> <given-names>P.</given-names></name> <name><surname>Gaur</surname> <given-names>P.</given-names></name> <name><surname>Choudhary</surname> <given-names>R.</given-names></name> <name><surname>Vaidyanathan</surname> <given-names>S.</given-names></name><etal/></person-group> (<year>2012</year>). <article-title>A draft of the genome and four transcriptomes of a medicinal and pesticidal angiosperm <italic>Azadirachta indica</italic>.</article-title> <source><italic>BMC Genomics</italic></source> <volume>13</volume>:<fpage>13</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-13-464</pub-id> <pub-id pub-id-type="pmid">22958331</pub-id></citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kumar</surname> <given-names>S.</given-names></name> <name><surname>Stecher</surname> <given-names>G.</given-names></name> <name><surname>Li</surname> <given-names>M.</given-names></name> <name><surname>Knyaz</surname> <given-names>C.</given-names></name> <name><surname>Tamura</surname> <given-names>K.</given-names></name></person-group> (<year>2018a</year>). <article-title>MEGA X: molecular evolutionary genetics analysis across computing platforms.</article-title> <source><italic>Mol. Biol. Evol.</italic></source> <volume>35</volume> <fpage>1547</fpage>&#x2013;<lpage>1549</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msy096</pub-id> <pub-id pub-id-type="pmid">29722887</pub-id></citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kumar</surname> <given-names>S.</given-names></name> <name><surname>Stecher</surname> <given-names>G.</given-names></name> <name><surname>Suleski</surname> <given-names>M.</given-names></name> <name><surname>Hedges</surname> <given-names>S. B.</given-names></name></person-group> (<year>2017</year>). <article-title>Timetree: a resource for timelines. timetrees, and divergence times.</article-title> <source><italic>Mol. Biol. Evol.</italic></source> <volume>34</volume> <fpage>1812</fpage>&#x2013;<lpage>1819</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msx116</pub-id> <pub-id pub-id-type="pmid">28387841</pub-id></citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kumar</surname> <given-names>Y.</given-names></name> <name><surname>Khan</surname> <given-names>F.</given-names></name> <name><surname>Rastogi</surname> <given-names>S.</given-names></name> <name><surname>Shasany</surname> <given-names>A. K.</given-names></name></person-group> (<year>2018b</year>). <article-title>Genome-wide detection of terpene synthase genes in holy basil (<italic>Ocimum sanctum L.</italic>).</article-title> <source><italic>PLoS One</italic></source> <volume>13</volume>:<fpage>e0207097</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0207097</pub-id> <pub-id pub-id-type="pmid">30444870</pub-id></citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kuravadi</surname> <given-names>N. A.</given-names></name> <name><surname>Yenagi</surname> <given-names>V.</given-names></name> <name><surname>Rangiah</surname> <given-names>K.</given-names></name> <name><surname>Mahesh</surname> <given-names>H. B.</given-names></name> <name><surname>Rajamani</surname> <given-names>A.</given-names></name> <name><surname>Shirke</surname> <given-names>M. D.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>Comprehensive analyses of genomes, transcriptomes and metabolites of neem tree.</article-title> <source><italic>Peerj</italic></source> <volume>3</volume>:<fpage>25</fpage>. <pub-id pub-id-type="doi">10.7717/peerj.1066</pub-id> <pub-id pub-id-type="pmid">26290780</pub-id></citation></ref>
<ref id="B31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Langmead</surname> <given-names>B.</given-names></name> <name><surname>Salzberg</surname> <given-names>S. L.</given-names></name></person-group> (<year>2012</year>). <article-title>Fast gapped-read alignment with bowtie 2.</article-title> <source><italic>Nat. Methods</italic></source> <volume>9</volume> <fpage>357</fpage>&#x2013;<lpage>359</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.1923</pub-id> <pub-id pub-id-type="pmid">22388286</pub-id></citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ley</surname> <given-names>S. V.</given-names></name></person-group> (<year>1994</year>). <article-title>Synthesis and chemistry of the insect antifeedant azadirachtin.</article-title> <source><italic>Pure Appl. Chem.</italic></source> <volume>66</volume> <fpage>2099</fpage>&#x2013;<lpage>2102</lpage>. <pub-id pub-id-type="doi">10.1351/pac199466102099</pub-id></citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Cheng</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>G.</given-names></name> <name><surname>Ding</surname> <given-names>W.</given-names></name> <name><surname>Duan</surname> <given-names>L.</given-names></name> <name><surname>Yang</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>Engineering yeast for the production of breviscapine by genomic analysis and synthetic biology approaches.</article-title> <source><italic>Nat. Commun.</italic></source> <volume>9</volume>:<fpage>448</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-018-02883-z</pub-id> <pub-id pub-id-type="pmid">29386648</pub-id></citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Suarez Duran</surname> <given-names>H. G.</given-names></name> <name><surname>Harnvanichvech</surname> <given-names>Y.</given-names></name> <name><surname>Stephenson</surname> <given-names>M. J.</given-names></name> <name><surname>Schranz</surname> <given-names>M. E.</given-names></name> <name><surname>Nelson</surname> <given-names>D.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Drivers of metabolic diversification: how dynamic genomic neighbourhoods generate new biosynthetic pathways in the brassicaceae.</article-title> <source><italic>New Phytol.</italic></source> <volume>227</volume> <fpage>1109</fpage>&#x2013;<lpage>1123</lpage>. <pub-id pub-id-type="doi">10.1111/nph.16338</pub-id> <pub-id pub-id-type="pmid">31769874</pub-id></citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lowe</surname> <given-names>T. M.</given-names></name> <name><surname>Eddy</surname> <given-names>S. R.</given-names></name></person-group> (<year>1997</year>). <article-title>tRNAscan-SE: a program for improved detection of transfer RNA genes in genomic sequence.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>25</volume> <fpage>955</fpage>&#x2013;<lpage>964</lpage>. <pub-id pub-id-type="doi">10.1093/nar/25.5.955</pub-id> <pub-id pub-id-type="pmid">9023104</pub-id></citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mithilesh</surname> <given-names>S.</given-names></name> <name><surname>Rakhi</surname> <given-names>C.</given-names></name></person-group> (<year>2014</year>). <article-title>Sustainable production of azadirachtin from differentiated <italic>in vitro</italic> cell lines of neem.</article-title> <source><italic>AoB Plants</italic></source> <volume>5</volume> <fpage>lt034</fpage>&#x2013;<lpage>lt034</lpage>.</citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Moriya</surname> <given-names>Y.</given-names></name> <name><surname>Itoh</surname> <given-names>M.</given-names></name> <name><surname>Okuda</surname> <given-names>S.</given-names></name> <name><surname>Yoshizawa</surname> <given-names>A. C.</given-names></name> <name><surname>Kanehisa</surname> <given-names>M.</given-names></name></person-group> (<year>2007</year>). <article-title>KAAS: an automatic genome annotation and pathway reconstruction server.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>35</volume> <fpage>W182</fpage>&#x2013;<lpage>W185</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkm321</pub-id> <pub-id pub-id-type="pmid">17526522</pub-id></citation></ref>
<ref id="B38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nawrocki</surname> <given-names>E. P.</given-names></name> <name><surname>Eddy</surname> <given-names>S. R.</given-names></name></person-group> (<year>2013</year>). <article-title>Infernal 1.1: 100-fold faster RNA homology searches.</article-title> <source><italic>Bioinform.</italic></source> <volume>29</volume> <fpage>2933</fpage>&#x2013;<lpage>2935</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btt509</pub-id> <pub-id pub-id-type="pmid">24008419</pub-id></citation></ref>
<ref id="B39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nowak</surname> <given-names>M. D.</given-names></name> <name><surname>Russo</surname> <given-names>G.</given-names></name> <name><surname>Schlapbach</surname> <given-names>R.</given-names></name> <name><surname>Huu</surname> <given-names>C. N.</given-names></name> <name><surname>Lenhard</surname> <given-names>M.</given-names></name> <name><surname>Conti</surname> <given-names>E.</given-names></name></person-group> (<year>2015</year>). <article-title>The draft genome of <italic>Primula veris</italic> yields insights into the molecular basis of heterostyly.</article-title> <source><italic>Genome Biol.</italic></source> <volume>16</volume>:<fpage>16</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-014-0567-z</pub-id> <pub-id pub-id-type="pmid">25651398</pub-id></citation></ref>
<ref id="B40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nutzmann</surname> <given-names>H. W.</given-names></name> <name><surname>Huang</surname> <given-names>A.</given-names></name> <name><surname>Osbourn</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). <article-title>Plant metabolic clusters - from genetics to genomics.</article-title> <source><italic>New Phytol.</italic></source> <volume>211</volume> <fpage>771</fpage>&#x2013;<lpage>789</lpage>. <pub-id pub-id-type="doi">10.1111/nph.13981</pub-id> <pub-id pub-id-type="pmid">27112429</pub-id></citation></ref>
<ref id="B41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Osuna-Cruz</surname> <given-names>C. M.</given-names></name> <name><surname>Paytuvi-Gallart</surname> <given-names>A.</given-names></name> <name><surname>Di Donato</surname> <given-names>A.</given-names></name> <name><surname>Sundesha</surname> <given-names>V.</given-names></name> <name><surname>Andolfo</surname> <given-names>G.</given-names></name> <name><surname>Cigliano</surname> <given-names>R. A.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>PRGdb 3.0: a comprehensive platform for prediction and analysis of plant disease resistance genes.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>46</volume> <fpage>D1197</fpage>&#x2013;<lpage>D1201</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkx1119</pub-id> <pub-id pub-id-type="pmid">29156057</pub-id></citation></ref>
<ref id="B42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ou</surname> <given-names>S.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Jiang</surname> <given-names>N.</given-names></name></person-group> (<year>2018</year>). <article-title>Assessing genome assembly quality using the LTR Assembly Index (LAI).</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>46</volume>:<fpage>e126</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gky730</pub-id> <pub-id pub-id-type="pmid">30107434</pub-id></citation></ref>
<ref id="B43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ou</surname> <given-names>S. J.</given-names></name> <name><surname>Jiang</surname> <given-names>N.</given-names></name></person-group> (<year>2018</year>). <article-title>LTR_retriever: a highly accurate and sensitive program for identification of long terminal repeat retrotransposons.</article-title> <source><italic>Plant Physiol.</italic></source> <volume>176</volume> <fpage>1410</fpage>&#x2013;<lpage>1422</lpage>. <pub-id pub-id-type="doi">10.1104/pp.17.01310</pub-id> <pub-id pub-id-type="pmid">29233850</pub-id></citation></ref>
<ref id="B44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pellicer</surname> <given-names>J.</given-names></name> <name><surname>Leitch</surname> <given-names>I. J.</given-names></name></person-group> (<year>2020</year>). <article-title>The Plant DNA C-values database (release 7.1): an updated online repository of plant genome size data for comparative studies.</article-title> <source><italic>New Phytol.</italic></source> <volume>226</volume> <fpage>301</fpage>&#x2013;<lpage>305</lpage>. <pub-id pub-id-type="doi">10.1111/nph.16261</pub-id> <pub-id pub-id-type="pmid">31608445</pub-id></citation></ref>
<ref id="B45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Powell</surname> <given-names>S.</given-names></name> <name><surname>Szklarczyk</surname> <given-names>D.</given-names></name> <name><surname>Trachana</surname> <given-names>K.</given-names></name> <name><surname>Roth</surname> <given-names>A.</given-names></name> <name><surname>Kuhn</surname> <given-names>M.</given-names></name> <name><surname>Muller</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2012</year>). <article-title>eggNOG v3.0: orthologous groups covering 1133 organisms at 41 different taxonomic ranges.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>40</volume> <fpage>D284</fpage>&#x2013;<lpage>D289</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkr1060</pub-id> <pub-id pub-id-type="pmid">22096231</pub-id></citation></ref>
<ref id="B46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Price</surname> <given-names>A. L.</given-names></name> <name><surname>Jones</surname> <given-names>N. C.</given-names></name> <name><surname>Pevzner</surname> <given-names>P. A.</given-names></name></person-group> (<year>2005</year>). <article-title>De novo identification of repeat families in large genomes.</article-title> <source><italic>Bioinform.</italic></source> <volume>21</volume> <fpage>I351</fpage>&#x2013;<lpage>I358</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bti1018</pub-id> <pub-id pub-id-type="pmid">15961478</pub-id></citation></ref>
<ref id="B47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qiao</surname> <given-names>X.</given-names></name> <name><surname>Yin</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>R.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>Different modes of gene duplication show divergent evolutionary patterns and contribute differently to the expansion of gene families involved in important fruit traits in pear (<italic>Pyrus bretschneideri</italic>).</article-title> <source><italic>Front. Plant Sci.</italic></source> <volume>9</volume>:<fpage>161</fpage>. <pub-id pub-id-type="doi">10.3389/fpls.2018.00161</pub-id> <pub-id pub-id-type="pmid">29487610</pub-id></citation></ref>
<ref id="B48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rodrigues</surname> <given-names>M.</given-names></name> <name><surname>Festucci-Buselli</surname> <given-names>R. A.</given-names></name> <name><surname>Silva</surname> <given-names>L. C.</given-names></name> <name><surname>Otoni</surname> <given-names>W. C.</given-names></name></person-group> (<year>2014</year>). <article-title>Azadirachtin biosynthesis induction in <italic>Azadirachta indica</italic> a. juss cotyledonary calli with elicitor agents.</article-title> <source><italic>Braz. Arch. Biol. Technol.</italic></source> <volume>57</volume> <fpage>155</fpage>&#x2013;<lpage>162</lpage>. <pub-id pub-id-type="doi">10.1590/s1516-89132014000200001</pub-id></citation></ref>
<ref id="B49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schmutterer</surname> <given-names>H.</given-names></name></person-group> (<year>1995</year>). <article-title>The neem tree, <italic>Azadirachta indica</italic> a. juss. and other meliaceous plants: source of unique natural products for integrated pest management, medicine, industry and other purposes.</article-title> <source><italic>Pap. Bibliogr. Soc. Am.</italic></source> <volume>107</volume> <fpage>1365</fpage>&#x2013;<lpage>1372</lpage>.</citation></ref>
<ref id="B50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Servant</surname> <given-names>N.</given-names></name> <name><surname>Varoquaux</surname> <given-names>N.</given-names></name> <name><surname>Lajoie</surname> <given-names>B. R.</given-names></name> <name><surname>Viara</surname> <given-names>E.</given-names></name> <name><surname>Chen</surname> <given-names>C. J.</given-names></name> <name><surname>Vert</surname> <given-names>J. P.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>HiC-Pro: an optimized and flexible pipeline for Hi-C data processing.</article-title> <source><italic>Genome Biol.</italic></source> <volume>16</volume>:<fpage>11</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-015-0831-x</pub-id> <pub-id pub-id-type="pmid">26619908</pub-id></citation></ref>
<ref id="B51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Simao</surname> <given-names>F. A.</given-names></name> <name><surname>Waterhouse</surname> <given-names>R. M.</given-names></name> <name><surname>Ioannidis</surname> <given-names>P.</given-names></name> <name><surname>Kriventseva</surname> <given-names>E. V.</given-names></name> <name><surname>Zdobnov</surname> <given-names>E. M.</given-names></name></person-group> (<year>2015</year>). <article-title>BUSCO: assessing genome assembly and annotation completeness with single-copy orthologs.</article-title> <source><italic>Bioinformatics</italic></source> <volume>31</volume> <fpage>3210</fpage>&#x2013;<lpage>3212</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv351</pub-id> <pub-id pub-id-type="pmid">26059717</pub-id></citation></ref>
<ref id="B52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Soares</surname> <given-names>D.</given-names></name> <name><surname>Godin</surname> <given-names>A.</given-names></name> <name><surname>Menezes</surname> <given-names>R.</given-names></name> <name><surname>Nogueira</surname> <given-names>R.</given-names></name> <name><surname>Brito</surname> <given-names>A.</given-names></name> <name><surname>Melo</surname> <given-names>I.</given-names></name><etal/></person-group> (<year>2014</year>). <article-title>Anti-inflammatory and antinociceptive activities of azadirachtin in mice.</article-title> <source><italic>Planta Med.</italic></source> <volume>80</volume> <fpage>630</fpage>&#x2013;<lpage>636</lpage>. <pub-id pub-id-type="doi">10.1055/s-0034-1368507</pub-id> <pub-id pub-id-type="pmid">24871207</pub-id></citation></ref>
<ref id="B53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Srivastava</surname> <given-names>S.</given-names></name> <name><surname>Srivastava</surname> <given-names>A. K.</given-names></name></person-group> (<year>2013</year>). <article-title>Production of the biopesticide azadirachtin by hairy root cultivation of azadirachta indica in liquid-phase bioreactors.</article-title> <source><italic>Appl. Biochem. Biotechnol.</italic></source> <volume>171</volume> <fpage>1351</fpage>&#x2013;<lpage>1361</lpage>. <pub-id pub-id-type="doi">10.1007/s12010-013-0432-7</pub-id> <pub-id pub-id-type="pmid">23955295</pub-id></citation></ref>
<ref id="B54"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stanke</surname> <given-names>M.</given-names></name> <name><surname>Steinkamp</surname> <given-names>R.</given-names></name> <name><surname>Waack</surname> <given-names>S.</given-names></name> <name><surname>Morgenstern</surname> <given-names>B.</given-names></name></person-group> (<year>2004</year>). <article-title>AUGUSTUS: a web server for gene finding in eukaryotes.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>32</volume> <fpage>W309</fpage>&#x2013;<lpage>W312</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkh379</pub-id> <pub-id pub-id-type="pmid">15215400</pub-id></citation></ref>
<ref id="B55"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stix</surname> <given-names>G.</given-names></name></person-group> (<year>1992</year>). <article-title>Village pharmacy. the neem tree yields products from pesticides to soap.</article-title> <source><italic>Sci. Am.</italic></source> <volume>266</volume>:<fpage>132</fpage>. <pub-id pub-id-type="doi">10.1038/scientificamerican0592-132</pub-id></citation></ref>
<ref id="B56"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tang</surname> <given-names>H.</given-names></name> <name><surname>Bowers</surname> <given-names>J. E.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Ming</surname> <given-names>R.</given-names></name> <name><surname>Alam</surname> <given-names>M.</given-names></name> <name><surname>Paterson</surname> <given-names>A. H.</given-names></name></person-group> (<year>2008</year>). <article-title>Synteny and collinearity in plant genomes.</article-title> <source><italic>Science</italic></source> <volume>320</volume> <fpage>486</fpage>&#x2013;<lpage>488</lpage>. <pub-id pub-id-type="doi">10.1126/science.1153917</pub-id> <pub-id pub-id-type="pmid">18436778</pub-id></citation></ref>
<ref id="B57"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tarailo-Graovac</surname> <given-names>M.</given-names></name> <name><surname>Chen</surname> <given-names>N.</given-names></name></person-group> (<year>2009</year>). <article-title>Using repeatmasker to identify repetitive elements in genomic sequences.</article-title> <source><italic>Curr. protoc. Bioinform.</italic></source> <volume>Chapter 4</volume> <fpage>Unit 4.10.</fpage> <pub-id pub-id-type="doi">10.1002/0471250953.bi0410s25</pub-id> <pub-id pub-id-type="pmid">19274634</pub-id></citation></ref>
<ref id="B58"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Trapnell</surname> <given-names>C.</given-names></name> <name><surname>Roberts</surname> <given-names>A.</given-names></name> <name><surname>Goff</surname> <given-names>L.</given-names></name> <name><surname>Pertea</surname> <given-names>G.</given-names></name> <name><surname>Kim</surname> <given-names>D.</given-names></name> <name><surname>Kelley</surname> <given-names>D. R.</given-names></name><etal/></person-group> (<year>2012</year>). <article-title>Differential gene and transcript expression analysis of RNA-seq experiments with tophat and cufflinks.</article-title> <source><italic>Nat. Protoc.</italic></source> <volume>7</volume> <fpage>562</fpage>&#x2013;<lpage>578</lpage>. <pub-id pub-id-type="doi">10.1038/nprot.2012.016</pub-id> <pub-id pub-id-type="pmid">22383036</pub-id></citation></ref>
<ref id="B59"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tu</surname> <given-names>L.</given-names></name> <name><surname>Su</surname> <given-names>P.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Gao</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Hu</surname> <given-names>T.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Genome of Tripterygium wilfordii and identification of cytochrome P450 involved in triptolide biosynthesis.</article-title> <source><italic>Nat. Commun.</italic></source> <volume>11</volume> <fpage>971</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-020-14776-1</pub-id> <pub-id pub-id-type="pmid">32080175</pub-id></citation></ref>
<ref id="B60"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaser</surname> <given-names>R.</given-names></name> <name><surname>Sovic</surname> <given-names>I.</given-names></name> <name><surname>Nagarajan</surname> <given-names>N.</given-names></name> <name><surname>Sikic</surname> <given-names>M.</given-names></name></person-group> (<year>2017</year>). <article-title>Fast and accurate de novo genome assembly from long uncorrected reads.</article-title> <source><italic>Genome Res.</italic></source> <volume>27</volume> <fpage>737</fpage>&#x2013;<lpage>746</lpage>. <pub-id pub-id-type="doi">10.1101/gr.214270.116</pub-id> <pub-id pub-id-type="pmid">28100585</pub-id></citation></ref>
<ref id="B61"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Veitch</surname> <given-names>G. E.</given-names></name> <name><surname>Beckmann</surname> <given-names>E.</given-names></name> <name><surname>Burke</surname> <given-names>B. J.</given-names></name> <name><surname>Boyer</surname> <given-names>A.</given-names></name> <name><surname>Maslen</surname> <given-names>S. L.</given-names></name> <name><surname>Ley</surname> <given-names>S. V.</given-names></name></person-group> (<year>2007</year>). <article-title>Synthesis of azadirachtin: a long but successful journey.</article-title> <source><italic>Angew. Chem. Int. Ed Engl.</italic></source> <volume>46</volume> <fpage>7629</fpage>&#x2013;<lpage>7632</lpage>. <pub-id pub-id-type="doi">10.1002/anie.200703027</pub-id> <pub-id pub-id-type="pmid">17665403</pub-id></citation></ref>
<ref id="B62"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Walker</surname> <given-names>B. J.</given-names></name> <name><surname>Abeel</surname> <given-names>T.</given-names></name> <name><surname>Shea</surname> <given-names>T.</given-names></name> <name><surname>Priest</surname> <given-names>M.</given-names></name> <name><surname>Abouelliel</surname> <given-names>A.</given-names></name> <name><surname>Sakthikumar</surname> <given-names>S.</given-names></name><etal/></person-group> (<year>2014</year>). <article-title>Pilon: an integrated tool for comprehensive microbial variant detection and genome assembly improvement.</article-title> <source><italic>PLoS One</italic></source> <volume>9</volume>:<fpage>14</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0112963</pub-id> <pub-id pub-id-type="pmid">25409509</pub-id></citation></ref>
<ref id="B63"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>D.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Zhu</surname> <given-names>J.</given-names></name> <name><surname>Yu</surname> <given-names>J.</given-names></name></person-group> (<year>2010</year>). <article-title>KaKs_calculator 2.0: a toolkit incorporating gamma-series methods and sliding window strategies.</article-title> <source><italic>Genomics Proteomics Bioinform.</italic></source> <volume>8</volume> <fpage>77</fpage>&#x2013;<lpage>80</lpage>. <pub-id pub-id-type="doi">10.1016/s1672-0229(10)60008-3</pub-id></citation></ref>
<ref id="B64"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>N.</given-names></name> <name><surname>Huo</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>Multi-tissue transcriptome analysis using hybrid-sequencing reveals potential genes and biological pathways associated with azadirachtin a biosynthesis in neem (<italic>azadirachta indica</italic>).</article-title> <source><italic>BMC Genomics</italic></source> <volume>21</volume>:<fpage>749</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-020-07124-6</pub-id> <pub-id pub-id-type="pmid">33115410</pub-id></citation></ref>
<ref id="B65"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>Q.</given-names></name> <name><surname>Chen</surname> <given-names>L. L.</given-names></name> <name><surname>Ruan</surname> <given-names>X. A.</given-names></name> <name><surname>Chen</surname> <given-names>D. J.</given-names></name> <name><surname>Zhu</surname> <given-names>A. D.</given-names></name> <name><surname>Chen</surname> <given-names>C. L.</given-names></name><etal/></person-group> (<year>2013</year>). <article-title>The draft genome of sweet orange (Citrus sinensis).</article-title> <source><italic>Nat. Genetics</italic></source> <volume>45</volume> <fpage>59</fpage>&#x2013;<lpage>U92</lpage>. <pub-id pub-id-type="doi">10.1038/ng.2472</pub-id> <pub-id pub-id-type="pmid">23179022</pub-id></citation></ref>
<ref id="B66"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name></person-group> (<year>2007</year>). <article-title>LTR_FINDER: an efficient tool for the prediction of full-length LTR retrotransposons.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>35</volume> <fpage>W265</fpage>&#x2013;<lpage>W268</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkm286</pub-id> <pub-id pub-id-type="pmid">17485477</pub-id></citation></ref>
<ref id="B67"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>J.</given-names></name> <name><surname>Wariss</surname> <given-names>H. M.</given-names></name> <name><surname>Tao</surname> <given-names>L. D.</given-names></name> <name><surname>Zhang</surname> <given-names>R. G.</given-names></name> <name><surname>Yun</surname> <given-names>Q. Z.</given-names></name> <name><surname>Hollingsworth</surname> <given-names>P.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>De novo genome assembly of the endangered <italic>Acer yangbiense</italic>, a plant species with extremely small populations endemic to yunnan province.</article-title> <source><italic>China. Gigascience</italic></source> <volume>8</volume>:<fpage>10</fpage>. <pub-id pub-id-type="doi">10.1093/gigascience/giz085</pub-id> <pub-id pub-id-type="pmid">31307060</pub-id></citation></ref>
<ref id="B68"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>Z.</given-names></name></person-group> (<year>1997</year>). <article-title>PAML: a program package for phylogenetic analysis by maximum likelihood.</article-title> <source><italic>Comput. Appl. Biosci.</italic></source> <volume>13</volume> <fpage>555</fpage>&#x2013;<lpage>556</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/13.5.555</pub-id> <pub-id pub-id-type="pmid">9367129</pub-id></citation></ref>
<ref id="B69"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Chen</surname> <given-names>F.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Zhao</surname> <given-names>Y.</given-names></name> <name><surname>Lohaus</surname> <given-names>R.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>The water lily genome and the early evolution of flowering plants.</article-title> <source><italic>Nature</italic></source> <volume>577</volume> <fpage>79</fpage>&#x2013;<lpage>84</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-019-1852-5</pub-id> <pub-id pub-id-type="pmid">31853069</pub-id></citation></ref>
<ref id="B70"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>X. T.</given-names></name> <name><surname>Zhang</surname> <given-names>S. C.</given-names></name> <name><surname>Zhao</surname> <given-names>Q.</given-names></name> <name><surname>Ming</surname> <given-names>R.</given-names></name> <name><surname>Tang</surname> <given-names>H. B.</given-names></name></person-group> (<year>2019</year>). <article-title>Assembly of allele-aware, chromosomal-scale autopolyploid genomes based on Hi-C data.</article-title> <source><italic>Nat. Plants</italic></source> <volume>5</volume> <fpage>833</fpage>&#x2013;<lpage>845</lpage>. <pub-id pub-id-type="doi">10.1038/s41477-019-0487-8</pub-id> <pub-id pub-id-type="pmid">31383970</pub-id></citation></ref>
<ref id="B71"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Xiao</surname> <given-names>J.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Liu</surname> <given-names>G.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name><etal/></person-group> (<year>2012</year>). <article-title>ParaAT: a parallel tool for constructing multiple protein-coding DNA alignments.</article-title> <source><italic>Biochem. Biophys. Res. Commun.</italic></source> <volume>419</volume> <fpage>779</fpage>&#x2013;<lpage>781</lpage>. <pub-id pub-id-type="doi">10.1016/j.bbrc.2012.02.101</pub-id> <pub-id pub-id-type="pmid">22390928</pub-id></citation></ref>
<ref id="B72"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>H. C.</given-names></name> <name><surname>Shamala</surname> <given-names>L. F.</given-names></name> <name><surname>Yi</surname> <given-names>X. K.</given-names></name> <name><surname>Yan</surname> <given-names>Z.</given-names></name> <name><surname>Wei</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>Analysis of terpene synthase family genes in <italic>Camellia sinensis</italic> with an emphasis on abiotic stress conditions.</article-title> <source><italic>Sci. Rep.</italic></source> <volume>10</volume>:<fpage>933</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-57805-1</pub-id> <pub-id pub-id-type="pmid">31969641</pub-id></citation></ref>
</ref-list>
<fn-group>
<fn id="footnote1">
<label>1</label>
<p><ext-link ext-link-type="uri" xlink:href="http://cegg.unige.ch/orthodb">http://cegg.unige.ch/orthodb</ext-link></p></fn>
<fn id="footnote2">
<label>2</label>
<p><ext-link ext-link-type="uri" xlink:href="https://github.com/galaxyproject/tools-iuc/tree/master/tools/minimap2">https://github.com/galaxyproject/tools-iuc/tree/master/tools/minimap2</ext-link></p></fn>
<fn id="footnote3">
<label>3</label>
<p><ext-link ext-link-type="uri" xlink:href="https://github.com/tpoorten/dotPlotly">https://github.com/tpoorten/dotPlotly</ext-link></p></fn>
<fn id="footnote4">
<label>4</label>
<p><ext-link ext-link-type="uri" xlink:href="https://www.girinst.org/repbase/">https://www.girinst.org/repbase/</ext-link></p></fn>
<fn id="footnote5">
<label>5</label>
<p><ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/protein/">https://www.ncbi.nlm.nih.gov/protein/</ext-link></p></fn>
<fn id="footnote6">
<label>6</label>
<p><ext-link ext-link-type="uri" xlink:href="https://www.ebi.ac.uk/interpro/">https://www.ebi.ac.uk/interpro/</ext-link></p></fn>
<fn id="footnote7">
<label>7</label>
<p><ext-link ext-link-type="uri" xlink:href="http://www.uniprot.org">http://www.uniprot.org</ext-link></p></fn>
</fn-group>
</back>
</article>