<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">842387</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.842387</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>PolyReco: A Method to Automatically Label Collinear Regions and Recognize Polyploidy Events Based on the <italic>K</italic>
<sub>
<italic>S</italic>
</sub> Dotplot</article-title>
<alt-title alt-title-type="left-running-head">Wang et al.</alt-title>
<alt-title alt-title-type="right-running-head">PolyReco</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Fushun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1601412/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Kang</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Ruolan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Hongquan</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Weijin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jia</surname>
<given-names>Zhanxiao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wang</surname>
<given-names>Chunyang</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1695473/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Information Science and Technology</institution>, <institution>Hebei Agricultural University</institution>, <addr-line>Baoding</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Hebei Key Laboratory of Agricultural Big Data</institution>, <addr-line>Baoding</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Life Science</institution>, <institution>Hebei Agricultural University</institution>, <addr-line>Baoding</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>State Key Laboratory of North China Crop Improvement and Regulation</institution>, <institution>Hebei Agricultural University</institution>, <addr-line>Baoding</addr-line>, <country>China</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Hebei Key Laboratory of Plant Physiology and Molecular Pathology</institution>, <institution>Hebei Agricultural University</institution>, <addr-line>Baoding</addr-line>, <country>China</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Department of Urban and Rural Construction</institution>, <institution>Hebei Agricultural University</institution>, <addr-line>Baoding</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/79898/overview">Manish Kumar Pandey</ext-link>, International Crops Research Institute for the Semi-Arid Tropics (ICRISAT), India</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/121109/overview">Victor A. Albert</ext-link>, University at Buffalo, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/258856/overview">Aamir W. Khan</ext-link>, University of Missouri, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Chunyang Wang, <email>shmwcy@hebau.edu.cn</email>
</corresp>
<fn fn-type="equal" id="fn1">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors share first authorship between correspondence and specialty section</p>
</fn>
<fn fn-type="other">
<p>This article was submitted to Plant Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>20</day>
<month>04</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>842387</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>12</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>03</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Wang, Zhang, Zhang, Liu, Zhang, Jia and Wang.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Wang, Zhang, Zhang, Liu, Zhang, Jia and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Polyploidization plays a critical role in producing new gene functions and promoting species evolution. Effective identification of polyploid types can be helpful in exploring the evolutionary mechanism. However, current methods for detecting polyploid types have some major limitations, such as being time-consuming and strong subjectivity, etc. In order to objectively and scientifically recognize collinearity fragments and polyploid types, we developed PolyReco method, which can automatically label collinear regions and recognize polyploidy events based on the <italic>K</italic>
<sub>
<italic>S</italic>
</sub> dotplot. Combining with whole-genome collinearity analysis, PolyReco uses DBSCAN clustering method to cluster <italic>K</italic>
<sub>
<italic>S</italic>
</sub> dots. According to the distance information in the <italic>x</italic>-axis and <italic>y</italic>-axis directions between the categories, the clustering results are merged based on certain rules to obtain the collinear regions, automatically recognize and label collinear fragments. According to the information of the labeled collinear regions on the <italic>y</italic>-axis, the polyploidization recognition algorithm is used to exhaustively combine and obtain the genetic collinearity evaluation index of each combination, and then draw the genetic collinearity evaluation index graph. Based on the inflection point on the graph, polyploid types and related chromosomes with polyploidy signal can be detected. The validation experiments showed that the conclusions of PolyReco were consistent with the previous study, which verified the effectiveness of this method. It is expected that this approach can become a reference architecture for other polyploid types classification methods.</p>
</abstract>
<kwd-group>
<kwd>clustering</kwd>
<kwd>collinearity fragment</kwd>
<kwd>polyploidy</kwd>
<kwd>DBSCAN</kwd>
<kwd>chromosome</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>Studying the process of polyploidization is essential for the in-depth understanding of evolutionary laws (<xref ref-type="bibr" rid="B6">Marcet-Houben and Gabald&#xf3;n 2015</xref>), and exploring the stability and chromosome rearrangement of the genome. Polyploidization of gymnosperms and almost all angiosperms are considered to be the main reason for the diversity of land plants (<xref ref-type="bibr" rid="B5">Li et al., 2016</xref>; <xref ref-type="bibr" rid="B4">Hao et al., 2017</xref>). Polyploidy can produce a large number of duplicated genes in the genome (<xref ref-type="bibr" rid="B11">Wang et al., 2018</xref>). These genes may play an important role in functional evolution, environmental adaptation, and new species formation (<xref ref-type="bibr" rid="B12">Wang et al., 2017</xref>; <xref ref-type="bibr" rid="B13">Wang et al., 2019</xref>). The recombination of some homologous chromosomes after polyploidization often causes the instability of the genome structure, and processes such as chromosome breakage and fusion often occur, which can lead to large-scale duplicated gene loss in the genome (<xref ref-type="bibr" rid="B15">Wang et al., 2007</xref>; <xref ref-type="bibr" rid="B14">Wang et al., 2011</xref>). If two species have a common ancestor, after polyploidization, although there will be differences between the genomes, the two species still have a relatively close relationship. This close relationship can be expressed in the form of collinearity. The more complete the collinearity fragment, the closer relationship between the two species is. Exploring the collinearity between species has big significance in understanding the origin of species and the evolution of the genome.</p>
<p>
<xref ref-type="bibr" rid="B1">Cheng et al. (2019)</xref> drew a <italic>K</italic>
<sub>
<italic>S</italic>
</sub> dotplot of homologous genes within Spirogloea muscicola, and found that it had recently experienced a whole genome triplication event. Through collinearity analysis, <xref ref-type="bibr" rid="B14">Wang et al., 2011</xref> found that <italic>Brassica rapa</italic> and <italic>Arabidopsis thaliana</italic> experienced a whole genome triplication event. By constructing a phylogenetic gene tree, <xref ref-type="bibr" rid="B2">Dong et al. (2021)</xref> revealed that a whole-genome duplication event occurred in Magnoliales and Laurales. <xref ref-type="bibr" rid="B18">Xu et al. (2020)</xref> by drawing a phylogenetic gene tree, found that Scutellaria baicalensis and Scutellaria barbata had a whole-genome duplication event about 13.28 million years ago. <xref ref-type="bibr" rid="B19">Yan et al., 2021</xref> by drawing the distribution graph of the synonymous substitution rate (<italic>K</italic>
<sub>
<italic>S</italic>
</sub>), found two WGD events in Juglans mandshurica and Juglans regia. By drawing the distribution graph of the synonymous substitution rate (<italic>K</italic>
<sub>
<italic>S</italic>
</sub>).</p>
<p>Although the <italic>K</italic>
<sub>
<italic>S</italic>
</sub> distribution graph combined with the molecular clock (<xref ref-type="bibr" rid="B7">Miyata et al., 1980</xref>) can calculate the doubling time, it is a challenge to determine the collinearity information between the chromosomes. In addition, the above method, which injects prior knowledge, manually marks the collinear area by observing the atlas, and then recognizes the polyploidization through the combination of the regions. This kind of recognition method has low recognition efficiency, high dependence on prior knowledge, as well as strong subjectivity, and easy to introduce human error. Due to the lack of objective evaluation criteria, the identification of the polyploid types of is still very challenging. In terms of the types of polyploidization and the choice of chromosomes, the same atlas will cause different personal perceptions. This deviation will affect the subsequent research on chromosome rearrangement (<xref ref-type="bibr" rid="B21">Zhang et al., 2021</xref>). Therefore, we develop a computational model PolyReco to accurately identify and characterize some polyploid types in atlas.</p>
<p>Considering only <italic>K</italic>
<sub>
<italic>S</italic>
</sub> values for identifying polyploid types might be insufficient, we add the gene positional information in PolyReco. Genes are aligned in sequential order on each of the chromosomes, so incorporating the gene positional information on chromosomes with <italic>K</italic>
<sub>
<italic>S</italic>
</sub> values will likely increase the accuracy of polyploid type classification. In this study, sequence comparisons were performed based on the whole genome data of <italic>Vitis vinifera</italic> and <italic>Salix sinopurpurea</italic>, combined with whole genome collinearity analysis, to obtain the summary data of homology information and <italic>K</italic>
<sub>
<italic>S</italic>
</sub> values between genomes. PolyReco comprehensively utilizes digital image processing technology and DBSCAN method, and realizes the automatic recognition and labeling of the collinear region based on the <italic>K</italic>
<sub>
<italic>S</italic>
</sub> dotplot of homologous genes. The model uses the collinear area as the unit and combines the combination strategy to construct the combination evaluation standard. According to the performance of the chromosome combination, determine the specific polyploidization and draw the combined figure of the polyploidization. This study aims to develop a polyploidization classification tool, which has the potential to take chromosome position information into account with the <italic>K</italic>
<sub>
<italic>S</italic>
</sub> values for boosting polyploid type prediction performance.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and Methods</title>
<sec id="s2-1">
<title>Data Sources</title>
<p>With the whole genome data of <italic>Salix sinopurpurea</italic> (Spu) and <italic>Brassica rapa</italic> (Bra) as the main research materials, comparative genomics was used to compare the collinearity between <italic>Salix sinopurpurea</italic> and the reference genome <italic>Vitis vinifera</italic> (Vvi), <italic>Brassica rapa</italic> and the reference genome <italic>Arabidopsis thaliana</italic> (Ath).</p>
<p>Genomes and their gene annotations of <italic>Salix sinopurpurea</italic> and <italic>Vitis vinifera</italic> were downloaded from Joint Genome Institute. Download the required documents for <italic>Brassica rapa</italic> and <italic>Arabidopsis thaliana</italic> at <ext-link ext-link-type="uri" xlink:href="http://brassicadb.org/">http://brassicadb.org/</ext-link>and <ext-link ext-link-type="uri" xlink:href="http://www.arabidopsis.org/shangxia">http://www.arabidopsis.org/shangxia</ext-link>, respectively.</p>
</sec>
<sec id="s2-2">
<title>Preprocessing of the Data Sources</title>
<p>Due to the huge amount of original genome data, in order to extract target data from the genome sequence and annotation files, the downloaded genome data is processed with a custom python script to obtain the blast results, which is convenient for subsequent research and analysis. Screen the original data of species genomes, information was extracted from the genome annotation files, which include chromosome number, gene start and end positions, gene transcription direction, and gene ID information, and then rename the gene ID and the number of the genes was given in order of their appearance on chromosomes. Map the gene ID in the CDS sequence and protein sequence file to the new ID of the corresponding gene in the genome annotation file. Label the processed genomic data with a unified naming method.</p>
</sec>
<sec id="s2-3">
<title>Homologous Sequence Alignment</title>
<p>Blastp was used to explore to align genomic sequences of different species. Screen out gene pairs with the expected value (E-value) not greater than 10&#x2013;5 and score evaluation (Score) higher than 100, so that the subsequent genome collinearity analysis results are more reliable.</p>
</sec>
<sec id="s2-4">
<title>Draw the <italic>K</italic>
<sub>
<italic>S</italic>
</sub> Dotplot of Homologous Genes</title>
<p>The WGDI (<xref ref-type="bibr" rid="B10">Sun et al., 2021</xref>) use MAFFT (<xref ref-type="bibr" rid="B17">Wong, Suchard, and Huelsenbeck 2008</xref>) or MUSCLE (<xref ref-type="bibr" rid="B3">Edgar 2004</xref>) to perform multiple sequence alignment, and calculates the synonymous substitution rate using the yn00 (<xref ref-type="bibr" rid="B20">Yang et al., 2000</xref>) or ng86 (<xref ref-type="bibr" rid="B8">Nei and Gojobori 1986</xref>) program of the PAML package. Finally, the visualization is realized by extracting block, and then output blockinfo file.</p>
</sec>
<sec id="s2-5">
<title>Collinear Fragment Labeling Method Based on Clustering</title>
<p>In this paper, the input for DBSCAN requires the blockinfo file generated by WGDI and the chromosome length information (len file) of the two species. By setting the epsilon (eps) and minimum points (MinPts), cluster analysis is performed on the collinearity fragments in the <italic>K</italic>
<sub>
<italic>S</italic>
</sub> dotplot. The collinear region was then obtained from the clustering results combined with certain rules for merging. And then realizes the automatic identification and labeling of the collinear region. The comparison result of a chromosome of the target species and a chromosome of the reference species is shown as a cell on the <italic>K</italic>
<sub>
<italic>S</italic>
</sub> dotplot, that is, a comparison unit.</p>
<p>The <italic>K</italic>
<sub>
<italic>S</italic>
</sub> dotplot between <italic>Salix sinopurpurea</italic> and <italic>Vitis vinifera</italic> genome homologous genes drawn by wgdi (<xref ref-type="fig" rid="F1">Figure 1A</xref>). The horizontal axis represents the chromosome of the target species (<italic>Salix sinopurpurea</italic>) and the vertical represents the chromosome of the reference species (<italic>Vitis vinifera</italic>). On the <italic>K</italic>
<sub>
<italic>S</italic>
</sub> dotplot, the chromosome number of <italic>Salix sinopurpurea</italic> is shown from left to right, and the chromosome number of <italic>Vitis vinifera</italic> is shown from top to bottom. The <italic>K</italic>
<sub>
<italic>S</italic>
</sub> value ranges from 0.00 to 2.00. As shown in the figure, different colored points correspond to different <italic>K</italic>
<sub>
<italic>S</italic>
</sub> values. It can be observed in <xref ref-type="fig" rid="F1">Figure 1A</xref>, in addition to the clear and complete homologous fragments of grape chromosome 4 with <italic>Salix sinopurpurea</italic> chromosomes 6 and 18, it also has fuzzy and unclear homologous fragments with <italic>Salix sinopurpurea</italic> chromosomes 1, 2 and 4. The reason why these fragments are unclear and incomplete is that they are doubled by the whole genome triplication events shared by older dicots. The collinearity of the homologous fragments produced by the whole genome triplication events shared by ancient dicotyledons is far inferior to that of the whole genome duplication events shared by the Salicaceae. The specific manifestation is that the <italic>K</italic>
<sub>
<italic>S</italic>
</sub> value is significantly large, belongs to the blue-purple system, scarce and fragmented seriously. The results showed the structural similarities and differences between genomes. The generated data and pictures provide references for follow-up research.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>
<bold>(A)</bold> <italic>K</italic>
<sub>
<italic>S</italic>
</sub> dotplot between Salix sinopurpurea and Vitis vinifera genome homologous genes <bold>(B)</bold> DBSCAN cluster recognition effect figure <bold>(C)</bold> Automatic label result of collinearity fragments based on DBSCAN.</p>
</caption>
<graphic xlink:href="fgene-13-842387-g001.tif"/>
</fig>
<p>Using the DBSCAN algorithm, by setting the eps to 50 and the MinPts to 3, cluster the <italic>K</italic>
<sub>
<italic>S</italic>
</sub> dotplot between <italic>Salix sinopurpurea</italic> and <italic>Vitis vinifera</italic> genome homologous genes. The algorithm outputs the clustering result figure (<xref ref-type="fig" rid="F1">Figure 1B</xref>), in which each category is represented by a rectangular box. The DBSCAN can cluster out complete collinearity fragments, in grape chromosome 14 and <italic>Salix sinopurpurea</italic> chromosome 5, as well as in grape chromosome 14 and <italic>Salix sinopurpurea</italic> chromosome 15. It also can identify fragmented collinearity fragments in grape chromosome 4 and <italic>Salix sinopurpurea</italic> chromosome 5, grape chromosome 13 and <italic>Salix sinopurpurea</italic> chromosome 16. These will accurately reflect the relationship between the collinearity fragments and improve the subsequent combination effect.</p>
<p>The model sorts the category in the same comparison unit from top to bottom to generate ID, and calculates the number of homologous gene points in the box (num), the length of the box (<italic>y</italic>1-<italic>y</italic>2), and the width of the box (<italic>x</italic>1-<italic>x</italic>2). The model then generates cluster.csv files that contain the target species chromosome number (chr1), the reference species chromosome number (chr2), ID, and the horizontal and vertical coordinates of the upper left corner point are l_<italic>x</italic> and l_<italic>y</italic>, respectively, num, the horizontal and vertical coordinates of the lower right point are r_<italic>x</italic> and r_<italic>y</italic>, respectively, <italic>y</italic>1-<italic>y</italic>2, and <italic>x</italic>1-<italic>x</italic>2. Part of the data in the cluster.csv file of grape chromosome 13 is shown in <xref ref-type="table" rid="T1">Table 1</xref>. Among them, chromosome 13 and chromosome 1 form a class. The coordinates of the upper left corner of this class are 145, 354, and the coordinates of the lower right corner are 257, 228. The number of homologous genes contained is 32. The length of the cluster box is 126 coordinate lengths, and the width is 112 coordinate lengths.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Partial data of grape chromosome 13 cluster.csv file.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">chr1</th>
<th align="center">chr2</th>
<th align="center">id</th>
<th align="center">l_x</th>
<th align="center">l_y</th>
<th align="center">num</th>
<th align="center">r_x</th>
<th align="center">r_y</th>
<th align="center">y1-y2</th>
<th align="center">x1-x2</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="center">13</td>
<td align="center">1</td>
<td align="center">145</td>
<td align="center">354</td>
<td align="center">32</td>
<td align="center">257</td>
<td align="center">228</td>
<td align="center">126</td>
<td align="center">112</td>
</tr>
<tr>
<td align="left">4</td>
<td align="center">13</td>
<td align="center">1</td>
<td align="center">1,266</td>
<td align="center">1,153</td>
<td align="center">10</td>
<td align="center">1,310</td>
<td align="center">1,097</td>
<td align="center">56</td>
<td align="center">44</td>
</tr>
<tr>
<td align="left">6</td>
<td align="center">13</td>
<td align="center">1</td>
<td align="center">838</td>
<td align="center">1,267</td>
<td align="center">17</td>
<td align="center">913</td>
<td align="center">1,097</td>
<td align="center">170</td>
<td align="center">75</td>
</tr>
<tr>
<td align="left">6</td>
<td align="center">13</td>
<td align="center">2</td>
<td align="center">735</td>
<td align="center">994</td>
<td align="center">26</td>
<td align="center">834</td>
<td align="center">840</td>
<td align="center">154</td>
<td align="center">99</td>
</tr>
<tr>
<td align="left">8</td>
<td align="center">13</td>
<td align="center">1</td>
<td align="center">592</td>
<td align="center">1,267</td>
<td align="center">56</td>
<td align="center">671</td>
<td align="center">1,155</td>
<td align="center">112</td>
<td align="center">79</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In order to perform a combined analysis on the identified collinearity fragments, the model read the cluster.csv file generated by clustering. The model uses <italic>y</italic>_gap and <italic>x</italic>_gap, which represents the gap in the longitudinal and horizontal directions of adjacent collinear segments, as the basis for judging overlap. The location information of the gene is combined to set the parameters gap and Slen. In the comparison unit, the parameter gap represents the mean value of <italic>y</italic>_gap. Through a series of experiments and continuous optimization of parameter selection, it is finally determined that 1/6 of the corresponding chromosome length of the target species is the value of parameter Slen. For all collinearity fragments whose num is greater than the specified value, one condition is that the collinearity fragments do not overlap, another is overlap. In the first condition, collinear fragments will be merged if 0 &#x2264;<italic>y</italic>_gap &#x2264; gap and 0 &#x2264;<italic>x</italic>_gap &#x2264; Slen. And in the second condition, there is overlap in the <italic>y</italic>-axis direction, they will be merged when it meets 0 &#x2264; <italic>x</italic>_gap &#x3c; Slen; if there is overlap in the <italic>x</italic>-axis direction, when it meets 0 &#x2264; <italic>y</italic>_gap &#x2264; gap, merge them. Finally, the model output the clustering result graph (<xref ref-type="fig" rid="F1">Figure 1C</xref>), in which the merged result is marked with a rectangular box, and the combine.csv file is generated.</p>
<p>The content of the combine.csv file is the same as the cluster.csv file. <xref ref-type="table" rid="T2">Table 2</xref> shows some data of the combine.csv file. Among them, chromosome 13 and chromosome 8 form two classes. The coordinates of the upper left corner of the first class are 592, 1,264, and the lower right corner are 671, 1,155. The number of homologous genes contained in this class is 56. The length of the labeled box &#x2206;y is 112 coordinate length, and the width &#x2206;x is 79 coordinate length; the upper left corner coordinate of the second class are 78, 1,089, and the lower right corner are 605, 446. The number of homologous genes contained in this class is 239, &#x2206;y is 643 coordinate length, &#x2206;x is 537.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Partial data of grape chromosome 13 combine.csv file.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">chr1</th>
<th align="center">chr2</th>
<th align="center">id</th>
<th align="center">l_x</th>
<th align="center">l_y</th>
<th align="center">num</th>
<th align="center">r_x</th>
<th align="center">r_y</th>
<th align="center">&#x2206;y</th>
<th align="center">&#x2206;x</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">8</td>
<td align="center">13</td>
<td align="center">1</td>
<td align="center">592</td>
<td align="center">1,267</td>
<td align="center">56</td>
<td align="center">671</td>
<td align="center">1,155</td>
<td align="center">112</td>
<td align="center">79</td>
</tr>
<tr>
<td align="left">8</td>
<td align="center">13</td>
<td align="center">2</td>
<td align="center">78</td>
<td align="center">1,089</td>
<td align="center">239</td>
<td align="center">605</td>
<td align="center">446</td>
<td align="center">643</td>
<td align="center">527</td>
</tr>
<tr>
<td align="left">9</td>
<td align="center">13</td>
<td align="center">1</td>
<td align="center">378</td>
<td align="center">257</td>
<td align="center">63</td>
<td align="center">478</td>
<td align="center">89</td>
<td align="center">168</td>
<td align="center">100</td>
</tr>
<tr>
<td align="left">10</td>
<td align="center">13</td>
<td align="center">1</td>
<td align="center">1,327</td>
<td align="center">1,267</td>
<td align="center">65</td>
<td align="center">1,426</td>
<td align="center">1,157</td>
<td align="center">110</td>
<td align="center">99</td>
</tr>
<tr>
<td align="left">10</td>
<td align="center">13</td>
<td align="center">2</td>
<td align="center">1,419</td>
<td align="center">1,088</td>
<td align="center">276</td>
<td align="center">1,934</td>
<td align="center">457</td>
<td align="center">631</td>
<td align="center">515</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-6">
<title>Polyploidy Recognition Algorithm</title>
<p>In order to determine the polyploid types and related chromosomes of the species, we develop the polyploidy recognition algorithm. The algorithm read the generated combine.csv file, look for the labeled box with the largest &#x2206;y, mark it, and then look up and down to find the labeled box with the length less than it. The result.csv file will be built by adopting exhaustive above process. Among them, comro represents the combination round, sumy represents the sum of &#x2206;y in same combination round. In order to determine the specific polyploidy of the species, the gene collinearity evaluation index line chart is drawn. The horizontal is the combined round, and the vertical is the corresponding gene collinearity evaluation index. The significant inflection point in the line chart represents the corresponding polyploid type. The gene collinearity evaluation index (MI) was calculated by dividing the cumulative collinearity fragments length to the corresponding chromosome length of the reference species in the len file to describe the performance of the polyploidy in the corresponding combination round, and the larger its value, the better the performance.<disp-formula id="equ1">
<mml:math id="m1">
<mml:mrow>
<mml:mi mathvariant="bold-italic">M</mml:mi>
<mml:mi mathvariant="bold-italic">I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="bold-italic">1</mml:mi>
</mml:mrow>
<mml:mi mathvariant="bold-italic">n</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:msub>
<mml:mi mathvariant="bold-italic">y</mml:mi>
<mml:mi mathvariant="bold-italic">m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">l</mml:mi>
<mml:mi mathvariant="bold-italic">e</mml:mi>
<mml:msub>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>Where <inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:mi>&#x394;</mml:mi>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <italic>n</italic> are the length and number of collinearity fragments in the same combination round, respectively; <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the corresponding chromosome length of the reference species, <italic>i</italic> is the corresponding chromosome number.</p>
<p>After determining the combination round, output the combined result graph. To describe the analysis of input, output and fetching the final results of the analysis, we made pseudo code. The pseudo code of the chromosome collinearity fragment labeling and polyploidy recognition algorithm is shown as below for a better understanding of the context and better assess the relevance of this paper.</p>
<p>
<statement content-type="algorithm" id="algorithm_1">
<label>Algorithm 1</label>
<p>Chromosome collinearity fragment labeling and polyploidy recognition algorithm.</p>
<p>
<inline-graphic xlink:href="fgene-13-842387-fx1.tif"/>
</p>
</statement>
</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec id="s3-1">
<title>
<italic>Salix sinopurpurea</italic> Polyploidy Recognition</title>
<p>Using PolyReco to objectively determine the polyploid types of <italic>Salix sinopurpurea</italic>. The model use <italic>Vitis vinifera</italic> as the reference genome to identify the target species <italic>Salix sinopurpurea</italic> polyploid type, and read the data of <italic>Vitis vinifera</italic> chromosome 4, 13 and 14. The DBSCAN algorithm obtains the collinearity fragments, and get the combine.csv file. Then using the polyploidy recognition algorithm to combine the labeled boxes exhaustively, and get the gene collinearity evaluation index table of each chromosome in different combination rounds (<xref ref-type="sec" rid="s10">Supplementary Table S1</xref>). In the colinearity evaluation index line chart (<xref ref-type="fig" rid="F2">Figure 2A</xref>), we can find that chromosomes 4, 13, and 14 of <italic>Vitis vinifera</italic> have obvious inflection points when the combined round is 2. Therefore, it is determined that the <italic>Salix sinopurpurea</italic> has a whole genome duplication event recently. This conclusion can be found in <xref ref-type="bibr" rid="B14">Wang et al., 2011</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>
<bold>(A)</bold> Collinearity evaluation index line chart of Vitis vinifera Chr.4, Chr.13, and Chr.14 <bold>(B)</bold> Combination figure of Salix sinopurpurea polyploidy. <bold>(A,B)</bold> correspond to each other. (1) The grape chromosome 4 (2) The grape chromosome 13 (3) The grape chromosome 14.</p>
</caption>
<graphic xlink:href="fgene-13-842387-g002.tif"/>
</fig>
<p>After determining the specific polyploidy, we can output the information of the labeled box participating in the polyploidy, and obtain the result.csv file of <italic>Vitis vinifera</italic> chromosomes 4, 13, and 14, in which the data of chromosome 13 is shown in <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>
<italic>Vitis vinifera</italic> Chr. 13 result.csv file.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">chr1</th>
<th align="center">chr2</th>
<th align="center">id</th>
<th align="center">l_x</th>
<th align="center">l_y</th>
<th align="center">num</th>
<th align="center">r_x</th>
<th align="center">r_y</th>
<th align="center">Sumy</th>
<th align="center">
<inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:mi>&#x394;</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">
<inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:mi>&#x394;</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">Comro</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">8</td>
<td align="center">13</td>
<td align="center">2</td>
<td align="center">78</td>
<td align="center">1,089</td>
<td align="center">239</td>
<td align="center">605</td>
<td align="center">446</td>
<td align="center">947</td>
<td align="center">643</td>
<td align="center">527</td>
<td align="center">1</td>
</tr>
<tr>
<td align="left">8</td>
<td align="center">13</td>
<td align="center">1</td>
<td align="center">592</td>
<td align="center">1,267</td>
<td align="center">56</td>
<td align="center">671</td>
<td align="center">1,155</td>
<td align="center">947</td>
<td align="center">112</td>
<td align="center">79</td>
<td align="center">1</td>
</tr>
<tr>
<td align="left">16</td>
<td align="center">13</td>
<td align="center">1</td>
<td align="center">1,163</td>
<td align="center">283</td>
<td align="center">42</td>
<td align="center">1,235</td>
<td align="center">91</td>
<td align="center">947</td>
<td align="center">192</td>
<td align="center">72</td>
<td align="center">1</td>
</tr>
<tr>
<td align="left">10</td>
<td align="center">13</td>
<td align="center">2</td>
<td align="center">1,419</td>
<td align="center">1,088</td>
<td align="center">276</td>
<td align="center">1,934</td>
<td align="center">457</td>
<td align="center">909</td>
<td align="center">631</td>
<td align="center">515</td>
<td align="center">2</td>
</tr>
<tr>
<td align="left">10</td>
<td align="center">13</td>
<td align="center">1</td>
<td align="center">1,327</td>
<td align="center">1,267</td>
<td align="center">65</td>
<td align="center">1,426</td>
<td align="center">1,157</td>
<td align="center">909</td>
<td align="center">110</td>
<td align="center">99</td>
<td align="center">2</td>
</tr>
<tr>
<td align="left">9</td>
<td align="center">13</td>
<td align="center">1</td>
<td align="center">378</td>
<td align="center">257</td>
<td align="center">63</td>
<td align="center">478</td>
<td align="center">89</td>
<td align="center">909</td>
<td align="center">168</td>
<td align="center">100</td>
<td align="center">2</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>According to the result.csv file, output the combined figure of the <italic>Salix sinopurpurea</italic> polyploidy (<xref ref-type="fig" rid="F2">Figure 2B</xref>). When the combination round is 2, get the two groups with the highest scores, among which the chromosomes 5 and 6 of the <italic>Salix sinopurpurea</italic> can be combined into a relatively complete chromosome 4 of <italic>Vitis vinifera</italic>, the corresponding MI is 67.87%; the chromosomes 2 and 18 of the <italic>Salix sinopurpurea</italic> can be combined into a relatively complete chromosome 4 of <italic>Vitis vinifera</italic>, and the corresponding MI is 42.19%. The chromosomes 8 and 16 of the <italic>Salix sinopurpurea</italic> can be combined into a relatively complete <italic>Vitis vinifera</italic> chromosome 13 with MI of 73.93%; the chromosomes 9 and 10 of the <italic>Salix sinopurpurea</italic> can be combined into a relatively complete <italic>Vitis vinifera</italic> chromosome 4, MI is 70.96%. The chromosomes 5, 15, and 17 of <italic>Salix sinopurpurea</italic> can be combined into a relatively complete <italic>Vitis vinifera</italic> chromosome 14 with MI of 88.74%; the chromosomes 10, 13, 16, and 17 of <italic>Salix sinopurpurea</italic> can be combined into a relatively complete <italic>Vitis vinifera</italic> chromosome 14, MI is 79.88%.</p>
</sec>
<sec id="s3-2">
<title>
<italic>Brassica rapa</italic> Polyploidy Recognition</title>
<p>In order to further verify the universality of the method, according to the procedure in 3.1, the model uses <italic>Arabidopsis thaliana</italic> as the reference genome to identify the polyploidy type of the target species <italic>Brassica rapa</italic>. Through reading the data of chromosomes 1, 3 and 5 of <italic>Arabidopsis thaliana</italic>, we finally get the gene collinearity evaluation index table of each chromosome in different combination rounds (<xref ref-type="sec" rid="s10">Supplementary Table S2</xref>). In the collinearity evaluation index line chart (<xref ref-type="fig" rid="F3">Figure 3A</xref>), it can be found that chromosomes 1, 3, and 5 of <italic>Arabidopsis thaliana</italic> had an obvious turning point when the combination round was 3. Therefore, it is determined that the <italic>Brassica rapa</italic>. had a whole genome triplication event recently. This conclusion can be found in Wang (2011a).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>
<bold>(A)</bold> Collinearity evaluation index line chart of <italic>Arabidopsis thaliana</italic> Chr.1, Chr.3 and Chr.5 <bold>(B)</bold> Combination figure of Brassica rapa polyploidy. <bold>(A,B)</bold> correspond to each other. (1) The <italic>Arabidopsis thaliana</italic> chromosome 1 (2) The <italic>Arabidopsis thaliana</italic> chromosome 3 (3) The <italic>Arabidopsis thaliana</italic> chromosome 5</p>
</caption>
<graphic xlink:href="fgene-13-842387-g003.tif"/>
</fig>
<p>After determining the specific polyploidy, the result.csv file of <italic>Arabidopsis thaliana</italic> chromosomes 1, 3, and 5 is obtained, in which the data of chromosome 3 is shown in <xref ref-type="table" rid="T4">Table 4</xref>.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>
<italic>Arabidopsis thaliana</italic> Chr. 3 result.csv file.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">chr1</th>
<th align="center">chr2</th>
<th align="center">id</th>
<th align="center">l_x</th>
<th align="center">l_y</th>
<th align="center">num</th>
<th align="center">r_x</th>
<th align="center">r_y</th>
<th align="center">Sumy</th>
<th align="center">
<inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:mi>&#x394;</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">
<inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:mi>&#x394;</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">Comro</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">3</td>
<td align="center">3</td>
<td align="center">1</td>
<td align="center">2,919</td>
<td align="center">5,436</td>
<td align="center">857</td>
<td align="center">4,001</td>
<td align="center">2,763</td>
<td align="center">4,793</td>
<td align="center">2,673</td>
<td align="center">1,082</td>
<td align="center">1</td>
</tr>
<tr>
<td align="left">9</td>
<td align="center">3</td>
<td align="center">1</td>
<td align="center">3,322</td>
<td align="center">1,458</td>
<td align="center">827</td>
<td align="center">4,386</td>
<td align="center">8</td>
<td align="center">4,793</td>
<td align="center">1,450</td>
<td align="center">1,064</td>
<td align="center">1</td>
</tr>
<tr>
<td align="left">6</td>
<td align="center">3</td>
<td align="center">1</td>
<td align="center">1,571</td>
<td align="center">2,132</td>
<td align="center">260</td>
<td align="center">2,021</td>
<td align="center">1,462</td>
<td align="center">4,793</td>
<td align="center">670</td>
<td align="center">450</td>
<td align="center">1</td>
</tr>
<tr>
<td align="left">1</td>
<td align="center">3</td>
<td align="center">1</td>
<td align="center">2,581</td>
<td align="center">5,431</td>
<td align="center">1,036</td>
<td align="center">3,948</td>
<td align="center">2,863</td>
<td align="center">3,799</td>
<td align="center">2,568</td>
<td align="center">1,367</td>
<td align="center">2</td>
</tr>
<tr>
<td align="left">4</td>
<td align="center">3</td>
<td align="center">1</td>
<td align="center">3</td>
<td align="center">1,231</td>
<td align="center">501</td>
<td align="center">631</td>
<td align="center">0</td>
<td align="center">3,799</td>
<td align="center">1,231</td>
<td align="center">628</td>
<td align="center">2</td>
</tr>
<tr>
<td align="left">5</td>
<td align="center">3</td>
<td align="center">1</td>
<td align="center">1,955</td>
<td align="center">5,432</td>
<td align="center">1,374</td>
<td align="center">3,668</td>
<td align="center">3,021</td>
<td align="center">3,533</td>
<td align="center">2,411</td>
<td align="center">1,713</td>
<td align="center">3</td>
</tr>
<tr>
<td align="left">7</td>
<td align="center">3</td>
<td align="center">1</td>
<td align="center">1,555</td>
<td align="center">1,130</td>
<td align="center">373</td>
<td align="center">1,998</td>
<td align="center">8</td>
<td align="center">3,533</td>
<td align="center">1,122</td>
<td align="center">443</td>
<td align="center">3</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>According to the result.csv file, output the combined figure of <italic>Brassica rapa</italic> polyploidy (<xref ref-type="fig" rid="F3">Figure 3B</xref>). When the combination round is 3, get the three groups with the highest scores, among which the chromosomes 7, 8, and 9 of the <italic>Brassica rapa</italic> can be combined into a relatively complete chromosome 1 of <italic>Arabidopsis thaliana</italic>, the corresponding MI is 89.88%; the chromosomes 2, 6, and 9 of the <italic>Brassica rapa</italic> can be combined into a relatively complete chromosome 1 of <italic>Arabidopsis thaliana</italic>, and the corresponding MI is 58.31%; the chromosomes 6, 7, and 10 of the <italic>Brassica rapa</italic> can be combined into a relatively complete chromosome 1 of <italic>Arabidopsis thaliana</italic>, and the corresponding MI is 49.38%. The chromosomes 3, 6, and 9 of the <italic>Brassica rapa</italic> can be combined into a relatively complete <italic>Arabidopsis thaliana</italic> chromosome 3 with MI of 88.16%; the chromosomes 1 and 4 of the <italic>Brassica rapa</italic> can be combined into a relatively complete <italic>Arabidopsis thaliana</italic> chromosome 3 with MI of 69.87%; the chromosomes 5 and 7 of the <italic>Brassica rapa</italic> can be combined into a relatively complete <italic>Arabidopsis thaliana</italic> chromosome 3 with MI of 64.98%. The chromosomes 2, 6, and 9 of the <italic>Brassica rapa</italic> can be combined into a relatively complete <italic>Arabidopsis thaliana</italic> chromosome 5, MI is 83.54%; the chromosomes 6 and 10 of the <italic>Brassica rapa</italic> can be combined into a relatively complete <italic>Arabidopsis thaliana</italic> chromosome 5, MI is 59.84%; the chromosomes 3 of the <italic>Brassica rapa</italic> can be combined into a relatively complete <italic>Arabidopsis thaliana</italic> chromosome 5, MI is 51.17%.</p>
<p>In <xref ref-type="fig" rid="F3">Figure 3B</xref>, there are two seemingly identical collinearity fragments among the four fragments formed by <italic>Arabidopsis thaliana</italic> chromosome 3 and <italic>Brassica rapa</italic> chromosome 4, with the naked eye. But only one is labeled and used, because it has a small number of homologous genes. So this method can break through the limitations of the human eye, and find chromosome fragments with strong collinearity, as well as provide a basis for objective judgment of polyploidy.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>The previous study has mostly used to observe the atlas with prior knowledge to identify the polyploid types of the species. This method has some major limitations, such as low efficiency, high dependence on prior knowledge, strong subjectivity, lack of objective evaluation criteria, and easy introduction of human error. In this paper, digital image processing technology was used to identify polyploid types based on clustering algorithms. The <italic>K</italic>
<sub>
<italic>S</italic>
</sub> dotplot of homologous genes was used as the research object, and the DBSCAN method was used to cluster. Then we can obtain the collinear fragments and automatically label collinear region. According to the gene collinearity evaluation index line chart of each combination, the model can determine the polyploid type and related chromosome combination. The study mainly focused on developing a polyploidization recognition algorithm and providing the method to speed up the evolutionary laws of gene structure associated with polyploidy research. PolyReco involves more than a simple labels of collinear regions, but also gives the polyploidy types through the collinearity evaluation index line chart and related chromosomes at the end. Compared with MCScanX (<xref ref-type="bibr" rid="B16">Wang et al., 2012</xref>), PolyReco labels the specific gene segments involved in the polyploidy events and improves the recognition efficiency of polyploidy. Compared to traditional methods, PolyReco reduces the dependence on prior knowledge, solves the limitations of the human eye in visual space, comply with artificial logic analysis and reasoning process. Moreover, the PolyReco can not only provides an effective method for large-scale rapid identification of genome polyploidy but also has important application value in distant hybrid breeding (<xref ref-type="bibr" rid="B9">Rabanus-Wallace et al., 2021</xref>).</p>
<p>In summary, the proposed PolyReco provides a reference model for processing automatically label collinear regions and recognize polyploidy. However, the <italic>K</italic>
<sub>
<italic>S</italic>
</sub> dotplot is sensitive to the size of the parameter Eps. When a large value is used for Eps, the fragmented collinearity segments are easy to cluster together. On the contrary, it is easy to separate continuous fragments so that complete collinearity fragments cannot be clustered. In the next step, we expect to study the DBSCAN clustering method based on adaptive Eps to further optimize the clustering effect.</p>
</sec>
</body>
<back>
<sec id="s5">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s10">Supplementary Material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s6">
<title>Author Contributions</title>
<p>FW and KZ conceived and designed the experiments, performed the experiments, analyzed the data, prepared figures and/or tables, authored or reviewed drafts of the paper. RZ performed the experiments, analyzed the data, prepared figures and/or tables, and authored drafts of the paper. HL designed the experiments and analyzed the data. WZ and ZJ analyzed the data, prepared figures and/or tables. CW conceived and designed the experiments, authored or reviewed drafts of the paper. Manuscript is approved by all authors for publication.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>This work was supported by Science and Technology Research Project of Hebei Province Colleges and Universities (No. QN2020421); the Scientific Research Project of Introducing Talents of Hebei Agricultural University (No. YJ201944); the Innovative Research Group Project of Hebei Natural Science Foundation (Grant No. C2020204111); the International Science and Technology Cooperation base Special Project of Hebei (Grant No. 20592901D); National Natural Science Foundation of China (31901864).</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2022.842387/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2022.842387/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.DOCX" id="SM1" mimetype="application/DOCX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table2.DOCX" id="SM2" mimetype="application/DOCX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xian</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Marin</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Keller</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Genomes of Subaerial Zygnematophyceae Provide Insights into Land Plant Evolution</article-title>. <source>Cell</source> <volume>179</volume> (<issue>5</issue>), <fpage>1057</fpage>&#x2013;<lpage>1067</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2019.10.019</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dong</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>The Genome of Magnolia Biondii Pamp. Provides Insights into the Evolution of Magnoliales and Biosynthesis of Terpenoids</article-title>. <source>Hortic. Res.</source> <volume>8</volume> (<issue>1</issue>), <fpage>38</fpage>. <pub-id pub-id-type="doi">10.1038/s41438-021-00471-9</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Edgar</surname>
<given-names>R. C.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>MUSCLE: a Multiple Sequence Alignment Method with Reduced Time and Space Complexity</article-title>. <source>BMC Bioinformatics</source> <volume>5</volume>, <fpage>113</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-5-113</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>The Abundance of Homoeologue Transcripts Is Disrupted by Hybridization and Is Partially Restored by Genome Doubling in Synthetic Hexaploid Wheat</article-title>. <source>Bmc Genomics</source> <volume>18</volume>, <fpage>149</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-017-3558-0</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Defoort</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tasdighian</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Maere</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Van de Peer</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>De Smet</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Gene Duplicability of Core Genes Is Highly Consistent across All Angiosperms</article-title>. <source>Plant Cell</source> <volume>28</volume> (<issue>2</issue>), <fpage>326</fpage>&#x2013;<lpage>344</lpage>. <pub-id pub-id-type="doi">10.1105/tpc.15.00877</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marcet-Houben</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gabald&#xf3;n</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Beyond the Whole-Genome Duplication: Phylogenetic Evidence for an Ancient Interspecies Hybridization in the Baker&#x27;s Yeast Lineage</article-title>. <source>Plos Biol.</source> <volume>13</volume> (<issue>8</issue>), <fpage>e1002220</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pbio.1002220</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Miyata</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yasunaga</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Nishida</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>1980</year>). <article-title>Nucleotide Sequence Divergence and Functional Constraint in mRNA Evolution</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A.</source> <volume>77</volume> (<issue>12</issue>), <fpage>7328</fpage>&#x2013;<lpage>7332</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.77.12.7328</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nei</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gojobori</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>1986</year>). <article-title>Simple Methods for Estimating the Numbers of Synonymous and Nonsynonymous Nucleotide Substitutions</article-title>. <source>Mol. Biol. Evol.</source> <volume>3</volume> (<issue>5</issue>), <fpage>418</fpage>&#x2013;<lpage>426</lpage>. <pub-id pub-id-type="doi">10.1093/oxfordjournals.molbev.a040410</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rabanus-Wallace</surname>
<given-names>M. T.</given-names>
</name>
<name>
<surname>Hackauf</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Mascher</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lux</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wicker</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Gundlach</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Chromosome-scale Genome Assembly Provides Insights into rye Biology, Evolution and Agronomic Potential</article-title>. <source>Nat. Genet.</source> <volume>53</volume> (<issue>4</issue>), <fpage>564</fpage>&#x2013;<lpage>573</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-021-00807-0</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Jiao</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Shan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>WGDI: A User-Friendly Toolkit for Evolutionary Analyses of Whole-Genome Duplications and Ancestral Karyotypes</article-title>. <source>bioRxiv</source>. <pub-id pub-id-type="doi">10.1101/2021.04.29.441969</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>An Overlooked Paleotetraploidization in Cucurbitaceae</article-title>. <source>Mol. Biol. Evol.</source> <volume>35</volume> (<issue>1</issue>), <fpage>16</fpage>&#x2013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msx242</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Hierarchically Aligning 10 Legume Genomes Establishes a Family-Level Genomics Platform</article-title>. <source>Plant Physiol.</source> <volume>174</volume> (<issue>1</issue>), <fpage>284</fpage>&#x2013;<lpage>300</lpage>. <pub-id pub-id-type="doi">10.1104/pp.16.01981</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Recursive Paleohexaploidization Shaped the Durian Genome</article-title>. <source>Plant Physiol.</source> <volume>179</volume> (<issue>1</issue>), <fpage>209</fpage>&#x2013;<lpage>219</lpage>. <pub-id pub-id-type="doi">10.1104/pp.18.00921</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>The Genome of the Mesopolyploid Crop Species Brassica Rapa</article-title>. <source>Nat. Genet.</source> <volume>43</volume> (<issue>10</issue>), <fpage>1035</fpage>&#x2013;<lpage>1039</lpage>. <pub-id pub-id-type="doi">10.1038/ng.919</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Bowers</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Feltus</surname>
<given-names>F. A.</given-names>
</name>
<name>
<surname>Paterson</surname>
<given-names>A. H.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Extensive Concerted Evolution of rice Paralogs and the Road to Regaining independence</article-title>. <source>Genetics</source> <volume>177</volume> (<issue>3</issue>), <fpage>1753</fpage>&#x2013;<lpage>1763</lpage>. <pub-id pub-id-type="doi">10.1534/genetics.107.073197</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Debarry</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>MCScanX: a Toolkit for Detection and Evolutionary Analysis of Gene Synteny and Collinearity</article-title>. <source>Nucleic Acids Res.</source> <volume>40</volume> (<issue>7</issue>), <fpage>e49</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkr1293</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wong</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Suchard</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Huelsenbeck</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Alignment Uncertainty and Genomic Analysis</article-title>. <source>Science</source> <volume>319</volume> (<issue>5862</issue>), <fpage>473</fpage>&#x2013;<lpage>476</lpage>. <pub-id pub-id-type="doi">10.1126/science.1151532</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Comparative Genome Analysis of Scutellaria Baicalensis and Scutellaria Barbata Reveals the Evolution of Active Flavonoid Biosynthesis</article-title>. <source>Genomics, Proteomics &#x26; Bioinformatics</source> <volume>18</volume> (<issue>3</issue>), <fpage>230</fpage>&#x2013;<lpage>240</lpage>. <pub-id pub-id-type="doi">10.1016/j.gpb.2020.06.002</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Xi</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>She</surname>
<given-names>R. X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>P. P.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>Y. J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Improved De Novo Chromosome&#x2010;level Genome Assembly of the Vulnerable walnut Tree Juglans Mandshurica Reveals Gene Family Evolution and Possible Genome Basis of Resistance to Lesion Nematode</article-title>. <source>Mol. Ecol. Resour.</source> <volume>21</volume> (<issue>6</issue>), <fpage>2063</fpage>&#x2013;<lpage>2076</lpage>. <pub-id pub-id-type="doi">10.1111/1755-0998.13394</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Nielsen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Goldman</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Pedersen</surname>
<given-names>A.-M. K.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Codon-substitution Models for Heterogeneous Selection Pressure at Amino Acid Sites</article-title>. <source>Genetics</source> <volume>155</volume> (<issue>1</issue>), <fpage>431</fpage>&#x2013;<lpage>449</lpage>. <pub-id pub-id-type="doi">10.1093/genetics/155.1.431</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>F. S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z. K.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>Z. X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C. Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Music Emotion Recognition Method Based on Multi Feature Fusion</article-title>. <source>Ijart</source> <volume>13</volume>, <fpage>1</fpage>&#x2013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1504/ijart.2021.10043883</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="thesis">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>M. H.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Comparative Genomics and Bioiformatics Research into Salicaceae Genomes</article-title>,&#x201d; (<publisher-loc>Tangshan, China</publisher-loc>: <publisher-name>North China University of Science and Technology</publisher-name>). <comment>Dissertation</comment>. </citation>
</ref>
</ref-list>
</back>
</article>