<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">757524</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.757524</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Feature Compression Applications of Genetic Algorithm</article-title>
<alt-title alt-title-type="left-running-head">Zou et&#x20;al.</alt-title>
<alt-title alt-title-type="right-running-head">Application of Genetic Algorithm</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zou</surname>
<given-names>Meiling</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jiang</surname>
<given-names>Sirong</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1409512/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Fang</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1484641/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhao</surname>
<given-names>Long</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1347540/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Chenji</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bao</surname>
<given-names>Yuting</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Yonghao</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1410254/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Xia</surname>
<given-names>Zhiqiang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1484611/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Institute of Tropical Biosciences and Biotechnology</institution>, <institution>Chinese Academy of Tropical Agriculture Sciences</institution>, <addr-line>Haikou</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Hainan University</institution>, <addr-line>Haikou</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Academy of Agriculture and Forestry Sciences</institution>, <institution>Qinghai University</institution>, <addr-line>Xining</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1519404/overview">Shilin Zhao</ext-link>, Vanderbilt University Medical Center, United&#x20;States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/516450/overview">Yang Xu</ext-link>, Yangzhou University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/695184/overview">Meiyue Wang</ext-link>, Stanford University, United&#x20;States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1622874/overview">Yu Wang</ext-link>, Vanderbilt University Medical Center, United&#x20;States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Zhiqiang Xia, <email>zqxia@hainanu.edu.cn</email>
</corresp>
<fn fn-type="equal" id="fn1">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this&#x20;work</p>
</fn>
<fn fn-type="other">
<p>This article was submitted to Statistical Genetics and Methodology, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>08</day>
<month>03</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>757524</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>08</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>02</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Zou, Jiang, Wang, Zhao, Zhang, Bao, Chen and Xia.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Zou, Jiang, Wang, Zhao, Zhang, Bao, Chen and Xia</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these&#x20;terms.</p>
</license>
</permissions>
<abstract>
<p>With the rapid development of molecular breeding technology and many new varieties breeding, a method is urgently needed to identify different varieties accurately and quickly. Using this method can not only help farmers feel convenient and efficient in the normal cultivation and breeding process but also protect the interests of breeders, producers and users. In this study, single nucleotide polymorphism (SNP) data of 533&#x20;<italic>Oryza sativa</italic>, 284&#x20;<italic>Solanum tuberosum</italic> and 247&#x20;<italic>Sus scrofa</italic> and 544&#x20;<italic>Manihot esculenta</italic> Crantz were used. The original SNPs were filtered and screened to remove the SNPs with deletion number more than 1% or the homozygous genotype 0/0 and 1/1 number less than 2. The correlation between SNPs were calculated, and the two adjacent SNPs with correlation R<sup>2</sup> &#x3e; 0.95 were retained. The genetic algorithm program was developed to convert the genotype format and randomly combine SNPs to calculate a set of a small number of SNPs which could distinguish all varieties in different species as fingerprint data, using Matlab platform. The successful construction of three sets of fingerprints showed that the method developed in this study was effective in animals and plants. The population structure analysis showed that the genetic algorithm could effectively obtain the core SNPs for constructing fingerprints, and the fingerprint was practical and effective. At present, the two-dimensional code of <italic>Manihot esculenta</italic> Crantz fingerprint obtained by this method has been applied to field planting. This study provides a novel idea for the <italic>Oryza sativa</italic>, <italic>Solanum tuberosum</italic>, <italic>Sus scrofa</italic> and <italic>Manihot esculenta</italic> Crantz identification of various species, lays foundation for the cultivation and identification of new varieties, and provides theoretical significance for many other species fingerprints construction.</p>
</abstract>
<kwd-group>
<kwd>fingerprint</kwd>
<kwd>DNA molecular markers</kwd>
<kwd>SNP</kwd>
<kwd>genetic algorithm</kwd>
<kwd>feature compression</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>A DNA molecular marker is a kind of genetic marker based on DNA molecular polymorphism, which directly reflects genetic variation at the DNA level (<xref ref-type="bibr" rid="B22">Muhammad et&#x20;al., 2018</xref>). With the rapid development of molecular biology technology, research on DNA molecular markers is becoming increasingly popular. At present, dozens of different molecular markers have been studied and used, such as restriction fragment length polymorphism (RFLP), random amplified polymorphic DNA (RAPD), amplified fragment length polymorphism (AFLP), and simple sequence repeats (SSR). Single nucleotide polymorphism (SNP), known as a third-generation DNA molecular marker technology, refers to the differences in single nucleotide between different alleles at the same locus. The most common one is the substitution of a single nucleotide, and the deletion or insertion of a single base usually occurs between purine base (A/G) and pyrimidine base (C/T) (<xref ref-type="bibr" rid="B12">Jooyeong et&#x20;al., 2020</xref>). SNP markers play an important role in distinguishing the differences between two genetic materials, so they are also considered the most promising genetic markers. SNP is highly different from the first-generation RFLP and the second-generation SSR markers; this difference is mainly manifested in two aspects. Firstly, the detection tool of SNP is no longer the length change of DNA fragments but directly uses the sequence change as a marker. Secondly, SNP marker analysis completely abandons traditional gel electrophoresis and replaces it with the latest DNA chip technology.</p>
<p>Reduced-representation genome sequencing (RRGS) refers to the use of restriction endonucleases to interrupt genomic DNA to carry out the high-throughput sequencing of specific fragments and obtain a large number of genetic polymorphism marker sequences, thereby fully representing the whole genome information of the target species sequencing strategy. The method is simple in experimental steps and low in cost, and it can obtain genetic polymorphism markers of the whole genome without referring to the genome. It has been widely used in ecology, evolution and genomics. Reduced genome can obtain representative genetic diversity SNPs in the whole genome, so it is widely used to construct population evolution, population structure, population history, genetic mapping and linkage map. At present, the commonly used RRGS technologies includin RAD (<xref ref-type="bibr" rid="B2">Baird et&#x20;al., 2008</xref>)<sub>,</sub> GBS (<xref ref-type="bibr" rid="B27">Robert et&#x20;al., 2011</xref>)<sub>,</sub> 2b-RAD (<xref ref-type="bibr" rid="B24">Peterson et&#x20;al., 2012</xref>)<sub>,</sub> dd-RAD (<xref ref-type="bibr" rid="B30">Shi et&#x20;al., 2012</xref>; <xref ref-type="bibr" rid="B23">Palaiokostas et&#x20;al., 2016</xref>) and AFSM (<xref ref-type="bibr" rid="B33">Xia et&#x20;al., 2014</xref>).</p>
<p>Genetic algorithm (GA) is a search algorithm that simulates natural evolution to explore the optimal solution (<xref ref-type="bibr" rid="B16">Kumar and Ghose, 2009</xref>; <xref ref-type="bibr" rid="B32">Wu and Rul&#x2019;kov, 1993</xref>), which belongs to the evolutionary algorithm. It is a computational model of biological evolution based on natural selection and genetic mechanism of Darwin&#x2019;s theory of biological evolution (<xref ref-type="bibr" rid="B8">Goldberg, 1989</xref>). The intersection and penetration between life science and engineering science is the main cause for the emergence of GA, and the highly parallel global random search algorithm used for solving problems is its essence. In the calculation process of GA, the GA cannot directly deal with the parameters in practical problems, and it can only deal with chromosomes expressed in the form of gene strings. Therefore, to use GA, the parameters that optimise the solution of the problem must be transformed into recognisable chromosome form via coding. Aiming at the problem to be solved, an initial population is formed by coding, and the fitness of each individual in the population is calculated. After screening the fitness function, the individuals with high fitness will be operated in the next step. In the operation process, there are three genetic operators, namely, selection operator, crossover operator and mutation operator. Among them, on the basis of the principle that the higher the fitness of the selected individual, the easier it is to be selected and the easier it is to produce the optimal solution, some individuals are selected and hybridised to produce new offspring. Some individuals are selected to mutate to form new offspring and finally form a new population. Re-calculations are conducted according to the process until the fitness of the best individual reaches the set threshold, or when the fitness of the best individual and group reaches the peak, or the number of iterations reaches the set value. The maximum number of iterations is typically adopted as the termination condition. The calculation process of GA is to perform certain operations on individuals depending on the group&#x2019;s environmental adaptability (adaptability evaluation), so as to realise the evolutionary process of survival of the fittest. From the perspective of optimal search, genetic operation can optimise the solution of the problem by generation and approach the optimal solution.</p>
<p>DNA fingerprinting is a technology based on molecular markers, which has been used to describe the molecular patterns of genotypes. Co-dominant marker (SSR, SNP and SCAR) is the most recommended marker for constructing DNA fingerprinting (<xref ref-type="bibr" rid="B1">Azevedo et&#x20;al., 2018</xref>). This technology is now considered a powerful tool for variety identification, strains and cloning, as well as genetic correlation between parents and offspring (<xref ref-type="bibr" rid="B15">Kopp et&#x20;al., 2002</xref>). The cost of developing DNA molecular markers is becoming low, the speed is becoming fast and the methods are becoming highly convenient and diverse. The DNA fingerprint is gradually improved. Some species have developed from using the first generation of molecular markers RFLP and AFLP to the second generation of molecular markers SSR and then to the third generation of molecular markers SNP, which is beginning to be popular at present. All have constructed fingerprints for a variety of identification or classification processes. With the development of fingerprints, the number of identifiable samples, accuracy and experimental stability have been improved (<xref ref-type="bibr" rid="B4">Chen et&#x20;al., 2000</xref>; <xref ref-type="bibr" rid="B34">Zhan et&#x20;al., 2012</xref>). C.E. McGregor et&#x20;al. used RAPD (20 primers), ISSR (6 primers), AFLP (2 primers) and SSR (5 primer pairs) to study DNA fingerprints based on PCR in 39&#x20;<italic>Solanum tuberosum</italic> varieties (<xref ref-type="bibr" rid="B21">McGregor et&#x20;al., 2000</xref>). A., Zhuk and others used eight SSR markers with the strongest polymorphism to analyse all <italic>Solanum tuberosum</italic> varieties (<italic>Solanum tuberosum</italic> L. subsp. tuberosum) listed in the Latvian plant genetic resources database. The variety fingerprint, genetic distance evaluation and cluster analysis were carried out (<xref ref-type="bibr" rid="B37">Zhuk, 2008</xref>). Zhao S J et&#x20;al. constructed a DNA fingerprint database by SSR markers and analysed the genetic diversity of the main varieties of 27 seedless watermelons in China (<xref ref-type="bibr" rid="B36">Zhao et&#x20;al., 2013</xref>). Li L et&#x20;al. collected 41 unique wheat varieties in Shandong Province by SSR technology and constructed a DNA fingerprint database (<xref ref-type="bibr" rid="B18">Li, 2013</xref>).</p>
<p>How to simply identify new materials and varieties is not only the need of research classification but also an important basis for the protection and utilisation of knowledge output. Therefore, a method that can accurately identify various varieties in different species is urgently needed. Population structure analysis plays a major auxiliary role in variety identification. Cluster analysis is used to observe the genetic relationship of varieties in the population for the further analysis of the origin and variety situation of the varieties. In this study, the SNP data of Gramineae <italic>Oryza sativa</italic>, polyploid <italic>Solanum tuberosum</italic> and mammalian <italic>Sus scrofa</italic> were used. After a series of algorithm screening and population structure analysis, 100 core SNPs were finally selected to construct fingerprints that can distinguish different varieties. These SNPs are expected to be applied to variety specificity and authenticity identification and seedling purity identification, which will provide new technical basis for the further excavation and utilisation of genetic resources of various species and registration and protection of varieties.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and Methods</title>
<sec id="s2-1">
<title>Sample Collection, Sequencing Libraries Preparation and Sequencing</title>
<p>The fresh leaves of 284&#x20;<italic>Solanum tuberosum</italic> accessions from the Academy of Agriculture and Forestry Sciences of Qinghai University (E101, N36) were collected, and DNA of all samples was extracted by the modified CTAB method. Through 1% agarose gel electrophoresis detection and concentration measurement, the working solution diluted to 100&#xa0;ng/&#x3bc;L according to the mother liquor concentration was stored at &#x2212;20&#xb0;C. The EcoRI-MspI and EcoRI-HpaII libraries of 284&#x20;<italic>Solanum tuberosum</italic> DNA samples were constructed by the AFSM method, and two mixed gene pools of EcoRI-MspI and EcoRI-HpaII were purified by OMEGA&#x2019;s E.Z.N.A.Cycle Pure Kit. After PCR amplification and monoclonal detection met the requirements, the EcoRI-MspI and EcoRI-HpaII libraries were mixed into a library at a ratio of 1:1 for Hiseq 2,500 sequencing.</p>
</sec>
<sec id="s2-2">
<title>SNP Data Acquisition</title>
<p>For <italic>Solanum tuberosum</italic>, the original Illumina sequencing data were optimised. The artificial data errors were reduced by a Perl script (<ext-link ext-link-type="uri" xlink:href="http://afsmseq.sourceforge.net/">http://afsmseq.sourceforge.net/</ext-link>), and the total number of sequencing reads was obtained. All the reads were assigned to each individual through the designed tags, and the number of reads of each individual was counted. Optimised sequencing reads were compared with the <italic>Solanum tuberosum</italic> PGSC_DM_v4.03 reference genome by Bowtie2 software, and subsequent SNPs were identified by the two software programs of SAMtools (<ext-link ext-link-type="uri" xlink:href="http://samtools.sourceforge.net/">http://samtools.sourceforge.net/</ext-link>) and VCFtools (<ext-link ext-link-type="uri" xlink:href="http://vcftools.sourceforge.net/">http://vcftools.sourceforge.net/</ext-link>). SNP data of <italic>Oryza sativa</italic> and <italic>Sus scrofa</italic> were from the RiceVarMap v2.0 database (<ext-link ext-link-type="uri" xlink:href="http://ricevarmap.ncpgr.cn/">http://ricevarmap.ncpgr.cn/</ext-link>) (<xref ref-type="bibr" rid="B3">Chen et&#x20;al., 2014</xref>) and the BIGD database (<ext-link ext-link-type="uri" xlink:href="https://bigd.big.ac.cn/">https://bigd.big.ac.cn/</ext-link>), respectively. SNPs of <italic>Manihot esculenta</italic> Crantz was from our laboratory (<xref ref-type="bibr" rid="B35">Zhang et&#x20;al., 2018</xref>).</p>
</sec>
<sec id="s2-3">
<title>Core SNPs Selection</title>
<p>SNPs were processed in the obtained samples. Genotypes were converted by using online editing tools of awk and sed under the Linux system. The SNPs with the deletion rate over 99%, as well as the homozygous genotype 0/0 and 1/1 number less than 2, were removed. The remaining SNPs were obtained as candidate SNPs. For the candidate SNPs, VCFtools was used to calculate the R<sup>2</sup> value among SNPs, and the adjacent SNPs with R<sup>2</sup> &#x3e; 0.95 were reserved for subsequent calculation.</p>
<p>The read_SNP.m program independently developed in the Matlab platform was used to transform the format of the selected SNPs and pre-process the data, remove all the labelling information in the vcf file and transform the genotype format. The obtained SNPs were used for the calculation of the subsequent&#x20;GA.</p>
</sec>
<sec id="s2-4">
<title>Application of Genetic Algorithm</title>
<p>With the main.m program in Matlab in the same folder, data.mat obtained in the last step was imported to the program automatically. The number of SNPs and the maximum number of iterations in the parameter area were set, then the program was run after deployment (<xref ref-type="fig" rid="F1">Figure&#x20;1</xref>). In the running process, the first step was to select the appropriate initial population and the maximum number of iterations and calculate the fitness of the initial population. The formula was as follows:<list list-type="simple">
<list-item>
<p>1) For linear ordering:</p>
</list-item>
</list>
<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mn>1.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2.0</mml:mn>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
<list list-type="simple">
<list-item>
<p>2) For nonlinear ordering:</p>
</list-item>
</list>
<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Calculation principle of genetic algorithm.</p>
</caption>
<graphic xlink:href="fgene-13-757524-g001.tif"/>
</fig>
<p>Nind is the number of individuals in a population. sp is the selected differential pressure. pos is a position in an ordered population.</p>
<p>After several iterations, the population number increased. The population with a high fitness was selected for cross-mutation with a new population, then a new population was generated again. The process was repeated until the maximum number of iterations was reached. Then, the procedure was ended. All SNP combinations that can be used in distinguishing <italic>Solanum tuberosum</italic> L. individuals were displayed on the interface. The calculation result was saved in the result file. The detailed information of all the SNP combinations were included in the result file. The SNP combinations were ranked from high to low according to the number of individuals identified. Therefore, a group was selected from several SNP combinations according to the distribution of SNPs on chromosomes for fingerprint construction. The main information obtained from the result file included the ID number of all the SNPs and their positions corresponding to total SNP database. With this information, the core SNP genotype data was collected and arranged.</p>
<p>The specific operation process of genetic algorithm in this study is as follows:</p>
<p>Encoding parameters: the genotypes at SNP loci were transformed into the following four forms by encoding<disp-formula id="e3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>45</mml:mn>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>.</mml:mo>
<mml:mo>/</mml:mo>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>97</mml:mn>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>98</mml:mn>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>104</mml:mn>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>The raw data were obtained from <italic>N</italic> (total number of SNPs) loci of <italic>I</italic> (number of individuals) individuals, and a two-dimensional matrix <italic>S</italic> of <italic>N</italic>&#x2a;<italic>I</italic> (SNP array of all samples) was obtained.</p>
<p>Setting of initial population: We set the population size to be <italic>P</italic>, and each individual randomly selects <italic>C</italic> SNP loci from the <italic>N</italic> SNP loci to form an initial population <italic>G</italic>, which is a two-dimensional matrix of <italic>P</italic>&#x2a;<italic>C</italic>.</p>
<p>Setting of adaptive function: For the individual of each population, we currently have <italic>C</italic> SNP sites randomly selected by it. Assuming that the individual may be any one of the <italic>I</italic> individual samples, data of the corresponding <italic>C</italic> SNP sites in the <italic>I</italic> samples are collected respectively.<disp-formula id="e4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>C</mml:mi>
</mml:munderover>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="equ1">
<mml:math id="m5">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ2">
<mml:math id="m6">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Thus, we can obtain the genotype number coding data e of the randomly selected SNP loci corresponding to each individual in the population.</p>
<p>Next, we find out the unique ones in the <italic>C</italic> genotypes that each individual may have, namely, the valuable SNPs that can be used to distinguish species. The more unique SNPs that an individual may have, the more individuals that it can identify. We believe that the higher the fitness is, and the individual fitness array F of the population can be obtained.</p>
<p>Selection probability:We find the sum <italic>S</italic> of the fitness of <italic>P</italic> individuals. Then the fitness/total fitness of each individual is used to calculate the selection probability of a specific individual in the inheritance<disp-formula id="e5">
<mml:math id="m7">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>P</mml:mi>
</mml:munderover>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
<disp-formula id="e6">
<mml:math id="m8">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>S</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="equ3">
<mml:math id="m9">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Cumulative probability: In a roulette-like manner, from 0 to 1, the selection range of each individual is added with the selection probability of the individual from the sum of the selection ranges of all the individuals before, and finally a number in the range of 0&#x2013;1 is generated to see which range the number falls in, so as to determine which individual is selected finally. for example, the range of the random number <italic>F(j)</italic> required for the selection of the <italic>j</italic>-number individual is as follows<disp-formula id="e7">
<mml:math id="m10">
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munderover>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2264;</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:munderover>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
<p>Select action: Randomly generating a number of 0&#x2013;1, selecting individual serial numbers corresponding to two ranges which are closest to the number, and generating a random number <italic>R</italic> which is consistent with uniform distribution on the numbers of 0&#x2013;1<disp-formula id="e8">
<mml:math id="m11">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>0,1</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<p>Obtaining the serial numbers of the two mated individuals and then mating the&#x20;two</p>
<p>Interlace operation: It is determined whether or not to perform a crossover operation according to a given crossover probability, and if the crossover operation is performed, a crossover bit <italic>B1</italic> needs to be randomly generated in the range [1,<italic>C</italic>-1]<disp-formula id="e9">
<mml:math id="m12">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
<disp-formula id="equ4">
<mml:math id="m13">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="bold">N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Taking the crossing position as a dividing line, and carrying out cross exchange on two groups of genes participating in mating, wherein if the error that the SNP locus repeatedly appears in a new genotype after crossing occurs, the SNPs not appearing in the new genotype can be listed from all the <italic>I</italic> SNP loci and sequentially replaced, if the cross operation is not carried out, the offspring keep the original genotype unchanged.</p>
<p>Mutation operation: According to the variation probability, if the mutation operation is performed, 0.3&#x2a;<italic>C</italic> mutation bits need to be randomly generated in the [1,<italic>C</italic>] range and counted as <italic>B2</italic> array<disp-formula id="e10">
<mml:math id="m14">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
<disp-formula id="equ5">
<mml:math id="m15">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ6">
<mml:math id="m16">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="bold">N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>The mutation site are first emptied, Then listing all the SNP sites which do not currently appear in the variant genotype and sequencing in ascending order, and selecting the SNP sites which do not appear as the variant SNP to be inserted into the variant genotype by taking the value of the variant site as a subscript.</p>
<p>Analog result: As a result, a new population is generated, the previous generation and the new generation should be integrated now, after discarding the SNP with repeated genotypes, the SNP information table of all the existing genotypes is updated, and then the population is eliminated according to the descending order of the fitness of all the genotypic individuals in the current population to ensure that the population number is <italic>P</italic> constant, then the highest fitness, the average fitness and the best SNP individuals of each generation are recorded, and The <italic>X</italic> (maximum number of iterations) generation is cycled in such a way that the last generation is the best combination of SNP loci with the highest fitness, average fitness, and legacy from elimination.</p>
</sec>
<sec id="s2-5">
<title>Evaluate the Number of Core SNPs for Fingerprint Construction</title>
<p>In the calculation process of GA, the number of SNPs was set as 80, 100, 200 and 300, and multiple sets of fingerprints were constructed. The correlation between each matrix of the fingerprint and the original SNP data matrix was calculated in the form of a matrix. The utility of fingerprints composed of different SNPs was evaluated, and fingerprints composed by the best SNP combinations were selected for subsequent verification and analysis.</p>
</sec>
<sec id="s2-6">
<title>Conversion of Genotype Data into Binary Coded Data</title>
<p>Given that SNP molecular markers have dimorphism, the data in the fingerprint database can be converted into binary coded data. Homozygous genotype 0/0 was expressed as 1, homozygous genotype 1/1 was expressed as 2, heterozygous genotype 0/1 was expressed as 0 and deletion base locus was marked as &#x2018;-&#x2019;. The data of each sample were statistically sorted, so that the genotype data of each sample were integrated to form a unique code, that is, the fingerprint code to distinguish each sample.</p>
</sec>
<sec id="s2-7">
<title>Fingerprint Verification</title>
<p>ADMIXTURE software was used to evaluate the population structure, and each sample in the population was divided into corresponding subgroups. Firstly, the file format required by the software was obtained by using PLINK software (<xref ref-type="bibr" rid="B26">Purcell et&#x20;al., 2007</xref>). Secondly, the value range of subgroup number K was set to 1 to 12, and the appropriate subgroup number K value was selected based on the cross-validation error value obtained from the running result. The software of VCF2DiffMatrix was used to calculate the genetic distances of the core SNP corresponding to the fingerprint library and the original SNP locus after filtration. The distance matrix was obtained and imported into the Neighbour software to calculate the tree file, which was classified according to the K value. Different colours were used to represent different subgroups (<ext-link ext-link-type="uri" xlink:href="https://itol.embl.de/upload.cgi">https://itol.embl.de/upload.cgi</ext-link>). The results were visualised, and the clustering effect was observed.</p>
<p>The SNPs were visualised by using the CMplot package in R (<ext-link ext-link-type="uri" xlink:href="https://www.r-project.org/">https://www.r-project.org/</ext-link>), and the SNP distribution variations before and after GA calculation were observed. At the same time, VCFtools were used to calculate the allele frequency of vcf file containing SNP data. The calculated allele frequency was imported into PIC_CALC 0.6 (<ext-link ext-link-type="uri" xlink:href="https://github.com/luansheng/PIC_CALC">https://github.com/luansheng/PIC_CALC</ext-link>) software to calculate the PIC value of each SNP locus, count the PIC value of SNP on each chromosome and verify the consistency of data distribution before and after GA calculation. The clustering relationship between the fingerprint database composed of core SNPs and the filtered SNP database was compared. Whether the genetic distance and genetic relationship between each individual are consistent with the obtained results were observed and compared as well (<xref ref-type="sec" rid="s12">Supplementary Figure&#x20;S1</xref>).</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec id="s3-1">
<title>SNPs Screening</title>
<p>Screening SNPs according to deletion rate, single variance and linkage relationships between each SNP pairs. The genotype 1/1 was converted to A, 0/0 was converted to B, 0/1 was converted to h and./. was converted to -, which were named as data.mat file. The SNPs with deletion rate of more than 1% and homozygous genotype 0/0 and 1/1 number less than 2 were filtered. A total of 174,819&#x20;high-quality SNPs out of the 7,546,976 original SNPs were selected in 533&#x20;<italic>Oryza sativa</italic> accessions. The remained high-quality SNPs were imported into Matlab program for subsequent calculation and analysis.</p>
<p>According to the same screening conditions, the data of <italic>Solanum tuberosum</italic> and <italic>Sus scrofa</italic> were filtered. Then the correlation among filtered SNPs were calculated. A total of 1,127&#x20;high-quality SNPs out of the 4,786,686 original SNPs were retained in 284&#x20;<italic>Solanum tuberosum</italic> accessions, 6,702&#x20;high-quality SNPs out of the original 3,667,783 SNPs were retained in 247&#x20;<italic>Sus scrofa</italic> accessions and 6,886&#x20;high-quality SNPs out of the original 1,641,026 SNPs were retained in 158&#x20;<italic>Manihot esculenta</italic> Crantz,respectively. (<xref ref-type="table" rid="T1">Table&#x20;1</xref>).</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>SNP number after filtration.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Species</th>
<th align="center">
<italic>Oryza sativa</italic>
</th>
<th align="center">
<italic>Solanum tuberosum</italic>
</th>
<th align="center">
<italic>Sus scrofa</italic>
</th>
<th align="center">
<italic>Manihot esculenta</italic> crantz</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Raw data</td>
<td align="center">7546976</td>
<td align="center">4786686</td>
<td align="center">3667783</td>
<td align="center">1641026</td>
</tr>
<tr>
<td align="left">After filtration</td>
<td align="center">174819</td>
<td align="center">1127</td>
<td align="center">6702</td>
<td align="center">6886</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-2">
<title>Establishment of Fingerprint of <italic>Oryza sativa</italic>, <italic>Solanum tuberosum</italic> and <italic>Sus scrofa</italic> Based on GA</title>
<p>Through GA calculation, the fingerprint matrix constructed by 80, 100, 200 and 300 SNPs was obtained. By calculating the correlation between the fingerprint matrix and original SNP matrix, the fitting degree with original data became high with the increase in the SNP number. However, when there were 100 SNPs, it had been basically fitted with the original data (<xref ref-type="fig" rid="F2">Figure&#x20;2A</xref>). From a numerical perspective, compared with the fingerprint composed of 80 SNPs, the correlation index between the fingerprint composed of 100 SNPs and the original data demonstrated a sudden increase (<xref ref-type="fig" rid="F2">Figure&#x20;2B</xref>). Therefore, the better fingerprint data of 533&#x20;<italic>Oryza sativa</italic> populations were composed of 100 SNPs, and the fingerprint composed of 100 core SNPs was finally retained. In the same way, the fingerprints composed of 80, 100, 200, and 300 SNPs calculated from <italic>Oryza sativa</italic> and <italic>Sus scrofa</italic> were compared with the original data, and the fingerprints composed of 100 SNPs had a good fitting degree with the original data.(<xref ref-type="sec" rid="s12">Supplementary Tables S2&#x2013;S4</xref>).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Calculation of the optimal number of SNPs in fingerprint. <bold>(A)</bold> The fitting diagram of comparing fingerprints composed of different SNPs with original SNP data. Dark blue represents the graph of original SNP data, orange represents the fingerprint curves of 80 SNPs, grey indicates the fingerprint curves of 100 SNPs, light blue shows the fingerprint curves of 200 SNPs and yellow illustrates the fingerprint curves of 300 SNPs. <bold>(B)</bold> The correlation value between each fingerprint and the original SNP&#x20;data.</p>
</caption>
<graphic xlink:href="fgene-13-757524-g002.tif"/>
</fig>
<p>The genotypic values of core SNPs of 533&#x20;<italic>Oryza sativa</italic> varieties, 284&#x20;<italic>Solanum tuberosum</italic> varieties and 247&#x20;<italic>Sus scrofa</italic> varieties were transformed into binary coding data 0, 1 and 2, respectively. The fingerprint database of all <italic>Oryza sativa</italic>, <italic>Solanum tuberosum</italic> and <italic>Sus scrofa</italic> varieties was obtained (<xref ref-type="sec" rid="s12">Supplementary Tables S5&#x2013;S7</xref>). The fingerprint codes obtained were converted into two-dimensional code format, which could better and more conveniently distinguish and identify different variety populations, it has been used in <italic>Manihot esculenta</italic> Crantz. (<xref ref-type="sec" rid="s12">Supplementary Table&#x20;S8</xref>).</p>
</sec>
<sec id="s3-3">
<title>Population Structure Analysis of <italic>Oryza sativa</italic>, <italic>Solanum tuberosum</italic> and <italic>Sus scrofa</italic> Populations</title>
<p>A total of 174,798 SNPs were obtained, which were evenly distributed on 12 chromosomes of <italic>Oryza sativa,</italic> after two-step filtration. The distribution, number and density of SNPs on 12 chromosomes of <italic>Oryza sativa</italic> were showed in <xref ref-type="fig" rid="F3">Figure&#x20;3A</xref>. At the whole genome level of <italic>Oryza sativa</italic>, the average polymorphism information content was 0.363. The polymorphism information content of most SNPs was between 0.365 and 0.37, and the polymorphism information content of very few SNPs was between 0.35 and 0.355. These values indicated that the polymorphism information content of <italic>Oryza sativa</italic> SNPs was high and distributed evenly. After selection and GA calculation, a fingerprint map constructed by 100 core SNPs was obtained. The SNPs were evenly distributed on 12 chromosomes of <italic>Oryza sativa</italic>. <xref ref-type="fig" rid="F3">Figure&#x20;3C</xref> shows the distribution, number and density of 100 SNPs of fingerprint on 12 chromosomes of <italic>Oryza sativa</italic>. The selected core SNPs were at the whole genome level of <italic>Oryza sativa</italic>, and the average polymorphism information content was 0.364. The PIC values of 100 SNPs were mostly distributed between 0.365 and 0.37, with a total number of 32. The data showed that the obtained 100 core SNPs calculated by GA were consistent with the original filtered&#x20;data.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Variation statistics of SNPs in <italic>Oryza sativa</italic> before and after genetic algorithm calculation. <bold>(A)</bold> The density distribution of original SNPs of <italic>Oryza sativa</italic>. <bold>(B)</bold> The PIC value statistics before and after genetic algorithm calculation. <bold>(C)</bold> The density distribution of 100 SNPs of <italic>Oryza sativa</italic> fingerprint.</p>
</caption>
<graphic xlink:href="fgene-13-757524-g003.tif"/>
</fig>
<p>The <italic>Oryza sativa</italic> population structure analysis of all the 174,798&#x20;high-quality SNPs was carried out by admixture software. The maximum cluster subgroup number (K) was estimated to be a certain value between 2 and 14, and the cross-validation error rate (CV error) under each K value was calculated. When K was from 1 to 9, CV error decreased rapidly. When K was greater than 9, CV error increased gradually and tended to be flat (<xref ref-type="fig" rid="F6">Figure&#x20;6A</xref>). Therefore, when K was equal to 9, CV error was the smallest, indicating that K &#x3d; 9 was the most suitable, that is, the whole <italic>Oryza sativa</italic> population was divided into nine subgroups, namely, subgroup 1 to subgroup 9. In R, according to K &#x3d; 9, the <italic>Oryza sativa</italic> population was divided into nine subgroups. K-means clustering analysis was carried out on the filtered original SNP database and the obtained fingerprint database, and the K-value matrix was obtained. Genetic distance was calculated by VCF2DiffMatrix software, which was imported into the Neighbour software to obtain a tree file. The nine subgroups were classified and expressed by nine different colours, and the evolutionary tree was visualised in iTOL to obtain clustering results. The clustering results of 100 SNP fingerprint database were basically similar to those of all 174,798 SNPs, indicating that the use of 100 SNPs for fingerprint construction was effective (<xref ref-type="fig" rid="F6">Figure&#x20;6B</xref>).</p>
<p>A total of 1,127 SNPs were obtained, after two-step filtration, which were distributed evenly on 12 chromosomes of <italic>Solanum tuberosum</italic>. <xref ref-type="fig" rid="F4">Figure&#x20;4A</xref> shows the distribution, number and density of SNPs on 12 chromosomes of <italic>Solanum tuberosum</italic>. The number of mutations on each chromosome ranged from 38 to 135, with an average of 94; the largest number of SNPs 135) was on chromosome 1. By contrast, there were 38 SNPs, the least number of SNPs, on chromosome 2. At the whole genome level of <italic>Solanum tuberosum</italic>, the average polymorphism information content was 0.249. The polymorphism information content of most SNPs was between 0.24 and 0.26, and the polymorphism information content of very few SNPs was between 0.2 and 0.22. These values indicated that the polymorphism information content of <italic>Solanum tuberosum</italic> SNPs was relatively even. After selection and GA calculation, a fingerprint constructed by 100 core SNPs was obtained, and the SNPs were distributed evenly on 12 chromosomes of <italic>Solanum tuberosum</italic>. <xref ref-type="fig" rid="F4">Figure&#x20;4C</xref> shows the distribution, number and density of 100 SNPs of fingerprint on 12 chromosomes of <italic>Solanum tuberosum</italic>. The statistical results showed that the number of SNPs on chromosome 1 was the largest (36). However, the one with the least number of SNPs occurred on chromosome 5, followed with chromosome 2. These observations showed that the distribution of sites was relatively even and consistent with the distribution of the original calculated data filtered. The selected core locus was at the whole genome level of <italic>Solanum tuberosum</italic>, and the average polymorphism information content was 0.255. The PIC values of 100 SNPs were mostly distributed between 0.24 and 0.26, with a total of number of 39. The average polymorphism information content of the SNPs on each chromosome was between 0.241 and 0.268. The highest average polymorphism information content was on chromosome 9, reaching 0.268, while the lowest was on chromosome 2, which was only 0.2411. After the calculation of GA, the obtained 100 core SNPs were consistent with the original filtered&#x20;data.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Variation statistics of SNPs in <italic>Solanum tuberosum</italic> before and after genetic algorithm calculation. <bold>(A)</bold> The density distribution of original SNPs of <italic>Solanum tuberosum</italic>. <bold>(B)</bold> The PIC value statistics before and after genetic algorithm calculation. <bold>(C)</bold> The density distribution of 100 SNPs of <italic>Solanum tuberosum</italic> fingerprint.</p>
</caption>
<graphic xlink:href="fgene-13-757524-g004.tif"/>
</fig>
<p>The admixture software was used to carry out the <italic>Solanum tuberosum</italic> population structure analysis of all 1,127&#x20;high-quality SNPs. The maximum number of cluster subsets (K) was inferred to be a certain value between 1 and 10, and the CV error under each K value was calculated. When K was from 1 to 7, CV error decreased rapidly. After K was greater than 7, CV error increased gradually and tended to be flat (<xref ref-type="fig" rid="F6">Figure&#x20;6A</xref>). Therefore, when K was equal to 7, CV error was the smallest, indicating that K &#x3d; 7 was the most suitable, that is, the whole <italic>Solanum tuberosum</italic> population was divided into seven subgroups, namely, subgroup 1 to subgroup 7. In R, according to K &#x3d; 7, the <italic>Solanum tuberosum</italic> population was divided into seven subgroups. K-means clustering analysis was carried out on the filtered original SNP database and the obtained fingerprint database, and the K-value matrix was obtained. Genetic distance was calculated by VCF2DiffMatrix software, which was imported into the Neighbour software to obtain a tree file. Seven subgroups were classified and expressed by seven different colours, and the evolutionary tree was visualised in iTOL to obtain clustering results. The clustering results of 100 SNP fingerprint database were basically similar to those of all 1,127 SNPs, indicating that the use of 100 SNPs for fingerprint construction was effective (<xref ref-type="fig" rid="F6">Figure&#x20;6B</xref>).</p>
<p>A total of 6,702 SNPs were obtained after two-step filtration, which were distributed evenly on 21 chromosomes of <italic>Sus scrofa</italic>. <xref ref-type="fig" rid="F5">Figure&#x20;5A</xref> shows the distribution, number and density of SNPs on 21 chromosomes of <italic>Sus scrofa</italic>. At the whole genome level of <italic>Sus scrofa</italic>, the average polymorphism information content was 0.217. The polymorphism information content of most SNPs was between 0.2 and 0.25, and the polymorphism information content of very few SNPs was between 0.3 and 0.35. After selection and GA calculation, a fingerprint constructed by 100 core SNPs was obtained, and the SNPs were distributed relatively evenly on 21 chromosomes of <italic>Sus scrofa</italic>. <xref ref-type="fig" rid="F5">Figure&#x20;5C</xref> shows the distribution, number and density of 100 SNPs of fingerprint on 21 chromosomes of <italic>Sus scrofa</italic>. After the GA calculation, the distributions of SNPs on each chromosome were relatively even, which were consistent with the distributions of the filtered original calculation data. The selected core locus was at the whole genome level of <italic>Sus scrofa</italic>, and the average polymorphism information content was 0.216. The PIC values of 100 SNPs were mostly distributed between 0.2 and 0.25, with a total of number of 49. After the calculation of GA, the obtained&#x20;100 core SNPs were consistent with the original filtered&#x20;data.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Variation statistics of SNPs in <italic>Sus scrofa</italic> before and after genetic algorithm calculation. <bold>(A)</bold> The density distribution of original SNPs of <italic>Sus scrofa</italic>. <bold>(B)</bold> The PIC value statistics before and after genetic algorithm calculation. <bold>(C)</bold> The density distribution of 100 SNPs of <italic>Sus scrofa</italic> fingerprint.</p>
</caption>
<graphic xlink:href="fgene-13-757524-g005.tif"/>
</fig>
<p>The admixture software was used to carry out the <italic>Sus scrofa</italic> population structure analysis of all 6,702&#x20;high-quality SNPs. The maximum number of cluster subsets (K) was inferred to be a certain value between 1 and 12, and the CV error under each K value was calculated. When K was from 1 to 9, CV error decreased rapidly. When K was greater than 9, CV error increased gradually and tended to be flat (<xref ref-type="fig" rid="F6">Figure&#x20;6A</xref>). Therefore, when K was equal to 9, CV error was the smallest, indicating that K &#x3d; 9 was the most suitable, that is, the whole <italic>Sus scrofa</italic> population was divided into nine subgroups, namely, subgroup 1 to subgroup 9. In R, according to K &#x3d; 9, the <italic>Sus scrofa</italic> population was divided into nine subgroups. K-means clustering analysis was carried out on the filtered original SNP database and the obtained fingerprint database, and the K-value matrix was obtained. Genetic distance was calculated by VCF2DiffMatrix software, which was imported into the Neighbour software to obtain a tree file. The nine subgroups were classified and expressed by nine different colours, and the evolutionary tree was visualised in iTOL to obtain clustering results. The clustering results of 100 SNP fingerprint database were basically similar to those of all 6,702 SNPs, indicating that the use of 100 SNPs for fingerprint construction was effective (<xref ref-type="fig" rid="F6">Figure&#x20;6B</xref>).</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Cluster comparison of each species before and after genetic algorithm calculation. <bold>(A)</bold> The CV error value statistics and the Nj cluster tree of 533&#x20;<italic>Oryza sativa</italic> populations. The left one shows the clustering of original data, and the right one shows the clustering of fingerprint data. Different colours represent different subgroups. <bold>(B)</bold>. The CV error value statistics and the Nj cluster tree of 284&#x20;<italic>Solanum tuberosum</italic> populations. <bold>(C)</bold>. The CV error value statistics and the Nj cluster tree of 247&#x20;<italic>Sus scrofa</italic> populations.</p>
</caption>
<graphic xlink:href="fgene-13-757524-g006.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>Application of Core SNP in <italic>Solanum tuberosum</italic> Variety Identification</title>
<p>To determine the genetic relationship among <italic>Solanum tuberosum</italic> varieties in detail and whether the fingerprint is true and effective, the genetic distances among 284 individuals were calculated by evaluating and statistically utilising the DNA-SNP database calculated by GA. The average genetic distance among all varieties was 0.15359, ranging from 0.05 to 0.415. When the genetic distance was the minimum value of 0.1, there were two groups of individuals (R3R4 and CIP09-29 and KW-59 and KW-11). Every two individuals in these two groups of individuals were similar in origin, and they all came from CIP. There were two groups of individuals (Sebago-1 and Ziyun and CIP09-29 and zhongshu18) when the genetic distance was the maximum value of 0.65. These two groups of individuals had different origins. The origin of Sebago-1 was unknown, while Ziyun came from China, CIP09-29 came from CIP and zhongshu18 came from China. We compared the genotypes of 100 SNP loci and found that the number of different loci between R3R4 and CIP09-29 is 28, the number of different loci between KW-59 and KW-11 is 25, and that between Sebago-1 and Ziyun The number of different loci is 38, and the number of different loci between CIP09-29 and zhongshu18 is 42, which is significantly more than the previous two groups. These findings showed that the fingerprint could show the origin and genetic relationship of the <italic>Solanum tuberosum</italic> population, and the results were reliable and available.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<sec id="s4-1">
<title>Construction of Fingerprint Based on SNP Genotype Data</title>
<p>DNA fingerprinting is an applied genetic germplasm analysis method established with the development of molecular biology. There are different traits and DNA sequences among various species and varieties, and varieties can be distinguished according to these differences (<xref ref-type="bibr" rid="B29">Sharon et&#x20;al., 1992</xref>).</p>
<p>At the end of the 1970s, the technology of studying the differences in DNA among different individuals was proposed and applied. The differences in DNA molecules can directly reflect the differences between individuals and are not affected by the external environment. Therefore, it gradually replaces the previous method of observing external phenotypic differences with the naked eyes to distinguish individuals. With the development of molecular biology, various DNA molecular marker technologies have been widely studied and applied, from RFLP, AFLP to later developed SNP. These molecular markers have played a critical role in the analysis and research of fingerprint constructions.</p>
<p>However, the previous molecular markers have some shortcomings. Relatively speaking, RFLP technology is cumbersome to operate and has low polymorphic information content, which has limitations in individual recognition. The RAPD technique has poor repeatability and unstable results, so it can obtain reliable results only by optimising the reaction conditions repeatedly. AFLP technology has high requirements on genome purity and reaction conditions. Compared with the previous ones, SSR technology has the advantages of high polymorphism detection rate, large information content, simple experimental operation and stable and reliable results, but it also has the disadvantages of low detection efficiency and low repeatability (<xref ref-type="bibr" rid="B31">Wang, 2006</xref>). Li et&#x20;al. (<xref ref-type="bibr" rid="B19">Li et&#x20;al., 2017</xref>)developed a novel method for SSR genotyping, named as AmpSeq-SSR, which combines multiplexing polymerase chain reaction (PCR),targeted deep sequencing and comprehensive analysis.The identification fingerprint of that rice variety construct by 3,105 ssr markers is obtain by the method, which offer much greater discriminative power than the 48 SSRs commonly used for rice. As the third-generation molecular marker, SNP has unique advantages, which solves the unrepeatability of SSR markers well. It has the following advantages. Firstly, the SNPs are abundant and almost all over the whole genome. Secondly, it is rich in polymorphism and has higher genetic stability than SSR. In addition, SNP detection methods are becoming increasingly simple and efficient and have excellent application prospects.</p>
<p>Traditionally, SSR is used to evaluate the genetic characteristics of species (<xref ref-type="bibr" rid="B20">Mba et&#x20;al., 2001</xref>; <xref ref-type="bibr" rid="B11">Hurtado et&#x20;al., 2008</xref>). However, SSR rarely produces multi-allele markers in <italic>Solanum tuberosum</italic>, which makes the study of genetic identification or genetic variation time-consuming and laborious. SNP is a double allele genotype marker, which is more abundant than SSR, easy to be automated, highly repetitive and can be shared among laboratories (<xref ref-type="bibr" rid="B25">Primmer et&#x20;al., 2002</xref>; <xref ref-type="bibr" rid="B6">Floro et&#x20;al., 2018</xref>). Therefore, SNP was selected to complete this&#x20;study.</p>
</sec>
<sec id="s4-2">
<title>Construction of Fingerprint by GA</title>
<p>In the field of artificial intelligence, finding the optimal solution or approximate optimal solution in a very large and complex space is a key problem that has always existed. For NP-hard problems (<xref ref-type="bibr" rid="B14">Komusiewicz et&#x20;al., 2014</xref>), if an improper search method is used, problems such as combinatorial explosion may occur. Therefore, researchers in related fields have been devoted to find a general search algorithm. Pre-exact algorithm and intelligent optimisation algorithm are common solutions to function optimisation problems. Accurate algorithms are time-consuming and suitable for small-scale problems. Intelligent optimisation algorithms mainly include simulated annealing algorithm proposed by reference (<xref ref-type="bibr" rid="B9">Holland, 2006</xref>) in 1953, GA proposed by reference (<xref ref-type="bibr" rid="B7">Glover, 1986</xref>) in 1973, tabu algorithm proposed by reference (<xref ref-type="bibr" rid="B5">Colorni et&#x20;al., 1992</xref>) in 1986, ant colony algorithm proposed by reference (<xref ref-type="bibr" rid="B10">Hopfield, 1982</xref>) in 1992 and neural network method used by reference (<xref ref-type="bibr" rid="B13">Jungnickel, 2008</xref>) in an innovative&#x20;way.</p>
<p>Genetic algorithm, Particle Swarm optimization, and Differential Evolution Algorithm are all branches of evolutionary algorithm. Many scholars have conducted research on these algorithms. Through continuous improvement, the performance of the algorithm has been improved, and the application field has been expanded. Therefore, it is necessary to discuss these algorithms. Characteristics, according to the adaptability of different application fields and algorithms, it will be very meaningful work to recommend different algorithms for use. After comparing these three algorithms, it is found that the DE algorithm has the best performance, and the algorithm is relatively stable, and repeated operations can converge to the same solution; the PSO algorithm has the second highest convergence speed, but the algorithm is unstable, and the final convergence result is easily affected by the parameter size. And the initial population; the GA algorithm has a relatively slow convergence speed, but in terms of dealing with noise problems, GA can solve the noise problem very well, but it is difficult for the DE algorithm to deal with this noise problem.</p>
<p>GA simulates the genetic evolutionary mechanism of human beings and organisms, which is mainly based on Darwin&#x2019;s theory of biological evolution: natural selection and survival of the fittest. The specific implementation process is as follows: &#x2460;The individuals that adapt to the environment and perform well are selected from the initial generation population. &#x2461;Using genetic operators, the screened individuals are combined, crossed and mutated, and the second-generation population is generated. &#x2462;Individuals with good environmental adaptability are selected from the second-generation population and combined, crossed and mutated to form the third-generation population. The evolution continues in this way until the final population is generated, that is, the approximate optimal solution of the problem (<xref ref-type="bibr" rid="B17">Kumar et&#x20;al., 2010</xref>). The principle of GA applied to this problem can be summarised as follows: 1) Firstly, a population <italic>S</italic> is determined. The number of individuals in this population is <italic>L</italic>, and each individual corresponds to <italic>M</italic> SNPs. 2) Set the fitness as <italic>Fi</italic>, and the fitness of each individual is:<disp-formula id="equ7">
<mml:math id="m17">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>
<italic>Ui</italic> represents the number of samples that the population can uniquely identify, and <italic>N</italic> represents the total number of samples. Let the probability be <italic>Pi</italic>, and the <italic>P</italic> value of each individual is:<disp-formula id="equ8">
<mml:math id="m18">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Order the <italic>P</italic> values and calculate sort (<italic>P</italic>). 3) Individuals with high <italic>P</italic> value are selected and hybridised to generate new individuals. 4) The generated sub-individuals are placed into the original population, and the individual fitness and probability are recalculated through step 2. The corresponding number of individuals with low probability are eliminated. 5) Steps 2, 3 and 4 are repeated until the set number of iterations of the population is reached. 6) When the iterations are completed, the individual with the highest probability in the population is selected, which is the best SNP combination to be&#x20;found.</p>
<p>The program written by GA can find the SNP combination that can distinguish all samples by computer language calculation. It has the advantages of high efficiency, convenience and high reliability, and it is a highlight of this&#x20;study.</p>
</sec>
<sec id="s4-3">
<title>Application Prospect of SNP Fingerprint Database Based on GA</title>
<p>In this paper, based on the SNP data of 533&#x20;<italic>Oryza sativa</italic>, 284&#x20;<italic>Solanum tuberosum</italic> and 247&#x20;<italic>Sus scrofa</italic> accessions, 100 core SNPs for fingerprint construction were selected from the original SNP data by GA calculation. All varieties of each species could be distinguished and identified by using these core SNPs. Each variety had its own fingerprint code, by which different varieties could be classified and identified. The two-dimensional code of <italic>Manihot esculenta</italic> Crantz fingerprint obtained by this method has been applied to field planting. Therefore, this method can be used for various species, has wide application and high practicability. In addition, the fingerprints obtained by GA were tested by the population structure. A small number of core SNPs used for fingerprint construction could basically restore the population structure of the original SNP data. Better restoration results were obtained in the populations of model crop <italic>Oryza sativa</italic>, autotetraploid <italic>Solanum tuberosum</italic> and mammalian <italic>Sus scrofa</italic>. For example, original SNPs from 533&#x20;<italic>Oryza sativa</italic> were to divide the population into nine subgroups, while the subgroup classifications of core SNPs were to divide the population into nine subgroups; the population structures were also highly consistent. Through the program written by GA, the core SNPs calculated had high accuracy; this set of core SNPs combination may provide new ideas and methods for identification analysis and genetic diversity analysis of each species (<xref ref-type="bibr" rid="B28">Semagn et&#x20;al., 2014</xref>)<sup>,38]</sup>.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>Conclusion</title>
<p>In this paper, using SNP data of three representative species and the concept of GA, the SNPs with a high deletion rate and high single variation rate were finally screened out. A total of 100 SNPs that could distinguish all varieties in the population for constructing the model crop <italic>Oryza sativa</italic>, olyploid crop <italic>Solanum tuberosum</italic> and mammal <italic>Sus scrofa</italic> were obtained to construct the fingerprint. The proposed method is innovative and widely applied; it provides a new idea for genetic research and animal and plant breeding and has good application prospect.</p>
</sec>
</body>
<back>
<sec id="s6">
<title>Data Availability Statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: NCBI BioProject PRJNA171289; CNCB-NGDC BioProject PRJCA005945; GenBank GCA_000003025.6.</p>
</sec>
<sec id="s7">
<title>Ethics Statement</title>
<p>Ethical review and approval was not required for the animal study because Animal data used in this article have been published.</p>
</sec>
<sec id="s8">
<title>Author Contributions</title>
<p>MZ and ZX are the experimental designers of this study; ZX provided experimental materials for this study; SJ, CZ, LZ, and YB were the executors of the main experimental study and analysis of experimental results. SJ, MZ, ZX complete data analysis, the writing of the first draft of the paper; YC and FW participated in the experimental design and analysis of the experimental results. MZ is the designer and main person in charge of the project; MZ and SJ is responsible for the revision of the paper; ZX directed experimental design, data analysis, and paper writing and revision. All authors read and agree on the final&#x20;text.</p>
</sec>
<sec id="s9">
<title>Funding</title>
<p>This work was supported by National Key R&#x26;D Program of China (2019YFD1000500), Integrated demonstration of key techniques for the industrial development of featured crops in rocky desertification areas of Yunnan-Guangxi-Guizhou provinces (SMH2019-2021), Central Public-interest Scientific Institution Basal Research Fund for Chinese Academy of Tropical Agricultural Sciences (No. 1630052019022) and Hainan University Startup Fund (KYQD(ZR)-20101).</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2022.757524/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2022.757524/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.PDF" id="SM1" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Azevedo</surname>
<given-names>A. O. N.</given-names>
</name>
<name>
<surname>Azevedo</surname>
<given-names>C. D. d. O.</given-names>
</name>
<name>
<surname>Santos</surname>
<given-names>P. H. A. D.</given-names>
</name>
<name>
<surname>Ramos</surname>
<given-names>H. C. C.</given-names>
</name>
<name>
<surname>Boechat</surname>
<given-names>M. S. B.</given-names>
</name>
<name>
<surname>Ar&#xea;des</surname>
<given-names>F. A. S.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Selection of Legitimate&#x20;dwarf Coconut Hybrid Seedlings Using DNA Fingerprinting</article-title>. <source>Crop Breed. Appl. Biotechnol.</source> <volume>18</volume> (<issue>4</issue>), <fpage>409</fpage>&#x2013;<lpage>416</lpage>. <pub-id pub-id-type="doi">10.1590/1984-70332018v18n4a60</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baird</surname>
<given-names>N. A.</given-names>
</name>
<name>
<surname>Etter</surname>
<given-names>P. D.</given-names>
</name>
<name>
<surname>Atwood</surname>
<given-names>T. S.</given-names>
</name>
<name>
<surname>Currey</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Shiver</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Lewis</surname>
<given-names>Z. A.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Rapid SNP Discovery and Genetic Mapping Using Sequenced RAD Markers</article-title>. <source>Plos One</source> <volume>3</volume> (<issue>10</issue>), <fpage>e3376</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0003376</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Genome-wide Association Analyses Provide Genetic and Biochemical Insights into Natural Variation in rice Metabolism</article-title>. <source>Nat. Genet.</source> <volume>46</volume>, <fpage>714</fpage>&#x2013;<lpage>721</lpage>. <pub-id pub-id-type="doi">10.1038/ng.3007</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y. H.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>J.&#x20;H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C. Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Weng</surname>
<given-names>M. L.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Rice Seed Identification by Computerized AFLP-DNA Fingerprint-Ing</article-title>. <source>Chin. Rice Res. Newsl.</source> (<issue>01</issue>), <fpage>4</fpage>&#x2013;<lpage>5</lpage>. </citation>
</ref>
<ref id="B5">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Colorni</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dorigo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Maniezzo</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>1992</year>). <source>Distributed Optimization by Ant Colonies. European Conference on Artificial Life</source>. <publisher-name>The MIT Press</publisher-name>. </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Floro</surname>
<given-names>V. O.</given-names>
</name>
<name>
<surname>Labarta</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Becerra L&#xf3;pez-Lavalle</surname>
<given-names>L. A.</given-names>
</name>
<name>
<surname>Martinez</surname>
<given-names>J.&#x20;M.</given-names>
</name>
<name>
<surname>Ovalle</surname>
<given-names>T. M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Household Determinants of the Adoption of Improved Cassava Varieties Using DNA Fingerprinting to Identify Varieties in Farmer Fields: A Case Study in Colombia</article-title>. <source>J.&#x20;Agric. Econ.</source> <volume>69</volume> (<issue>2</issue>), <fpage>518</fpage>&#x2013;<lpage>536</lpage>. <pub-id pub-id-type="doi">10.1111/1477-9552.12247</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Glover</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>1986</year>). <article-title>Future Paths for Integer Programming and Links to Artificial Intelligence</article-title>. <source>Comput. Operations Res.</source> <volume>13</volume> (<issue>5</issue>), <fpage>533</fpage>&#x2013;<lpage>549</lpage>. <pub-id pub-id-type="doi">10.1016/0305-0548(86)90048-1</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Goldberg</surname>
<given-names>D. E.</given-names>
</name>
</person-group> (<year>1989</year>). <source>Genetic Algorithms in Search, Optimization, and Machine Learning</source>. <publisher-name>Addison-Wesley. Publishing Company</publisher-name>. </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Holland</surname>
<given-names>J.&#x20;H.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Erratum: Genetic Algorithms and the Optimal Allocation of Trials</article-title>. <source>Siam J.&#x20;Comput.</source> <volume>2</volume> (<issue>2</issue>), <fpage>88</fpage>&#x2013;<lpage>105</lpage>. <pub-id pub-id-type="doi">10.1137/0203026</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hopfield</surname>
<given-names>J.&#x20;J.</given-names>
</name>
</person-group> (<year>1982</year>). <article-title>Neural Networks and Physical Systems with Emergent Collective Computational Abilities</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>79</volume> (<issue>8</issue>), <fpage>2554</fpage>&#x2013;<lpage>2558</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.79.8.2554</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hurtado</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Olsen</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Buitrago</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ospina</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Marin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Duque</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Comparison of Simple Sequence Repeat (SSR) and Diversity Array Technology (DArT) Markers for Assessing Genetic Diversity in Cassava (Manihot Esculenta Crantz)</article-title>. <source>Plant Genet. Res.</source> <volume>6</volume>, <fpage>208</fpage>&#x2013;<lpage>214</lpage>. <pub-id pub-id-type="doi">10.1017/s1479262108994181</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jooyeong</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Bo</surname>
<given-names>R. C.</given-names>
</name>
<name>
<surname>Juyeon</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hyukjin</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Highly Selective Detection of Single Nucleotide Polymorphism (SNP) Using a Dumbbell DNA Probe with a gap-filling Approach</article-title>. <source>J.&#x20;Ind. Eng. Chem.</source> <volume>88</volume>, <fpage>78</fpage>&#x2013;<lpage>83</lpage>. <pub-id pub-id-type="doi">10.1016/j.jiec.2020.03.028</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jungnickel</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>The Greedy Algorithm</article-title>. <source>Springer Berlin Heidelberg</source> <volume>1999</volume>, <fpage>135</fpage>&#x2013;<lpage>161</lpage>. <pub-id pub-id-type="doi">10.1007/3-540-26908-8_5</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Komusiewicz</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bulteau</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>H&#xfc;ffner</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Niedermeier</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Multivariate Algorithmics for NP-Hard String Problems</article-title>. <source>Bull. Eatcs</source> <volume>114</volume>(<issue>114</issue>). </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kopp</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Smart</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Maynard</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Tuskan</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Abrahamson</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Predicting Within-Family Variability in Juvenile Height Growth of Salix Based upon Similarity Among Parental AFLP Fingerprintsfingerprints</article-title>. <source>Theor. Appl. Genet.</source> <volume>105</volume>, <fpage>106</fpage>&#x2013;<lpage>112</lpage>. <pub-id pub-id-type="doi">10.1007/s00122-001-0855-3</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ghose</surname>
<given-names>M. K.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Overview of Information Security Using Genetic Algorithm and Chaos</article-title>. <source>Inf. Security J.&#x20;A Glob. Perspective</source> <volume>18</volume> (<issue>6</issue>), <fpage>306</fpage>&#x2013;<lpage>315</lpage>. <pub-id pub-id-type="doi">10.1080/19393550903327558</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Husian</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Upreti</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Genetic Algorithm: Review and Application</article-title>. <source>Computer</source> <volume>2</volume> (<issue>2</issue>), <fpage>451</fpage>&#x2013;<lpage>454</lpage>. <pub-id pub-id-type="doi">10.2139/ssrn.3529843</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Establishiment of DNA Fingerprinting for Wheat in Shandong Province by SSR Markers</article-title>. <source>J.&#x20;Plant Genet. Resour.</source> <volume>14</volume> (<issue>35</issue>), <fpage>537</fpage>&#x2013;<lpage>541</lpage>. </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>An Accurate and Efficient Method for Large-Scale SSR Genotyping and Applications</article-title>. <source>Nucleic Acids Res.</source> <volume>45</volume> (<issue>10</issue>), <fpage>e88</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkx093</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mba</surname>
<given-names>R. E. C.</given-names>
</name>
<name>
<surname>Stephenson</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Edwards</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Melzer</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nkumbira</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gullberg</surname>
<given-names>U.</given-names>
</name>
<etal/>
</person-group> (<year>2001</year>). <article-title>Simple Sequence Repeat (SSR) Markers Survey of the Cassava (Manihot Esculenta Crantz) Genome: towards an SSR-Based Molecular Genetic Map of Cassava</article-title>. <source>Theor. Appl. Genet.</source> <volume>102</volume>, <fpage>21</fpage>&#x2013;<lpage>31</lpage>. <pub-id pub-id-type="doi">10.1007/s001220051614</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>McGregor</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Lambert</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Greyling</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Louw</surname>
<given-names>J.&#x20;H.</given-names>
</name>
<name>
<surname>Warnich</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2000</year>). <source>A Comparative Assessment of DNA Fingerprinting Techniques (RAPD, ISSR, AFLP and SSR) in Tetraploid Potato (<italic>Solanum tuberosum</italic> L.) Germplasm</source>. <publisher-loc>Netherlands</publisher-loc>: <publisher-name>Euphytica</publisher-name>. </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Muhammad</surname>
<given-names>A. N.</given-names>
</name>
<name>
<surname>Muhammad</surname>
<given-names>A. N.</given-names>
</name>
<name>
<surname>Muhammad</surname>
<given-names>Q. S.</given-names>
</name>
<name>
<surname>Y&#x131;ld&#x131;z</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Gonul</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mehtap</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>DNA Molecular Markers in Plant Breeding: Current Status and Recent Advancements in Genomic Selection and Genome Editing</article-title>. <source>Plant Breed.</source> <volume>32</volume> (<issue>2</issue>), <fpage>261</fpage>&#x2013;<lpage>285</lpage>. <pub-id pub-id-type="doi">10.1080/13102818.2017.1400401</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Palaiokostas</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ferraresso</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Franch</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Houston</surname>
<given-names>R. D.</given-names>
</name>
<name>
<surname>Bargelloni</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Genomic Prediction of Resistance to Pasteurellosis in Gilthead Sea Bream (Sparus Aurata) Using 2b-RAD Sequencing</article-title>. <source>G</source> <volume>6</volume> (<issue>11</issue>), <fpage>3693</fpage>&#x2013;<lpage>3700</lpage>. <pub-id pub-id-type="doi">10.1534/g3.116.035220</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peterson</surname>
<given-names>B. K.</given-names>
</name>
<name>
<surname>Weber</surname>
<given-names>J.&#x20;N.</given-names>
</name>
<name>
<surname>Kay</surname>
<given-names>E. H.</given-names>
</name>
<name>
<surname>Fisher</surname>
<given-names>H. S.</given-names>
</name>
<name>
<surname>Hoekstra</surname>
<given-names>H. E.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Double Digest RADseq: an Inexpensive Method for De Novo SNP Discovery and Genotyping in Model and Non-model Species</article-title>. <source>Plos One</source> <volume>7</volume> (<issue>5</issue>), <fpage>e37135</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0037135</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Primmer</surname>
<given-names>C. R.</given-names>
</name>
<name>
<surname>Borge</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lindell</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Saetre</surname>
<given-names>G.-P.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Single-nucleotide Polymorphism Characterization in Species with Limited Available Sequence Information: High Nucleotide Diversity Revealed in the Avian Genome</article-title>. <source>Mol. Ecol.</source> <volume>11</volume> (<issue>3</issue>), <fpage>603</fpage>&#x2013;<lpage>612</lpage>. <pub-id pub-id-type="doi">10.1046/j.0962-1083.2001.01452.x</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Purcell</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Neale</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Todd-Brown</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Thomas</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>M. A. R.</given-names>
</name>
<name>
<surname>Bender</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>PLINK: A Tool Set for Whole-Genome Association and Population-Based Linkage Analyses</article-title>. <source>Am. J.&#x20;Hum. Genet.</source> <volume>81</volume> (<issue>3</issue>), <fpage>559</fpage>&#x2013;<lpage>575</lpage>. <pub-id pub-id-type="doi">10.1086/519795</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Robert</surname>
<given-names>J.&#x20;E.</given-names>
</name>
<name>
<surname>Jeffrey</surname>
<given-names>C. G.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jesse</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Ken</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Edward</surname>
<given-names>S. B.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>A Robust, Simple Genotyping-By-Sequencing (GBS) Approach for High Diversity Species</article-title>. <source>Plos One</source> <volume>6</volume> (<issue>5</issue>), <fpage>e19379</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0019379</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Semagn</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Babu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hearne</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Olsen</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Single Nucleotide Polymorphism Genotyping Using Kompetitive Allele Specific PCR (KASP): Overview of the Technology and its Application in Crop Improvement</article-title>. <source>Mol. Breed.</source> <volume>33</volume>, <fpage>1</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1007/s11032-013-9917-x</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sharon</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hillel</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Vainstein</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lavi</surname>
<given-names>U.</given-names>
</name>
</person-group> (<year>1992</year>). <article-title>Application of DNA Fingerprints for Identification and Genetic Analysis of Carica Papaya and Other Carica Species</article-title>. <source>Euphytica</source> <volume>62</volume> (<issue>2</issue>), <fpage>119</fpage>&#x2013;<lpage>126</lpage>. <pub-id pub-id-type="doi">10.1007/bf00037937</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Eli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>John</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Mikhail</surname>
<given-names>V. M.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>2b-RAD: a Simple and Flexible Method for Genome-wide Genotyping</article-title>. <source>Nat. Methods</source> <volume>9</volume> (<issue>8</issue>), <fpage>808</fpage>&#x2013;<lpage>810</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2023</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Z. H.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>DNA Fingerprinting and its Application in Crop Variety Resources</article-title>. <source>Mol. Plant Breed.</source> (<issue>03</issue>), <fpage>425</fpage>&#x2013;<lpage>430</lpage>. <pub-id pub-id-type="doi">10.3969/j.issn.1672-416X.2006.03.020</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>C. W.</given-names>
</name>
<name>
<surname>Rul&#x27;kov</surname>
<given-names>N. F.</given-names>
</name>
</person-group> (<year>1993</year>). <article-title>Studying Chaos via 1-D Maps-A Tutorial</article-title>. <source>IEEE Trans. Circuits Syst.</source> <volume>40</volume> (<issue>10</issue>), <fpage>707</fpage>&#x2013;<lpage>721</lpage>. <pub-id pub-id-type="doi">10.1109/81.246147</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xia</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>AFSM Sequencing Approach: a Simple and Rapid Method for Genome-wide SNP and Methylation Site Discovery and Genetic Mapping</article-title>. <source>Sci. Rep.</source> <volume>4</volume>, <fpage>7300</fpage>. <pub-id pub-id-type="doi">10.1038/srep07300</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhan</surname>
<given-names>Q. C.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>Z. F.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z. X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z. H.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>S. Z.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Construction of DNA Fingerprint Using SSR Marker for Hybrid Rice Cultivars Approved by Hunan Province</article-title>. <source>Agric. Biotechnol.</source> <volume>1</volume> (<issue>04</issue>), <fpage>5</fpage>. </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Genome-Wide Association Studies of 11 Agronomic Traits in Cassava (Manihot Esculenta Crantz)</article-title>. <source>Front. Plant Sci.</source> <volume>9</volume>, <fpage>503</fpage>. <pub-id pub-id-type="doi">10.3389/fpls.2018.00503</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>G. J.</given-names>
</name>
<name>
<surname>Xu-Jiang</surname>
<given-names>L. U.</given-names>
</name>
<name>
<surname>Nan</surname>
<given-names>H. E.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W. G.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Construction of DNA Fingerprinting and Analysis of Genetic Diversity with SSR Markers for Seedless Watermelon Major Varieties in China</article-title>. <source>J.&#x20;Plant Genet. Resour.</source> <volume>14</volume> (<issue>6</issue>), <fpage>1142</fpage>&#x2013;<lpage>1146</lpage>. <pub-id pub-id-type="doi">10.3724/sp.j.1006.2009.01451</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhuk</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2008</year>). <source>Latvian State Forestry Research Inst. Silava., Characterization of Latvian Potato Genetic Resources by DNA Fingerprinting with SSR Markers</source>. <publisher-loc>Latvia</publisher-loc>: <publisher-name>Latvian Journal of Agronomy</publisher-name>, <fpage>171</fpage>&#x2013;<lpage>178</lpage>. </citation>
</ref>
</ref-list>
</back>
</article>