<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1208695</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2023.1208695</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Ensemble learning-based approach for automatic classification of termite mushrooms</article-title>
<alt-title alt-title-type="left-running-head">Duong et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2023.1208695">10.3389/fgene.2023.1208695</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Duong</surname>
<given-names>Thi Kim Chi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2374642/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Tran</surname>
<given-names>Van Lang</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2141400/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Nguyen</surname>
<given-names>The Bao</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Nguyen</surname>
<given-names>Thi Thuy</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ho</surname>
<given-names>Ngoc Trung Kien</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2289488/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Nguyen</surname>
<given-names>Thanh Q.</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1492996/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Information Technology, Lac Hong University</institution>, <addr-line>Dong Nai Province</addr-line>, <country>Vietnam</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Faculty of Engineering and Technology, Thu Dau Mot University</institution>, <addr-line>Binh Duong Province</addr-line>, <country>Vietnam</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>HUFLIT Journal of Science, Ho Chi Minh City University of Foreign Languages and Information Technology</institution>, <addr-line>Ho Chi Minh City</addr-line>, <country>Vietnam</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Railway-Metro Engineering</institution>, <institution>Ho Chi Minh City University of Transport</institution>, <addr-line>Ho Chi Minh City</addr-line>, <country>Vietnam</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/905447/overview">Hiep Xuan Huynh</ext-link>, Can Tho University, Vietnam</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1101292/overview">Hasan Zulfiqar</ext-link>, University of Electronic Science and Technology of China, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2332781/overview">Van Hoa Nguyen</ext-link>, An Giang University, Vietnam</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2313058/overview">Minh Chon Nguyen</ext-link>, Can Tho University, Vietnam</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Van Lang Tran, <email>langtv@huflit.edu.vn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>11</day>
<month>10</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1208695</elocation-id>
<history>
<date date-type="received">
<day>19</day>
<month>04</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>13</day>
<month>09</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Duong, Tran, Nguyen, Nguyen, Ho and Nguyen.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Duong, Tran, Nguyen, Nguyen, Ho and Nguyen</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Termite mushrooms are edible fungi that provide significant economic, nutritional, and medicinal value. However, identifying these mushroom species based on morphology and traditional knowledge is ineffective due to their short development time and seasonal nature. This study proposes a novel method for classifying termite mushroom species. The method utilizes Gradient Boosting machine learning techniques and sequence encoding on the Internal Transcribed Spacer (ITS) gene dataset to construct a machine learning model for identifying termite mushroom species. The model is trained using ITS sequences obtained from the National Center for Biotechnology Information (NCBI) and the Barcode of Life Data Systems (BOLD). Ensemble learning techniques are applied to classify termite mushroom species. The proposed model achieves good results on the test dataset, with an accuracy of 0.91 and an average AUCROC value of 0.99. To validate the model, eight ITS sequences collected from termite mushroom samples in An Linh commune, Phu Giao district, Binh Duong province, Vietnam were used as the test data. The results show consistent species identification with predictions from the NCBI BLAST software. The results of species identification were consistent with the NCBI BLAST prediction software. This machine-learning model shows promise as an automatic solution for classifying termite mushroom species. It can help researchers better understand the local growth of these termite mushrooms and develop conservation plans for this rare and valuable plant resource.</p>
</abstract>
<kwd-group>
<kwd>ITS</kwd>
<kwd>molecular biology</kwd>
<kwd>DNA barcode</kwd>
<kwd>termite mushrooms</kwd>
<kwd>termite fungal taxonomy</kwd>
<kwd>ensemble learning</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Termitomyces mushrooms are a type of mushroom that nature has gifted us, known for their high nutritional value and delicious taste (<xref ref-type="bibr" rid="B21">Pegler, 1994</xref>). In addition to its high nutritional value, this termite mushroom is also known for its medicinal properties in many countries around the world. Termitomyces mushrooms have antibacterial properties, such as <italic>Termitomyces clypeatus</italic> against <italic>Pseudomonas aeruginosa</italic>, <italic>Termitomyces eurhizus</italic> against <italic>Proteus vulgaris</italic> and <italic>Scherichia coli</italic>, and <italic>Termitomyces microcarpus</italic> against <italic>Bacillus cereus</italic> and <italic>Proteus vulgaris</italic> (<xref ref-type="bibr" rid="B9">Giri, 2012</xref>). <italic>Termitomyces clypeatus</italic> also supports the treatment of chickenpox (<xref ref-type="bibr" rid="B5">Dutta and Acharya, 2014</xref>). The valuable compounds of these rare and valuable mushroom species are obtained through biomass cultivation (<xref ref-type="bibr" rid="B32">Lu et al., 2008</xref>) cultivated <italic>Termitomyces albuminosus</italic> to test its efficacy in pain reduction and anti-inflammation while <italic>Termitomyces striatus</italic> was used for other extracted compounds. <italic>Termitomyces heimii</italic> and <italic>Termitomyces microcarpus</italic> are used in the treatment of fever, colds, and fungal infections and in promoting cancer therapy (<xref ref-type="bibr" rid="B30">Venkatachalapathi and Paulsamy, 2016</xref>). There are about 30 species of Termitomyces mushrooms worldwide, and 10 species in Vietnam, with <italic>Termitomyces clypeatus</italic> and <italic>Termitomyces microcarpus</italic> being common in Binh Duong. Although very effective economically, the natural yield of these mushrooms is declining significantly, and they have not yet been cultivated sustainably, as they only grow seasonally.</p>
<p>Correctly identifying the name of a termite fungus species is an important task in biological research. Experts use traditional methods to classify and identify termite fungi based on their morphology. The overall structure of a termite fungus includes a cap, flesh, membrane, and stem, which may have rings and boxes (<xref ref-type="bibr" rid="B19">Mossebo et al., 2009</xref>). However, fungal structures vary from species to species, especially when mutations occur. Moreover, identifying samples lacking morphological characteristics can be difficult (<xref ref-type="bibr" rid="B25">Roe et al., 2010</xref>). A method for identifying new species of organisms that are often used to identify edible and medicinal mushrooms is based on molecular techniques. In this approach, molecular techniques such as DNA barcoding have been successfully used in recent years to identify species (<xref ref-type="bibr" rid="B11">Hebert et al., 2003</xref>; <xref ref-type="bibr" rid="B28">Somervuo et al., 2016</xref>). These molecular methods are based on analyzing genetic markers and have proven to be highly effective in identifying species, especially when combined with traditional morphological methods. Overall, incorporating molecular techniques into the identification process of termite fungi can provide more accurate and efficient identification, especially in cases where traditional morphological methods fall short.</p>
<p>One commonly utilized gene group in molecular identification is the group that encodes rRNA. This group is highly effective for finding similarities and differences when comparing different organisms due to the relatively conserved nature of most rRNA molecules (De Peer et al., 1996). For fungi, the rDNA ITS (Internal Transcribed Spacer) region, which includes two sequences, ITS1 and ITS2, flanking the 5.8S sequence, is widely accepted as the molecular region for species identification by most mycologists (<xref ref-type="bibr" rid="B16">K&#xf5;ljalg et al., 2013</xref>), as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. The ITS region is also used for predicting fungal species using machine-learning. This approach involves using the ITS sequence data to train a machine-learning model, which can then be used to accurately classify and identify different fungal species automatically. By combining molecular techniques such as machine-learning with traditional morphological identification methods, researchers can achieve more accurate and efficient identification of fungal species, aiding in both research and conservation efforts.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The ITS sequences region (<xref ref-type="bibr" rid="B31">White et al., 1990</xref>).</p>
</caption>
<graphic xlink:href="fgene-14-1208695-g001.tif"/>
</fig>
<p>The ITS sequence data for fungi can be accessed from two major datasets, BOLD (Barcode of Life Data) and the National Center for Biotechnology Information (NCBI). Both contain a vast collection of ITS sequences for all fungal species. Machine learning-based classification of fungal species using ITS sequences has been proposed by several researchers, including (<xref ref-type="bibr" rid="B26">Schloss et al., 2009</xref>; <xref ref-type="bibr" rid="B27">Schoch et al., 2012</xref>; <xref ref-type="bibr" rid="B3">Delgado-Serrano et al., 2016</xref>; <xref ref-type="bibr" rid="B4">Deshpande et al., 2016</xref>; <xref ref-type="bibr" rid="B6">Edgar, 2016</xref>; <xref ref-type="bibr" rid="B18">Meher et al., 2019</xref>; <xref ref-type="bibr" rid="B2">Das et al., 2023</xref>). A comprehensive list of the techniques and data used in fungal classification studies is provided in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Relevant works that used machine-learning based on ITS dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">References</th>
<th align="left">Tool</th>
<th align="left">No. of sequence per category</th>
<th align="left">The source of barcode sequences of fungal species</th>
<th align="left">Feature technical and ML algorithm</th>
<th align="left">Accuracy of the best model</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<xref ref-type="bibr" rid="B26">Schloss et al. (2009)</xref>
</td>
<td align="left">MOTHUR</td>
<td align="left">-</td>
<td align="left">The SILVA Database Project, Bremen, March 2009</td>
<td align="left">K-mer (k &#x3d; 5), The k-nearest neighbor (kNN) algorithm, and PGMA (unweighted-pair group method using average linkages) algorithms</td>
<td align="left">0.86</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B3">Delgado-Serrano et al. (2016)</xref>
</td>
<td align="left">Mycofar</td>
<td align="left">-</td>
<td align="left">NCBI GeneBank</td>
<td align="left">K-mer (k &#x3d; 5), Na&#xef;ve Bayes classifier</td>
<td align="left">0.87</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B4">Deshpande et al. (2016)</xref>
</td>
<td align="left">RDP</td>
<td align="left">10</td>
<td align="left">The Warcup dataset (18878 sequences belonging to 8551 species)</td>
<td align="left">K-mer (k &#x3d; 8) Bayesian regression.</td>
<td align="left">0.87</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B6">Edgar (2016)</xref>
</td>
<td align="left">SINTAX</td>
<td align="left">14</td>
<td align="left">RDP Warcup ITS (18878 sequences belonging to 8551 species)</td>
<td align="left">K-mer (k &#x3d; 8)Naive Bayesian Classifier</td>
<td align="left">0.87</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B18">Meher et al. (2019)</xref>
</td>
<td align="left">funbarRF</td>
<td align="left">10</td>
<td align="left">BOLD systems</td>
<td align="left">K-mer (k &#x3d; 4) Random Forest.</td>
<td align="left">0.89</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B2">Das et al. (2023)</xref>
</td>
<td align="left">CNN_FunBar</td>
<td align="left">20</td>
<td align="left">UNITE &#x2b; INSDC (4504529 sequences belonging to 44167 species)</td>
<td align="left">K-mer (k &#x3d; 6), CNN</td>
<td align="left">0.86</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The mentioned studies have successfully utilized supervised machine-learning techniques such as Naive Bayes classification, kNN, and Bayesian regression models for classifying fungal species. However, only (<xref ref-type="bibr" rid="B3">Delgado-Serrano et al., 2016</xref>) identified the fungal species at the genus level, while other studies only determined the species names. As ITS sequence data from the NCBI GeneBank were used, this data is not sufficient for identifying the labels of termite fungi found in these GenBank. For example, the ITS sequence of the termite fungus genus <italic>Termitomyces euripus</italic> in the NCBI GeneBank has only one sequence, while there are six labels for this fungal genus in BOLD. Additionally, the lengths of ITS sequences vary widely, ranging from 200 bases to 2000 bases, and the number of sequences between fungal genera varies greatly, from one to 500 sequences. Due to these limitations with ITS data for termite fungi, classical machine-learning algorithms struggle to accurately classify the labels of termite fungi. Our study focuses on identifying the labels of termite fungal genera using ITS sequence data collected from both the NCBI GeneBank and BOLD GenBank. The K-mer technique and natural language processing (NLP) were combined to extract features, and modern classification methods such as XGBoost (Extreme Gradient Boosting), Random Forest, and CatBoost are experimented with to build an automatic termite fungal species classifier. The proposed research is structured as follows: the method presents the concepts related to ITS sequence data, feature extraction techniques, the overall proposed model, experimental results, and finally, the study&#x2019;s conclusion.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>2 Methods</title>
<sec id="s2-1">
<title>2.1 ITS sequence data</title>
<p>Termite fungi are valuable but endangered, and urgent research and conservation efforts are needed. However, data on ITS sequences for termite fungi in GenBank are incomplete, making it crucial to synthesize data from different sources. In this article, ITS sequence data from two GenBank, BOLD and NCBI, was compiled by us. Specifically, 101 ITS sequences were obtained from BOLD, with the number of sequences for each genus ranging from 1 to 12. At NCBI, 1740 ITS sequences were obtained, with the number of sequences for each species ranging from 1 to 799. After synthesizing the ITS sequence data from these two GenBank and removing termite fungal species with fewer than 7 sequences, 1704 sequences belonging to 17 termite fungal species were obtained. The labels of each termite fungal species are presented in detail in <xref ref-type="table" rid="T2">Table 2</xref>. This data can be used for further research and conservation efforts for these valuable and endangered fungi.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Termitomyces species used for the training dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">No</th>
<th align="center">Termitomyces species label</th>
<th align="center">No. of sequences</th>
<th align="center">Lable</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="right">1</td>
<td align="left">
<italic>Uncultured Termitomyces</italic>
</td>
<td align="center">799</td>
<td align="center">16</td>
</tr>
<tr>
<td align="right">2</td>
<td align="left">
<italic>Termitomyces</italic> sp.</td>
<td align="center">483</td>
<td align="center">10</td>
</tr>
<tr>
<td align="right">3</td>
<td align="left">
<italic>Termitomyces intermedius</italic>
</td>
<td align="center">94</td>
<td align="center">8</td>
</tr>
<tr>
<td align="right">4</td>
<td align="left">
<italic>Termitomyces symbiont</italic>
</td>
<td align="center">60</td>
<td align="center">14</td>
</tr>
<tr>
<td align="right">5</td>
<td align="left">
<italic>Termitomyces microcarpus</italic>
</td>
<td align="center">34</td>
<td align="center">9</td>
</tr>
<tr>
<td align="right">6</td>
<td align="left">
<italic>Termitomyces clypeatus</italic>
</td>
<td align="center">33</td>
<td align="center">3</td>
</tr>
<tr>
<td align="right">7</td>
<td align="left">
<italic>Termitomyces cylindricus</italic>
</td>
<td align="center">30</td>
<td align="center">4</td>
</tr>
<tr>
<td align="right">8</td>
<td align="left">
<italic>Termitomyces striatus</italic>
</td>
<td align="center">29</td>
<td align="center">13</td>
</tr>
<tr>
<td align="right">9</td>
<td align="left">
<italic>Termitomyces DKA-2007</italic>
</td>
<td align="center">24</td>
<td align="center">0</td>
</tr>
<tr>
<td align="right">10</td>
<td align="left">
<italic>Termitomyces heimii</italic>
</td>
<td align="center">24</td>
<td align="center">7</td>
</tr>
<tr>
<td align="right">11</td>
<td align="left">
<italic>Termitomyces bulborhizus</italic>
</td>
<td align="center">17</td>
<td align="center">2</td>
</tr>
<tr>
<td align="right">12</td>
<td align="left">
<italic>Termitomyces fuliginosus</italic>
</td>
<td align="center">16</td>
<td align="center">6</td>
</tr>
<tr>
<td align="right">13</td>
<td align="left">
<italic>Termitomyces eurrhizus</italic>
</td>
<td align="center">15</td>
<td align="center">5</td>
</tr>
<tr>
<td align="right">14</td>
<td align="left">
<italic>Termitomyces albuminosus</italic>
</td>
<td align="center">14</td>
<td align="center">1</td>
</tr>
<tr>
<td align="right">15</td>
<td align="left">
<italic>Termitomyces</italic> sp. symbiont of <italic>Macrotermes bellicosus</italic>
</td>
<td align="center">12</td>
<td align="center">11</td>
</tr>
<tr>
<td align="right">16</td>
<td align="left">
<italic>Termitomyces</italic> sp. symbiont of <italic>Macrotermes subhyalinus</italic>
</td>
<td align="center">10</td>
<td align="center">12</td>
</tr>
<tr>
<td align="right">17</td>
<td align="left">Uncultured Ascomycota</td>
<td align="center">10</td>
<td align="center">15</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The ITS region of termite mushrooms collected from Binh Duong province, Vietnam, was sequenced, and the resulting sequences have a length ranging from 669 to 1050 base pairs. These termite mushroom samples have a morphology similar to that of <italic>Termitomyces clypeatus, Termitomyces microcarpus</italic> and <italic>Termitomyces striatus.</italic> The sequence data for these eight termite mushroom samples have been published and stored in the NCBI GeneBank. For more detailed information about these termite mushroom samples, please refer to <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Information of Termitomyces species in Binh Duong Province, Viet Nam.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">ID_sequences</th>
<th align="center">Binh Duong termitomyces species in NCBI</th>
<th align="center">Website</th>
<th align="center">Length of sequences</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">KU569480</td>
<td align="left">
<italic>Termitomyces clypeatus</italic>
</td>
<td align="left">
<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/search/all/?term=KU569480">https://www.ncbi.nlm.nih.gov/search/all/?term&#x3d;KU569480</ext-link>
</td>
<td align="left">980</td>
</tr>
<tr>
<td align="left">MF163136-BD5</td>
<td align="left">
<italic>Termitomyces clypeatus</italic>
</td>
<td align="left">
<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/search/all/?term=MF163136">https://www.ncbi.nlm.nih.gov/search/all/?term&#x3d;MF163136</ext-link>
</td>
<td align="left">720</td>
</tr>
<tr>
<td align="left">MF163152.1</td>
<td align="left">
<italic>Termitomyces clypeatus</italic>
</td>
<td align="left">
<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/search/all/?term=MF163152">https://www.ncbi.nlm.nih.gov/search/all/?term&#x3d;MF163152</ext-link>
</td>
<td align="left">938</td>
</tr>
<tr>
<td align="left">MF163445-BD3</td>
<td align="left">
<italic>Termitomyces</italic> sp.</td>
<td align="left">
<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/search/all/?term=MF163445">https://www.ncbi.nlm.nih.gov/search/all/?term&#x3d;MF163445</ext-link>
</td>
<td align="left">669</td>
</tr>
<tr>
<td align="left">MF163446-BD6</td>
<td align="left">
<italic>Termitomyces</italic> sp.</td>
<td align="left">
<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/search/all/?term=MF163446">https://www.ncbi.nlm.nih.gov/search/all/?term&#x3d;MF163446</ext-link>
</td>
<td align="left">1020</td>
</tr>
<tr>
<td align="left">MT672480.1</td>
<td align="left">
<italic>Termitomyces microcarpus</italic>
</td>
<td align="left">
<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/search/all/?term=MT672480.1">https://www.ncbi.nlm.nih.gov/search/all/?term&#x3d;MT672480.1</ext-link>
</td>
<td align="left">721</td>
</tr>
<tr>
<td align="left">MT730584.1</td>
<td align="left">
<italic>Termitomyces clypeatus</italic>
</td>
<td align="left">
<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/search/all/?term=MT730584">https://www.ncbi.nlm.nih.gov/search/all/?term&#x3d;MT730584</ext-link>
</td>
<td align="left">608</td>
</tr>
<tr>
<td align="left">MF163149-BD4</td>
<td align="left">
<italic>Termitomyces</italic> sp.</td>
<td align="left">
<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/search/all/?term=MF163149">https://www.ncbi.nlm.nih.gov/search/all/?term&#x3d;MF163149</ext-link>
</td>
<td align="left">812</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-2">
<title>2.2 Feature generation</title>
<p>The extraction of features from biological sequences is a crucial step in computational biology. Biological sequences are typically composed of a string of letters, which must be converted into numerical vectors before they can be utilized in machine-learning algorithms (<xref ref-type="bibr" rid="B14">Kamath et al., 2014</xref>). The K-mer feature technique has been employed to represent information for ITS sequences to classify species based on barcodes, as demonstrated by previous studies (<xref ref-type="bibr" rid="B26">Schloss et al., 2009</xref>; <xref ref-type="bibr" rid="B4">Deshpande et al., 2016</xref>). In 2016, Delgado-Serrano utilized K-mer encodings to transform ITS sequences into numerical vectors. The accuracy of the prediction model was affected by the size of the K-mer utilized (<xref ref-type="bibr" rid="B3">Delgado-Serrano et al., 2016</xref>). In our proposed approach, a combination of K-mer and CountVectorizer techniques was employed to encode ITS sequences into numerical vectors. An illustration of the methodology utilized to digitize sequence information is presented in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Illustrate the use of the K-mer method to encode ITS sequences into numeric vectors, in the example the size of K-mer was 7.</p>
</caption>
<graphic xlink:href="fgene-14-1208695-g002.tif"/>
</fig>
<p>In <xref ref-type="fig" rid="F2">Figure 2</xref>, The process of digitizing ITS sequences has been illustrated. This process is similar to that of using Natural Language Processing (NLP) tools from Sklearn to convert our K-mer words into numerical vectors. These vectors, which represent the count of each K-mer in the vocabulary, have the same length as unigrams.</p>
</sec>
<sec id="s2-3">
<title>2.3 Ensemble learning</title>
<p>Supervised machine-learning techniques are widely used in computational biology to solve various problems. Several traditional machine-learning algorithms such as k-nearest neighbors, Na&#xef;ve Bayes, and decision trees have been successful in identifying mushroom species based on barcode data (<xref ref-type="bibr" rid="B26">Schloss et al., 2009</xref>; <xref ref-type="bibr" rid="B3">Delgado-Serrano et al., 2016</xref>; <xref ref-type="bibr" rid="B4">Deshpande et al., 2016</xref>). However, these models have relatively low accuracy. In our research, two solutions were tested: i) The first set utilized well-known classification methods like Na&#xef;ve Bayes and Random forest to predict the names of termite mushroom species; ii) In the second, automated models for predicting termite mushroom species with higher accuracy were built by us using Ensemble learning algorithms such as XGBoost and CatBoost.</p>
</sec>
<sec id="s2-4">
<title>2.4 Gradient-boosted decision trees (GBDTs)</title>
<p>Gradient Boosting Decision Trees (GBDT) (<xref ref-type="bibr" rid="B8">Friedman, 2001</xref>) is a method that uses decision tree ensembles to predict target values. A GBDT is constructed by splitting observations based on the attribute values of the input data. The model can find the best way to divide data and determine the most time-consuming part of the partitioning process. To build a GBDT model with T trees from a dataset consisting of n samples, the prediction process according to the GBDT method is as follows:<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mo>.</mml:mo>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>K</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the predicted value of the <italic>i</italic>
<sup>
<italic>th</italic>
</sup> sample at the <italic>k</italic>
<sup>
<italic>th</italic>
</sup> iteration</p>
<p>The cost function of GBDT has two parts: a training error and regularization, as follows:<disp-formula id="e2">
<mml:math id="m3">
<mml:mrow>
<mml:mtext>Cost</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x302;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mi>&#x3a9;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf2">
<mml:math id="m4">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>T</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mi>&#x3bb;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2200;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mover accent="true">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf3">
<mml:math id="m5">
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of leaf nodes, <inline-formula id="inf4">
<mml:math id="m6">
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the score for a leaf node, <inline-formula id="inf5">
<mml:math id="m7">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the leaf penalty coefficient, and ensures that leaf nodes&#x2019; scores are not too large.</p>
<sec id="s2-4-1">
<title>2.4.1 CatBoost</title>
<p>CatBoost is an algorithm used to boost gradients on decision trees. It is used to process datasets with a large number of input features of the categorical data type (<xref ref-type="bibr" rid="B22">Prokhorenkova et al., 2018</xref>). In the field of computational biology, CatBoost has been applied for various purposes such as identifying bacterial genes at the 16S rRNA level (<xref ref-type="bibr" rid="B17">Meharunnisa and Sornam, 2022</xref>) or building a feature extraction package for DNA, RNA, and protein sequences (<xref ref-type="bibr" rid="B24">Robson, 2022</xref>). Our proposals have used the CatBoost algorithm to build a model for termite mushroom species classification.</p>
</sec>
<sec id="s2-4-2">
<title>2.4.2 XGBoost</title>
<p>XGBoost is a powerful machine-learning algorithm that builds upon the initial gradient-boosting machine (<xref ref-type="bibr" rid="B8">Friedman, 2001</xref>; <xref ref-type="bibr" rid="B1">Chen et all., 2015</xref>), is an upgraded version of gradient boosting that boasts many superior improvements (<xref ref-type="bibr" rid="B23">Ren et al., 2017</xref>; <xref ref-type="bibr" rid="B13">Jiang et al., 2019</xref>); (Zhong et al., 2018). These improvements, achieved through parallel computation on different datasets, have significantly increased processing speed, making XGBoost up to 10 times faster than GBM. XGBoost has been successfully applied in many fields, including computational biology.</p>
</sec>
</sec>
<sec id="s2-5">
<title>2.5 Building the best classifier base on ensemble learning</title>
<p>CatBoost is a viable option for gene sequence data analysis, as indicated by recent research (<xref ref-type="bibr" rid="B24">Robson, 2022</xref>). In our experiments with termite mushroom data, it was observed that CatBoost performed comparably to XGBoost in terms of prediction accuracy. However, a relatively longer training time is required by CatBoost than that of XGBoost to achieve a similar level of performance. Therefore, XGBoost was chosen as the primary algorithm for our prediction model.</p>
<p>The XGBoost model&#x2019;s performance depends on several key parameters such as <italic>&#x27;max_depth</italic>&#x27;, &#x27;<italic>gamma</italic>&#x27;, <italic>&#x27;n_estimators&#x27;</italic>, and &#x27;<italic>learning_rate&#x27;</italic>. These parameters are known as hyperparameters and can be adjusted manually during training or automatically. The proposed enhanced model uses the Bayesian Optimization technique (<xref ref-type="bibr" rid="B15">Klein et al., 2017</xref>) specifically Random search, to tune the hyperparameters. Bayesian Optimization was applied to tune the four main parameters of the XGBoost classifier: &#x27;max_depth&#x27;, &#x27;gamma&#x27;, &#x27;n_estimators&#x27;, and &#x27;learning_rate&#x27;.</p>
<p>To improve the predictive performance of the model, cross-validation with k &#x3d; 5 was performed to select the best classification model, in addition to using Bayesian Optimization to tune hyperparameters. A new dataset, which consisted of n data samples and m features, was obtained from the results of phase 1. An optimization parameter was then used as input for <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref> to build an optimal classification model.</p>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1. Building the best XGBoost classifier</label>
<p>
<list list-type="simple">
<list-item>
<p>
<bold>Input:</bold> <inline-formula id="inf6">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mi>m</mml:mi>
</mml:msup>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mn>0,1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mo>&#x2200;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mover accent="true">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>; hyperparameter is <italic>&#x19f;</italic> &#x3d; {&#x27;max_depth&#x27;: int(max_depth), &#x27;gamma&#x27;: Gama, &#x27;n_estimators&#x27;: int(n_estimators), &#x27;learning_rate&#x27;:learning_rate }</p>
</list-item>
<list-item>
<p>
<bold>Output:</bold> Best_Model</p>
</list-item>
<list-item>
<p>Begin</p>
</list-item>
<list-item>
<p>
<bold>&#x2003;</bold>1: <bold>Initialize</bold>: FeatureImportances&#x3d;{}</p>
</list-item>
<list-item>
<p>
<bold>&#x2003;</bold>2: Model &#x2190; <bold>XGBoostClassifier (</bold>
<italic>&#x19f;</italic>
<bold>)</bold>
</p>
</list-item>
<list-item>
<p>
<bold>&#x2003;</bold>3: KFold&#x2190; StratifiedKFold (n_splits&#x3d;5, shuffle &#x3d; True, random_state&#x3d;2020)</p>
</list-item>
<list-item>
<p>
<bold>&#x2003;</bold>4: <bold>For</bold> i&#x3d;1 <bold>each</bold> KFold</p>
</list-item>
<list-item>
<p>
<bold>&#x2003;&#x2003;</bold>&#x2022; Divide the <inline-formula id="inf7">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> dataset into <inline-formula id="inf8">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf9">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<bold>&#x2003;&#x2003;</bold>&#x2022; Train the model based on early-ending hyperparameters</p>
</list-item>
<list-item>
<p>
<bold>&#x2003;</bold>5: <bold>Calculate</bold> the roc_auc_score, accuracy_score, precision_score, recall_score, and f1_score over iterations</p>
</list-item>
<list-item>
<p>
<bold>&#x2003;</bold>6: <bold>Select</bold> the best model based on Step 4</p>
</list-item>
<list-item>
<p>
<bold>&#x2003;</bold>7: <bold>Visualize</bold> the mean value from Step 4</p>
</list-item>
<list-item>
<p>
<bold>&#x2003;</bold>8: <bold>Return</bold> the Best_Model from Step 4</p>
</list-item>
<list-item>
<p>End</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
<sec id="s2-6">
<title>2.6 Building a model for predicting the termite fungus species name</title>
<p>Our study has developed an automated process consisting of four stages to predict the species name of a new termite fungus. The first stage involves collecting termite fungus data from ITS gene sequence repositories. In the second stage, sequence features are extracted and encoded. The third stage involves building a classifier by constructing and tuning parameters to find the optimal classifier. Finally, in the fourth stage, the classifier is used to predict new termite fungus samples. <xref ref-type="fig" rid="F3">Figure 3</xref> provides a detailed description of this process.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Detailed model of the proposed method. <bold>(A)</bold> Collected Data: The study collected a total of 1796 ITS sequences of mushroom fungus from GenBank NCBI and BOLD. After filtering out termite fungus species sequences with less than 10 sequences, the final count of ITS sequences was 1704. <bold>(B)</bold> Data Preprocessing: The ITS sequences were split into smaller sequences, following the rules described in <xref ref-type="fig" rid="F2">Figure 2</xref>, using K-mer with a size of 7. The longest ITS sequence was 2470 bases, corresponding to a vector length of 14425 when encoded. <bold>(C)</bold>. Training: The training process used an 80:20 split ratio and employed hyperparameter optimization for the training model. The model was optimized using the k-fold Cross-Validation technique with k &#x3d; 5, and BayesianOptimization was performed to fine-tune the following parameters: <italic>&#x27;max_depth&#x27;</italic>: (5,10), <italic>&#x27;gamma&#x27;</italic>: (0,1), <italic>&#x27;learning_rate&#x27;</italic>:(0,1<italic>), &#x27;n_estimators&#x27;</italic>:(100,400). The model with the highest accuracy was selected for the classification. <bold>(D)</bold> Prediction: Mushroom samples collected in Binh Duong Province, Vietnam, and downloaded from NCBI were used as the test set. These samples were subjected to K-mer with a size of 7 and then CountVectorizer was applied. Finally, the best model from stage c was applied to predict the species of new termite fungi.</p>
</caption>
<graphic xlink:href="fgene-14-1208695-g003.tif"/>
</fig>
</sec>
<sec id="s2-7">
<title>2.7 Performance metrics</title>
<p>In our study, the terms &#x201c;<italic>true</italic>&#x201d; and &#x201c;<italic>false</italic>&#x201d; predictions can arise from the model&#x2019;s misclassification or failure to predict accurately, such as false negatives or false positives, or other concepts applied to the prediction targets. Specifically, the phrase &#x201c;<italic>predicting the species of Termitomyces</italic>&#x201d; is referred to as a true positive (TP), while the phrase &#x201c;<italic>correctly excluding the species of Termitomyces</italic>&#x201d; is referred to as a true negative (TN). On the other hand, the phrase &#x201c;<italic>predicting the species of Termitomyces incorrectly</italic>&#x201d; is designated as a false positive (FP), and a &#x201c;<italic>missed or misclassified prediction</italic>&#x201d; is considered a false negative (FN). These conditions are utilized as stopping points during initial data training. To evaluate the performance of our proposed model, various methods were applied to assess its machine-learning abilities on DNA sequence data (<xref ref-type="bibr" rid="B10">Gupta, P., et al., 2021</xref>). These methods include the following:<list list-type="simple">
<list-item>
<p>&#x2756; Accuracy: The proportion of correctly predicted cases is known as accuracy, and it can be calculated using the following formula:</p>
</list-item>
</list>
<disp-formula id="equ1">
<mml:math id="m12">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<list list-type="simple">
<list-item>
<p>&#x2756; Sensitivity: Recall (pr) was the hit rate (hit rate), and the true positive rate (TPR) was the ratio of correct positive classifications to the total number of positive and recall cases and it can be calculated using the following formula:</p>
</list-item>
</list>
<disp-formula id="equ2">
<mml:math id="m13">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<list list-type="simple">
<list-item>
<p>&#x2756; Specificity: True negative (TN) (or specificity in clinical medicine) was the correct exclusion rate out of the total number of negative cases, it can be calculated using the following formula:</p>
</list-item>
</list>
<disp-formula id="equ3">
<mml:math id="m14">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<list list-type="simple">
<list-item>
<p>&#x2756; False Positive Rate/Fallout (FPR) was an expression of the rate of mislabeling of negative to positive samples across all negative samples, it was calculated by the following formula:</p>
</list-item>
</list>
<disp-formula id="equ4">
<mml:math id="m15">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>specificity</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<list list-type="simple">
<list-item>
<p>&#x2756; Precision: Since the dataset had a larger sample, this led to an imbalanced input dataset for the prediction model. Therefore, we used precision to determine the ratio of actually positive cases to the total number of cases labeled &#x201c;positive&#x201d; by the model. Precision is a term that refers to the &#x201c;deterministic&#x201d; or accurate positive classification of a model:</p>
</list-item>
</list>
<disp-formula id="equ5">
<mml:math id="m16">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<list list-type="simple">
<list-item>
<p>&#x2756; F1 score: This was defined as the harmonic mean between precision and recall:</p>
</list-item>
</list>
<disp-formula id="equ6">
<mml:math id="m17">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mfrac>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<list list-type="simple">
<list-item>
<p>&#x2756; Receiver operating characteristics (ROCs) were used to calculate the model&#x2019;s classification performance in the condition of unbalanced data set classes. A ROC curve was produced for each pair (TPR, FPR) for different thresholds, with each point on the curve representing one pair (TPR, FPR) for one threshold. This curve shows us the relationship between the True Positive Rate (TPR) and the False Positive Rate (FPR). The ROC Curve and the ROC AUC score are important tools for evaluating binary classification models. To evaluate multi-class classifiers, the OvR (One vs. Rest) technique was used, which compares each class with all other classes simultaneously. In this case, one class was chosen to be the &#x201c;positive&#x201d; class, while all other classes (the remaining part) were considered &#x201c;negative&#x201d; classes. In the experiment, the last label class 16 was selected as the &#x201c;positive&#x201d; class and the remaining classes were considered &#x201c;negative&#x201d;. In this way, the multi-class classification output was reduced into binary classification, allowing the utilization of all known binary classification metrics to evaluate the classification model.</p>
</list-item>
</list>
</p>
</sec>
</sec>
<sec sec-type="results|discussion" id="s3">
<title>3 Results and discussion</title>
<sec id="s3-1">
<title>3.1 Result of each stage in the proposed process</title>
<p>In the experimental process, Python 3.9 and the libraries Scikit-learn, Biopython, XGBoost, CatBoost, and Bayesian optimization were employed to construct a mushroom classification model following the proposed process depicted in <xref ref-type="fig" rid="F3">Figure 3</xref>. The results of each stage a, b, c, and d are attached.<list list-type="simple">
<list-item>
<p>&#x2756; During stage a: Data was collected through the following steps: (a.1) retrieving data from the NCBI and BOLD GenBanks, which yielded 1740 sequences of 28 mushroom species; (a.2) selecting 17 species that had at least 10 sequences per species.</p>
</list-item>
<list-item>
<p>&#x2756; During stage b: Data preprocessing was performed in two steps: The ITS sequence strings were separated by applying K-mer with a length of <italic>k</italic> &#x3d; 7, and then the ITS sequences were converted into numerical data by vectorizing them, and the data labels were also converted into numerical values The section provides details on the number of classes and corresponding data.</p>
</list-item>
<list-item>
<p>&#x2756; During stage c: The best prediction model was built, consisting of (c.1) a classification model and (c.2) an optimized set of hyperparameters.</p>
</list-item>
<list-item>
<p>&#x2756; Finally, during stage d: The performance of the proposed model was displayed in step (d.1), while the predictions of eight ITS sequences collected in Thu Dau Mot, Binh Duong province were shown in step (d.2).</p>
</list-item>
</list>
</p>
</sec>
<sec id="s3-2">
<title>3.2 Select the appropriate K-mer sizes for the classifiers</title>
<p>The accuracy of predictive models based on sequence data is significantly impacted by the size of K-mers (<xref ref-type="bibr" rid="B3">Delgado-Serrano et al., 2016</xref>). To explore this impact, a study was conducted using different K-mer lengths, which resulted in varying classification accuracies. The sequence in <xref ref-type="fig" rid="F3">Figure 3</xref> was used to build a classifier, with machine-learning algorithms such as Naive Bayes (MultinomialNB), RandomForest, XGBboost, and Catboost. The classifier&#x2019;s results for each K-mer size are presented in <xref ref-type="fig" rid="F4">Figure 4</xref>. We found that each algorithm produced different predictive results (accuracy) for each K-mer size. Specifically, Catboost produced results ranging from 0.87&#x2013;0.88, XGBboost had results from 0.88&#x2013;0.91, and RandomForest yielded results from 0.87&#x2013;0.89. However, the Naive Bayes (MultinomialNB) model had the lowest accuracy, ranging from 0.59&#x2013;0.61. The classifier&#x2019;s results for each K-mer size are presented in <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Accuracy of machine-learning algorithms according to the K-mer sizes. It was found that different predictive results (accuracy) were produced by each algorithm for each K-mer size such as: Catboost produced results ranging from 0.87&#x2013;0.88, XGBboost had results from 0.88&#x2013;0.91, and RandomForest yielded results from 0.87&#x2013;0.89. However, the Naive Bayes (MultinomialNB) model had the lowest accuracy, ranging from 0.59&#x2013;0.61. <xref ref-type="fig" rid="F5">Figure 5</xref> presents detailed information on the impact of K-mer size on the ROC Curve (AUCROC) achieved by each algorithm. Notably, the XGBoost algorithm achieved the highest classification AUCROC when the K-mer size was set to 7, which was also used to build the automated model for predicting mushroom species names.</p>
</caption>
<graphic xlink:href="fgene-14-1208695-g004.tif"/>
</fig>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Accuracy of machine-learning algorithms according to the K-mer sizes.</p>
</caption>
<graphic xlink:href="fgene-14-1208695-g005.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F6">Figure 6</xref> presents detailed information on the impact of K-mer size on the highest accuracy achieved by each algorithm. Notably, the XGBoost algorithm achieved the highest classification accuracy when the K-mer size was set to 7, which was also used to build the automated model for predicting mushroom species names.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>The ROC curve using the OvR macro-average for each class in the XGBoost method by size <italic>K-mer</italic> &#x3d; 7.</p>
</caption>
<graphic xlink:href="fgene-14-1208695-g006.tif"/>
</fig>
</sec>
<sec id="s3-3">
<title>3.3 Performance analysis in other machine-learning algorithms</title>
<p>Apart from using accuracy as a measure of the classification model&#x2019;s performance, other metrics such as precision, recall, F1 score, or AUCROC are also utilized to evaluate the classifiers&#x2019; performance. A summary of the performance of the surveyed machine-learning algorithms is presented in <xref ref-type="table" rid="T4">Table 4</xref>.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Synthesize the performance of machine-learning algorithms.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Method</th>
<th align="center">AUROC</th>
<th align="center">Accuracy</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1 score</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Naive Bayes</td>
<td align="center">0.93</td>
<td align="center">0.60</td>
<td align="center">0.84</td>
<td align="center">0.60</td>
<td align="center">0.62</td>
</tr>
<tr>
<td align="left">RandomForest</td>
<td align="center">0.98</td>
<td align="center">0.88</td>
<td align="center">0.88</td>
<td align="center">0.88</td>
<td align="center">0.88</td>
</tr>
<tr>
<td align="left">XGBboost</td>
<td align="center">0.99</td>
<td align="center">0.91</td>
<td align="center">0.90</td>
<td align="center">0.91</td>
<td align="center">0.90</td>
</tr>
<tr>
<td align="left">Catboost</td>
<td align="center">0.99</td>
<td align="center">0.87</td>
<td align="center">0.87</td>
<td align="center">0.87</td>
<td align="center">0.87</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Furthermore, the AUCROC for each class was calculated using the ROC curve method with the OvR macro-average for the multi-class model utilized (<xref ref-type="bibr" rid="B20">Pedregosa et al., 2011</xref>). In this study, the last class (class 16) was designated as the positive class, while all other classes were considered negative classes. The visual representations of each class&#x2019;s results are presented in <xref ref-type="fig" rid="F6">Figure 6</xref>.</p>
</sec>
<sec id="s3-4">
<title>3.4 Comparative analysis for prediction of fungal species</title>
<p>Previous models for predicting fungal species accuracy have been evaluated using the <italic>K-mer</italic> method and machine-learning techniques such as <italic>k-Nearest</italic> Neighbor, Na&#xef;ve Bayes, and Random Forest, with results presented in <xref ref-type="table" rid="T5">Table 5</xref>. Our proposed approach demonstrates superior performance when utilizing a <italic>K-mer</italic> size of 7 with the XGBoost classification algorithm. <xref ref-type="table" rid="T5">Table 5</xref> presents a comparison of various classifiers&#x27; performance for predicting fungal species using ITS sequences.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Performance comparison of fungal classifiers using ITS sequencing.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Ref</th>
<th align="center">Method</th>
<th align="center">Accuracy</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<xref ref-type="bibr" rid="B26">Schloss et al. (2009)</xref>
</td>
<td align="left">K-mer (k &#x3d; 5), The k-nearest neighbor (kNN) algorithm, and PGMA algorithms</td>
<td align="center">0.86</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B3">Delgado-Serrano et al. (2016)</xref>
</td>
<td align="left">K-mer (k &#x3d; 5), Na&#xef;ve Bayes classifier model</td>
<td align="center">0.87</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B4">Deshpande et al. (2016)</xref>
</td>
<td align="left">K-mer (k &#x3d; 8) Bayesian regression.</td>
<td align="center">0.87</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B18">Meher et al. (2019)</xref>
</td>
<td align="left">K-mer (k &#x3d; 4), Random Forest.</td>
<td align="center">0.89</td>
</tr>
<tr>
<td align="left">Our proposal</td>
<td align="left">K-mer (k &#x3d; 7), XGBoost</td>
<td align="center">0.91</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-5">
<title>3.5 Compare the prediction results of the proposed model with the results of BLAST</title>
<p>The ITS sequences of termite fungi collected from Binh Duong province, Vietnam, were published on NCBI and are detailed in <xref ref-type="table" rid="T3">Table 3</xref>. Our proposed classification model predicted species identification with comparable results to those obtained from NCBI. For instance, sequences MF163150-BD1, MF163151-BD2, and MF163147-BD7 were identified as the same species as those on NCBI. Moreover, the species identification of MF163149-BD4 was consistent with the identification on NCBI. However, for MF163445-BD3, MF163446-BD6, and MF163149-BD4, the identification was previously unknown or unclear. Our proposed classification model successfully identified MF163445-BD3 and MF163446-BD6 as <italic>Termitomyces striatus</italic>, consistent with the type strain of the collected fungi. The results for MF163149-BD4 were also consistent with the species identification on NCBI. <xref ref-type="table" rid="T6">Table 6</xref> presents the details of the species identification results.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Result in comparison of the species identification of ITS sequences of termite fungi collected in Binh Duong province, Vietnam, with the identification on NCBI.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">ID_sequences</th>
<th align="center">Binh Duong termitomyces species in NCBI</th>
<th align="center">Binh Duong termitomyces species in our proposal</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">KU569480</td>
<td align="left">
<italic>Termitomyces clypeatus</italic>
</td>
<td align="left">
<italic>Termitomyces clypeatus</italic>
</td>
</tr>
<tr>
<td align="left">MF163136-BD5</td>
<td align="left">
<italic>Termitomyces clypeatus</italic>
</td>
<td align="left">
<italic>Termitomyces clypeatus</italic>
</td>
</tr>
<tr>
<td align="left">MF163152.1</td>
<td align="left">
<italic>Termitomyces clypeatus</italic>
</td>
<td align="left">
<italic>Termitomyces clypeatus</italic>
</td>
</tr>
<tr>
<td align="left">MF163445-BD3</td>
<td align="left">
<italic>Termitomyces</italic> sp.</td>
<td align="left">
<italic>Termitomyces striatus</italic>
</td>
</tr>
<tr>
<td align="left">MF163446-BD6</td>
<td align="left">
<italic>Termitomyces</italic> sp.</td>
<td align="left">
<italic>Termitomyces striatus</italic>
</td>
</tr>
<tr>
<td align="left">MT672480.1</td>
<td align="left">
<italic>Termitomyces microcarpus</italic>
</td>
<td align="left">
<italic>Termitomyces microcarpus</italic>
</td>
</tr>
<tr>
<td align="left">MT730584.1</td>
<td align="left">
<italic>Termitomyces clypeatus</italic>
</td>
<td align="left">
<italic>Termitomyces clypeatus</italic>
</td>
</tr>
<tr>
<td align="left">MF163149-BD4</td>
<td align="left">
<italic>Termitomyces</italic> sp.</td>
<td align="left">
<italic>Termitomyces</italic> sp.</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Accurately identifying new species is crucial for studying biodiversity and formulating conservation policies for endangered species (<xref ref-type="bibr" rid="B29">Van Velzen et al., 2012</xref>). Traditional methods of species identification based on physical characteristics can be difficult, prompting the use of DNA barcoding as an alternative approach (<xref ref-type="bibr" rid="B12">Hibbett et al., 2011</xref>). In this study, a novel computational method is proposed that utilizes K-mer techniques and NLP vectorization to convert DNA barcode sequence data into digital features. The XGBoost algorithm is then employed to build a model capable of predicting termite mushroom species using the ITS sequence as a DNA barcode.</p>
<p>The performance of the developed model was evaluated on 1704 sequences of 17 mushroom species obtained from two ITS GenBanks. The evaluation was conducted using standard classification metrics such as accuracy, precision, recall, F1-score, and AUCROC.</p>
<p>Our proposed model was assessed by comparing its predictions with the species identification results on NCBI, demonstrating complete consistency with the identified species of the ITS sequences of mushrooms, as well as predicting the species names of two sequences that had not previously been identified. An example of this is the <italic>Termitomyces striatus</italic> mushroom specimen found in Binh Duong province, Vietnam, which was correctly identified by our proposed model. Furthermore, when compared to four other research groups&#x27; machine-learning models for predicting termite mushroom species names, our proposed model achieved an accuracy of 0.91 and an average AUCROC score of 0.99, demonstrating its efficacy in species identification. These results suggest that our proposed model is a valuable tool for identifying termite fungi species in Binh Duong province, Vietnam, and could be applied to other mushroom species as well.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>This study presents a computational model to predict termite fungus species based on DNA barcodes. The paper also introduces a new method for creating features based on K-mer techniques, NLP vectorization to digitize sequence data, and an optimized classifier. The results showed that the model was evaluated based on the standard classification systems&#x2019; measures, including accuracy, precision, recall, <italic>f1</italic>-score, and AUCROC. The model was evaluated on 17 termite mushroom species and achieved high accuracy when compared with species identification results on NCBI. These results suggest that the proposed model can be an effective tool for identifying termite mushroom species based on DNA barcodes. Furthermore, the proposed method can also be used to predict other species.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>TD: study conception and design; TBN, TTN, and NH: data collection, analysis and interpretation of results; VT: draft manuscript, preparation; TQN: draft manuscript. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec sec-type="COI-statement" id="s7">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s8">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Benesty</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Khotilovich</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cho</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Xgboost: extreme gradient boosting</article-title>. <comment>R. package version 0.4-2</comment> <volume>1</volume> (<issue>4</issue>), <fpage>1</fpage>&#x2013;<lpage>4</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Das</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Rai</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mishra</surname>
<given-names>D. C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>CNN_FunBar: advanced learning technique for fungi ITS region classification</article-title>. <source>Genes.</source> <volume>14</volume> (<issue>3</issue>), <fpage>634</fpage>. <pub-id pub-id-type="doi">10.3390/genes14030634</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Delgado-Serrano</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Restrepo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bustos</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Zambrano</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Anzola</surname>
<given-names>J. M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Mycofier: A new machine learning-based classifier for fungal ITS sequences</article-title>. <source>BMC Res. Notes</source> <volume>9</volume> (<issue>1</issue>), <fpage>402</fpage>&#x2013;<lpage>408</lpage>. <pub-id pub-id-type="doi">10.1186/s13104-016-2203-3</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deshpande</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Greenfield</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Charleston</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Porras-Alfaro</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kuske</surname>
<given-names>C. R.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Fungal identification using a Bayesian classifier and the Warcup training set of internal transcribed spacer sequences</article-title>. <source>Mycologia</source> <volume>108</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.3852/14-293</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dutta</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Acharya</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Traditional and ethno-medicinal knowledge of mushrooms in West Bengal, India</article-title>. <source>Asian J. Pharm. Clin. Res.</source> <volume>7</volume> (<issue>4</issue>), <fpage>36</fpage>&#x2013;<lpage>41</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Edgar</surname>
<given-names>R. C.</given-names>
</name>
</person-group> (<year>2016</year>). <source>Sintax: A simple non-bayesian taxonomy classifier for 16S and ITS sequences</source>. <publisher-name>biorxiv</publisher-name>.<fpage>074161</fpage>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Friedman</surname>
<given-names>J. H.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Greedy function approximation: A gradient boosting machine</article-title>. <source>Ann. statistics</source> <volume>29</volume>, <fpage>1189</fpage>&#x2013;<lpage>1232</lpage>. <pub-id pub-id-type="doi">10.1214/aos/1013203451</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Giri</surname>
<given-names>S. B.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Antimicrobial activities of basidiocarps of wild edible mushrooms of West Bengal, India</article-title>. <source>Int. J. PharmTech Res.</source> <volume>4</volume> (<issue>4</issue>), <fpage>1554</fpage>&#x2013;<lpage>1560</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gupta</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Melkani</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Maggu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rathee</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Genome sequencing and classifier</article-title>. <source>Int. J. Adv. Eng. Manag.</source> <volume>4</volume> (<issue>4</issue>), <fpage>1554</fpage>&#x2013;<lpage>1560</lpage>. <pub-id pub-id-type="doi">10.35629/5252-030617591767</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hebert</surname>
<given-names>P. D.</given-names>
</name>
<name>
<surname>Cywinska</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ball</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>DeWaard</surname>
<given-names>J. R.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Biological identifications through DNA barcodes</article-title>. <source>Proc. R. Soc. Lond. Ser. B Biol. Sci.</source> <volume>270</volume> (<issue>1512</issue>), <fpage>313</fpage>&#x2013;<lpage>321</lpage>. <pub-id pub-id-type="doi">10.1098/rspb.2002.2218</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hibbett</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Ohman</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Glotzer</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Nuhn</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kirk</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Nilsson</surname>
<given-names>R. H.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Progress in molecular and morphological taxon discovery in Fungi and options for formal classification of environmental sequences</article-title>. <source>Fungal Biol. Rev.</source> <volume>25</volume> (<issue>1</issue>), <fpage>38</fpage>&#x2013;<lpage>47</lpage>. <pub-id pub-id-type="doi">10.1016/j.fbr.2011.01.001</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tong</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A pedestrian detection method based on genetic algorithm for optimize XGBoost training parameters</article-title>. <source>IEEE Access</source> <volume>7</volume>, <fpage>118310</fpage>&#x2013;<lpage>118321</lpage>. <pub-id pub-id-type="doi">10.1109/access.2019.2936454</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kamath</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>De Jong</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Shehu</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Effective automated feature construction and selection for classification of biological sequences</article-title>. <source>PloS one</source> <volume>9</volume> (<issue>7</issue>), <fpage>e99982</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0099982</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Klein</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Falkner</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bartels</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hennig</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Hutter</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Fast bayesian optimization of machine learning hyperparameters on large datasets</article-title>,&#x201d; in <source>Artificial intelligence and statistics</source> (<publisher-name>PMLR</publisher-name>), <fpage>528</fpage>&#x2013;<lpage>536</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>K&#xf5;ljalg</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Nilsson</surname>
<given-names>R. H.</given-names>
</name>
<name>
<surname>Abarenkov</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Tedersoo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Taylor</surname>
<given-names>A. F.</given-names>
</name>
<name>
<surname>Bahram</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Towards a unified paradigm for sequence&#x2010;based identification of fungi</article-title>. <source>Mol. Ecol.</source> <volume>22</volume>, <fpage>5271</fpage>&#x2013;<lpage>5277</lpage>. <pub-id pub-id-type="doi">10.1111/mec.12481</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>Y.-Y.</given-names>
</name>
<name>
<surname>Ao</surname>
<given-names>Z.-H.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Z.-M.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>H.-Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.-M.</given-names>
</name>
<name>
<surname>Dou</surname>
<given-names>W.-F.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Analgesic and anti-inflammatory effects of the dry matter of culture broth of <italic>Termitomyces albuminosus</italic> and its extracts</article-title>. <source>J. Ethnopharmacol.</source> <volume>120</volume> (<issue>3</issue>), <fpage>432</fpage>&#x2013;<lpage>436</lpage>. <pub-id pub-id-type="doi">10.1016/j.jep.2008.09.021</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Meharunnisa</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sornam</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>CatBoost encoded tree-based model for the identification of microbes at genes level in 16S rRNA sequence</article-title>,&#x201d; in <source>Communication and intelligent systems: Proceedings of ICCIS 2021</source> (<publisher-loc>Singapore</publisher-loc>: <publisher-name>Springer Nature Singapore</publisher-name>), <fpage>1137</fpage>&#x2013;<lpage>1156</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meher</surname>
<given-names>P. K.</given-names>
</name>
<name>
<surname>Sahu</surname>
<given-names>T. K.</given-names>
</name>
<name>
<surname>Gahoi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tomar</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Rao</surname>
<given-names>A. R.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>funbarRF: DNA barcode-based fungal species prediction using multiclass Random Forest supervised learning model</article-title>. <source>BMC Genet.</source> <volume>20</volume> (<issue>1</issue>), <fpage>2</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1186/s12863-018-0710-z</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mossebo</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Njounkou</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Piatek</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kengni</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Diasbe</surname>
<given-names>M. D.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>
<italic>Termitomyces striatus</italic> f. pileatus f. nov. and f. brunneus f. nov. from Cameroon with a key to central African species</article-title>. <source>Mycotaxon</source> <volume>107</volume> (<issue>1</issue>), <fpage>315</fpage>&#x2013;<lpage>329</lpage>. <pub-id pub-id-type="doi">10.5248/407.315</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pedregosa</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Varoquaux</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Gramfort</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Michel</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Thirion</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Grisel</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Scikit-learn: machine learning in Python</article-title>. <source>J. Mach. Learn. Res.</source> <volume>12</volume>, <fpage>2825</fpage>&#x2013;<lpage>2830</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pegler</surname>
<given-names>D. N.</given-names>
</name>
<name>
<surname>Vanhaecke</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>1994</year>). <article-title>Termitomyces of southeast asia</article-title>. <source>Kew Bull.</source> <volume>49</volume>, <fpage>717</fpage>&#x2013;<lpage>736</lpage>. <pub-id pub-id-type="doi">10.2307/4118066</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Prokhorenkova</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gusev</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Vorobev</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dorogush</surname>
<given-names>A. V.</given-names>
</name>
<name>
<surname>Gulin</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>CatBoost: unbiased boosting with categorical features</article-title>,&#x201d; in <conf-name>NIPS&#x27;18: Proceedings of the 32nd International Conference on Neural Information Processing Systems</conf-name>.</citation>
</ref>
<ref id="B23">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ren</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>A novel image classification method with CNN-XGBoost model</article-title>,&#x201d; in <conf-name>Digital Forensics and Watermarking: 16th International Workshop, IWDW 2017</conf-name>, <conf-loc>Magdeburg, Germany</conf-loc>, <conf-date>August 23-25, 2017</conf-date> (<publisher-name>Springer International Publishing</publisher-name>), <fpage>378</fpage>&#x2013;<lpage>390</lpage>. <comment>Proceedings 16</comment>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Robson</surname>
<given-names>P. B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>MathFeature: feature extraction package for DNA, RNA and protein sequences based on mathematical descriptors</article-title>. <source>Briefings Bioinforma.</source> <volume>23</volume>, <fpage>1</fpage>&#x2013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab434</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Roe</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Rice</surname>
<given-names>A. V.</given-names>
</name>
<name>
<surname>Bromilow</surname>
<given-names>S. E.</given-names>
</name>
<name>
<surname>Cooke</surname>
<given-names>J. E. K.</given-names>
</name>
<name>
<surname>Sperling</surname>
<given-names>F. A. H.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Multilocus species identification and fungal DNA barcoding: insights from blue stain fungal symbionts of the mountain pine beetle</article-title>. <source>Mol. Ecol. Resour.</source> <volume>10</volume>, <fpage>946</fpage>&#x2013;<lpage>959</lpage>. <pub-id pub-id-type="doi">10.1111/j.1755-0998.2010.02844.x</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schloss</surname>
<given-names>P. D.</given-names>
</name>
<name>
<surname>Westcott</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Ryabin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hall</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Hartmann</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hollister</surname>
<given-names>E. B.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Introducing mothur: open-source, platform-independent, community-supported software for describing and comparing microbial communities</article-title>. <source>Appl. Environ. Microbiol.</source> <volume>75</volume> (<issue>23</issue>), <fpage>7537</fpage>&#x2013;<lpage>7541</lpage>. <pub-id pub-id-type="doi">10.1128/AEM.01541-09</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schoch</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Seifert</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Huhndorf</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Robert</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Spouge</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Levesque</surname>
<given-names>C. A.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Nuclear ribosomal internal transcribed spacer (ITS) region as a universal DNA barcode marker for Fungi</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>109</volume> (<issue>16</issue>), <fpage>6241</fpage>&#x2013;<lpage>6246</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1117018109</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Somervuo</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Koskela</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pennanen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Henrik Nilsson</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ovaskainen</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Unbiased probabilistic taxonomic classification for DNA barcoding</article-title>. <source>Bioinformatics</source> <volume>32</volume> (<issue>19</issue>), <fpage>2920</fpage>&#x2013;<lpage>2927</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btw346</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Velzen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Weitschek</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Felici</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bakker</surname>
<given-names>F. T.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>DNA barcoding of recently diverged species: relative performance of matching methods</article-title>. <source>PloS one</source> <volume>7</volume> (<issue>1</issue>), <fpage>e30490</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0030490</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Venkatachalapathi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Paulsamy</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Exploration of wild medicinal mushroom species in walayar valley, the southern western ghats of coimbatore district Tamil nadu</article-title>. <source>Mycosphere</source> <volume>7</volume> (<issue>2</issue>), <fpage>118</fpage>&#x2013;<lpage>130</lpage>. <pub-id pub-id-type="doi">10.5943/mycosphere/7/2/3</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>White</surname>
<given-names>T. J.</given-names>
</name>
<name>
<surname>Bruns</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S. J. W. T.</given-names>
</name>
<name>
<surname>Taylor</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1990</year>). <article-title>Amplification and direct sequencing of fungal ribosomal RNA genes for phylogenetics</article-title>. <source>PCR Protoc. a guide methods Appl.</source> <volume>18</volume> (<issue>1</issue>), <fpage>315</fpage>&#x2013;<lpage>322</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>