<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1650244</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2025.1650244</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>HFTC: a hierarchical fungal taxonomic classification model for ITS sequences using low-dimensional embedding features</article-title>
<alt-title alt-title-type="left-running-head">Wang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2025.1650244">10.3389/fgene.2025.1650244</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Jiawei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3090210/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Qiao</surname>
<given-names>Shaojie</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3199472/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xiang</surname>
<given-names>Dongsheng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3219337/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liao</surname>
<given-names>Yangcheng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3219325/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wang</surname>
<given-names>Chao</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3006687/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Software Engineering, Chengdu University of Information Technology</institution>, <addr-line>Chengdu</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Center for Genomic and Personalized Medicine, Guangxi key Laboratory for Genomic and Personalized Medicine, Guangxi Collaborative Innovation Center for Genomic and Personalized Medicine, University Engineering Research Center of Digital Medicine and Healthcare, Guangxi Medical University</institution>, <addr-line>Nanning</addr-line>, <addr-line>Guangxi</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/493051/overview">Bruno Fosso</ext-link>, University of Bari Aldo Moro, Italy</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/362670/overview">Alam Sher</ext-link>, Anhui Agricultural University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/905691/overview">Isha Monga</ext-link>, Weill Cornell Medicine, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Shaojie Qiao, <email>sjqiao@cuit.edu.cn</email>; Chao Wang, <email>wangchao@sr.gxmu.edu.cn</email>, <email>siraowang@foxmail.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>10</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1650244</elocation-id>
<history>
<date date-type="received">
<day>19</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Wang, Qiao, Xiang, Liao and Wang.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Wang, Qiao, Xiang, Liao and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Fungal identification through ITS sequencing is pivotal for biodiversity and ecological studies, yet existing methods often face challenges with high-dimensional features and inconsistent taxonomy predictions.</p>
</sec>
<sec>
<title>Method</title>
<p>We proposed HFTC, a hierarchical fungal taxonomic classifier built upon a multi-level random forest (RF) architecture. Notably, HFTC incorporates a bidirectional k-mer strategy to capture contextual information from both sequence orientations. By leveraging Word2Vec embedding, it reduces feature dimensionality from 4<sup>
<italic>k</italic>
</sup> to only 200, significantly improving computational efficiency while preserving rich sequence context.</p>
</sec>
<sec>
<title>Result</title>
<p>Experimental results demonstrate that HFTC outperforms Mothur, RDP, Sintax, QIIME2, and CNN-Duong, achieving a Matthews correlation coefficient (MCC) of 95.31% despite uneven class distributions. Its overall accuracy (ACC) reaches 95.25%. At the species level, it attains a hierarchical accuracy (HA) of 95.10%, surpassing the best-performing deep learning baseline, CNN-Duong, by 3.2%. Moreover, HFTC exhibits the smallest discrepancy between ACC and HA (1.60%), in contrast to CNN-Duong, which shows the largest gap (35.00%), highlighting HFTC&#x2019;s superior hierarchical consistency.</p>
</sec>
<sec>
<title>Discussion</title>
<p>HFTC offers a scalable and accurate approach for fungal taxonomic classification. Its compact feature representation and hierarchical architecture make it particularly suitable for microbial diversity research. The source code and datasets are publicly accessible at <ext-link ext-link-type="uri" xlink:href="https://github.com/wjjw0731/HFTC/tree/master">https://github.com/wjjw0731/HFTC/tree/master</ext-link>.</p>
</sec>
</abstract>
<kwd-group>
<kwd>fungal identification</kwd>
<kwd>ITS sequencing</kwd>
<kwd>hierarchical classification</kwd>
<kwd>Word2Vec embedding</kwd>
<kwd>random forests</kwd>
</kwd-group>
<contract-num rid="cn001">62272065 62002051</contract-num>
<contract-num rid="cn002">2024GXNSFBA010372</contract-num>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">Natural Science Foundation of Guangxi Province<named-content content-type="fundref-id">10.13039/501100004607</named-content>
</contract-sponsor>
<counts>
<page-count count="13"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Fungi are indispensable to the Earth&#x2019;s ecosystem balance and have a profound impact on human life (<xref ref-type="bibr" rid="B39">Shoemaker et al., 2017</xref>) for their essential roles in biodiversity conservation, organic matter decomposition, medicine and food production, and agricultural bio-control application (<xref ref-type="bibr" rid="B23">Lennon and Locey, 2020</xref>). Despite their ecological and biotechnological significances, fungi remain vastly underexplored. Recent estimates suggest that approximately 12 million fungal species may exist, yet only approximately 150,000 have been formally described (<xref ref-type="bibr" rid="B15">Hawksworth and L&#xfc;cking, 2017</xref>), implying that over 99% of fungal diversity remains undocumented. Therefore, accurate and scalable species identification is essential to advance microbial diversity research and functional inference.</p>
<p>Traditional fungal identification approaches, based on morphology, anatomy, or sectional analysis (<xref ref-type="bibr" rid="B34">Raja et al., 2017</xref>), are impractical when morphological traits are absent or ambiguous. Although whole-genome sequencing provides the highest resolution for identification, it remains expensive, computationally intensive, and time-consuming (<xref ref-type="bibr" rid="B13">Gan et al., 2024</xref>; <xref ref-type="bibr" rid="B36">Saada et al., 2024</xref>). As a result, DNA metabarcoding has emerged as a widely adopted alternative for microbial community profiling. This approach targets a small, species-specific, and easily amplified genomic region (<xref ref-type="bibr" rid="B34">Raja et al., 2017</xref>; <xref ref-type="bibr" rid="B11">Ficetola et al., 2010</xref>). Several key rRNA gene regions, such as ITS, LSU, SSU, RPB2, and TEF, have been used for fungal species identification (<xref ref-type="bibr" rid="B50">Zhang et al., 2020</xref>; <xref ref-type="bibr" rid="B38">Schoch et al., 2012</xref>). Compared to LSU and SSU, which evolve slowly and lack resolution for closely related species, ITS provides superior species- and strain-level discrimination. Although slower-evolving than protein-coding markers such as RPB2 or TEF, ITS offers a balanced level of variability and conservation (<xref ref-type="bibr" rid="B33">Nilsson et al., 2019</xref>; <xref ref-type="bibr" rid="B25">Lindahl et al., 2013</xref>). Consequently, ITS has been designated as the universal DNA barcode for fungi (<xref ref-type="bibr" rid="B5">Bradshaw et al., 2023</xref>; <xref ref-type="bibr" rid="B31">Nilsson et al., 2008</xref>). To support ITS-based research, several databases, including UNITE (<xref ref-type="bibr" rid="B21">K&#xf5;ljalg et al., 2020</xref>), Warcup (<xref ref-type="bibr" rid="B7">Deshpande et al., 2016</xref>), and BOLD (<xref ref-type="bibr" rid="B35">Ratnasingham and Hebert, 2007</xref>), have been developed. Among these, UNITE is comprehensive and frequently updated, containing nearly 10 million sequences grouped into 2.4 million species hypotheses (SHs) (<xref ref-type="bibr" rid="B21">K&#xf5;ljalg et al., 2020</xref>). UNITE provides a valuable taxonomic database for taxonomists. However, it also presents challenges such as data noise, taxonomic imbalance, and ambiguous categories, which must be addressed for reliable classification.</p>
<p>Barcoding-based methods for microbiome classification are divided into alignment-based and alignment-free categories (<xref ref-type="bibr" rid="B3">Borozan et al., 2015</xref>). Alignment-based methods [e.g., BLAST (<xref ref-type="bibr" rid="B1">Altschul et al., 1990</xref>)] rely on pairwise sequence similarity searches, which are accurate but computationally intensive. In contrast, alignment-free methods offer speed and scalability and are increasingly adopted for high-throughput microbiome analysis (<xref ref-type="bibr" rid="B43">van Zyl et al., 2025</xref>; <xref ref-type="bibr" rid="B52">Zhou et al., 2019</xref>). These methods typically use k-mer frequency vectors (KFVs) for sequence representation (<xref ref-type="bibr" rid="B16">Jenike et al., 2024</xref>), leading to high-dimensional (4<sup>
<italic>k</italic>
</sup>), sparse representation, sensitivity to noise, and limited interpretability (<xref ref-type="bibr" rid="B48">Wichmann et al., 2023</xref>), which can hinder model performance and increase computational burden. Most existing models adopt a flat classification architecture, using a single model to predict all taxonomic ranks simultaneously. This structure lacks hierarchical awareness and often yields inconsistent predictions. For instance, a sample misclassified at a higher level (e.g., phylum or class) may still appear correct at lower levels (e.g., genus or species), which is biologically invalid. Despite this, accuracy at each level is typically reported in isolation, without accounting for upstream errors. For instance, if one sequence has the correct phylum but incorrect class and another has the wrong phylum but correct class, one of them will always be counted as correct in single-rank evaluations (either at the phylum or class level), although both are taxonomically inconsistent. This can inflate the perceived performance and obscure the true reliability of the model in practical taxonomic applications.</p>
<p>In this study, we address these challenges by proposing HFTC, a hierarchical fungal taxonomic classifier based on ITS sequences. First, to mitigate the effects of data imbalance and noise, we rigorously curated the UNITE database to construct a high-quality ITS dataset. Second, to reduce dimensionality and improve contextual representation, we adopted a bi-directional k-mer (Bi-kmer) strategy to capture richer sequence context information and applied Word2Vec embedding (<xref ref-type="bibr" rid="B48">Wichmann et al., 2023</xref>; <xref ref-type="bibr" rid="B47">Wang et al., 2020</xref>; <xref ref-type="bibr" rid="B30">Neelima and Mehrotra, 2023</xref>; <xref ref-type="bibr" rid="B2">Asim et al., 2020</xref>) to compress the feature space from 4<sup>
<italic>k</italic>
</sup> to only 200 dimensions. Third, we developed a multi-level random forest (RF) architecture to ensure taxonomic consistency. Together, these efforts enable more accurate, efficient, and consistent classification of fungal species. Experimental results demonstrate that HFTC significantly outperforms baseline approaches in both feature dimensions and hierarchical consistency.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<p>
<xref ref-type="fig" rid="F1">Figure 1</xref> illustrates the overall workflow of HFTC, which comprises four main stages&#x2014;from raw ITS sequence processing to final taxonomic prediction&#x2014;to address key challenges in fungal classification.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Schematic workflow of HFTC, a hierarchical fungal taxonomic classification framework. The pipeline consists of four main stages: data preparation, feature engineering, model development, and model evaluation.</p>
</caption>
<graphic xlink:href="fgene-16-1650244-g001.tif">
<alt-text content-type="machine-generated">Flowchart of a system for classifying fungi sequences. Data preparation involves cleaning data from UNITE by removing ambiguous labels, excluding invalid sequences, and filtering based on sequence count. The training set is randomly split, and the test set is structured by sequence size. Features are engineered using k-mer and skip-gram methods, concatenated into vectors. Model development visualizes classification of k. fungi, showing hierarchical decisions for sequence homology. Model evaluation employs hierarchical fine-tuning classification, assessing using ACC, HA, precision, recall, F1-score, and MCC.</alt-text>
</graphic>
</fig>
<sec id="s2-1">
<title>2.1 Data extraction and preprocessing</title>
<p>The UNITE database focuses on the eukaryotic nuclear ribosomal ITS region, where sequences are clustered into SHs based on pairwise similarity thresholds of 0.5% (<xref ref-type="bibr" rid="B20">K&#xf5;ljalg et al., 2013</xref>; <xref ref-type="bibr" rid="B19">Karsch-Mizrachi et al., 2018</xref>). For this study, we retrieved the full fungal ITS dataset from the UNITE dataset (v9.0) (<xref ref-type="bibr" rid="B12">UNITE Community, 2023</xref>), which integrates fungal sequences from both UNITE and INSD. The crude dataset includes 6,499,364 sequences.</p>
<p>To ensure data quality, we implemented a rigorous data preprocessing pipeline. First, 1,325,964 sequences labeled as unidentified or ambiguous (e.g., <italic>incertae sedis</italic>) and 66,707 erroneous sequences containing non-standard nucleotide bases were excluded (<xref ref-type="bibr" rid="B32">Nilsson et al., 2018</xref>). To further improve annotation reliability and reduce noise, 295,108 sequences from SHs with fewer than 10 representative sequences were excluded. These rare taxa typically lack sufficient intra-class variation to support stable training and are more susceptible to misannotation, potentially introducing bias. This filtering process represents a necessary trade-off between taxonomic inclusiveness and model reliability. Given the scale of this study&#x2014;spanning over 25,000 fungal SHs&#x2014;it constitutes a challenging large-scale multi-class classification task. The exclusion of underrepresented SHs has minimal impact on overall taxonomic coverage as the curated dataset already captures the major fungal lineages. Instead, this strategy significantly improves training efficiency and consistency without compromising fungal diversity.</p>
<p>Among the remaining SHs, the number of sequences varied widely, ranging from 10 to tens of thousands. To address this imbalance, we randomly sampled 10 sequences from each SH with more than 10 sequences. This step minimized sequence redundancy and ensured uniform representation across taxa. Since SHs in the UNITE database are clustered based on ITS sequence similarity and serve as proxies for species, this sampling strategy not only balances data distribution but also approximates species-level stratified sampling. It helps prevent model overfitting to overrepresented taxa, thereby promoting robustness and generalizability. As a result, the final training set comprised 251,630 sequences representing 25,163 fungal species. The data and associated metadata have been deposited in Zenodo in accordance with community metadata standards, available at <ext-link ext-link-type="uri" xlink:href="https://zenodo.org/uploads/14826761">https://zenodo.org/uploads/14826761</ext-link>.</p>
<p>To rigorously assess model performance, we constructed five independent test sets using only sequences that were completely absent from the training set, ensuring that no sequences overlap. In particular, we constructed five independent test sets (Test10&#x2013;Test30) by randomly selecting 10&#x2013;30 representative sequences per species, respectively. Dataset distributions are summarized in <xref ref-type="table" rid="T1">Table 1</xref>, and the most species-rich taxa at each taxonomic level are given in <xref ref-type="sec" rid="s11">Supplementary Image S1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Taxonomic coverage information for training and testing datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Dataset</th>
<th align="center">Kingdom</th>
<th align="center">Phylum</th>
<th align="center">Class</th>
<th align="center">Order</th>
<th align="center">Family</th>
<th align="center">Genus</th>
<th align="center">Species</th>
<th align="center">Total no.</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Training</td>
<td align="center">1</td>
<td align="center">15</td>
<td align="center">56</td>
<td align="center">181</td>
<td align="center">604</td>
<td align="center">2,683</td>
<td align="center">25,163</td>
<td align="center">251,630</td>
</tr>
<tr>
<td align="center">Test10</td>
<td align="center">1</td>
<td align="center">15</td>
<td align="center">52</td>
<td align="center">164</td>
<td align="center">520</td>
<td align="center">2,034</td>
<td align="center">15,027</td>
<td align="center">150,270</td>
</tr>
<tr>
<td align="center">Test15</td>
<td align="center">1</td>
<td align="center">14</td>
<td align="center">48</td>
<td align="center">146</td>
<td align="center">452</td>
<td align="center">1,646</td>
<td align="center">9,965</td>
<td align="center">149,475</td>
</tr>
<tr>
<td align="center">Test20</td>
<td align="center">1</td>
<td align="center">12</td>
<td align="center">44</td>
<td align="center">138</td>
<td align="center">417</td>
<td align="center">1,398</td>
<td align="center">7,082</td>
<td align="center">141,640</td>
</tr>
<tr>
<td align="center">Test25</td>
<td align="center">1</td>
<td align="center">12</td>
<td align="center">40</td>
<td align="center">120</td>
<td align="center">360</td>
<td align="center">1,165</td>
<td align="center">5,304</td>
<td align="center">132,600</td>
</tr>
<tr>
<td align="center">Test30</td>
<td align="center">1</td>
<td align="center">9</td>
<td align="center">36</td>
<td align="center">110</td>
<td align="center">335</td>
<td align="center">1,029</td>
<td align="center">4,149</td>
<td align="center">124,441</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-2">
<title>2.2 Sequence feature representation</title>
<p>In this work, we used Word2Vec to embed ITS sequences into dense numerical vectors. Each ITS sequence was regarded as a sentence, with k-mers serving as words whose contextual patterns capture taxonomic information. To construct a comprehensive sequence corpus, k-mers were extracted in both the forward and reverse directions (<xref ref-type="bibr" rid="B29">Mar&#xe7;ais et al., 2024</xref>) using a sliding window of length <italic>k</italic> and stride length <italic>L</italic>. This bidirectional approach captures richer local sequence context and improves embedding robustness. For Word2Vec training, we adopted the skip-gram model instead of the Continuous Bag-of-Words (CBOW) model. Skip-gram performs better in capturing rare or infrequent k-mers by directly predicting surrounding context words from a given center word (<xref ref-type="bibr" rid="B41">TH et al., 2015</xref>). In contrast, CBOW averages the context to predict the center word, which tends to oversmooth representations and underperform on sparse biological sequences such as fungal ITS data (<xref ref-type="bibr" rid="B6">Chiu et al., 2016</xref>).</p>
<p>After training, each k-mer was mapped to an <italic>N</italic>-dimensional embedding vector. To represent a full ITS sequence, we computed the average of all embedded k-mer vectors from each direction separately and then concatenated them to obtain a final 2<italic>N</italic>-dimensional sequence-level vector.</p>
</sec>
<sec id="s2-3">
<title>2.3 Construction of HFTC</title>
<p>To address the inconsistency of classification results across taxonomic levels, we proposed HFTC, a novel model that aligned predictions with phylogenetic relationships from phylum to species (<xref ref-type="bibr" rid="B18">Ji et al., 2023</xref>). Unlike conventional flat models, which predict all taxonomic ranks simultaneously and may yield spuriously high accuracy, HFTC decomposes the task into sequential subtasks, each handled by an independently trained sub-classifier. By integrating predictions across levels, HFTC reduces overall complexity and ensures taxonomic consistency (<xref ref-type="bibr" rid="B49">Zhang and Zhou, 2013</xref>).</p>
<p>To further support this hierarchical design, we adopted random forests (<xref ref-type="bibr" rid="B27">Breiman, 2001</xref>) as the base classifiers for each level. Compared with more complex models such as neural networks (NNs), RFs are more robust to class imbalance, require fewer computational resources, and are less sensitive to hyperparameter tuning (<xref ref-type="bibr" rid="B10">Fern&#xe1;ndez-Delgado et al., 2014</xref>). Moreover, the hierarchical tree-like structure of HFTC naturally aligns with the modular design, whereas NNs struggle with vanishing gradients and poor generalization in long-tailed settings (<xref ref-type="bibr" rid="B51">Zhang et al., 2023</xref>; <xref ref-type="bibr" rid="B22">Le Guillarme and Thuiller, 2022</xref>). Additionally, many taxonomic groups contain very few sequences, making it infeasible to train dedicated classifiers. To address this, taxa with fewer than 1,000 sequences were grouped into an &#x201c;Other&#x201d; category. Due to the strong multi-class capability of RFs, a single classifier could then be trained from this node to directly predict species-level labels, despite the diversity and imbalance within the group. This strategy reduces the number of required sub-classifiers and streamlines the classification pipeline. <xref ref-type="table" rid="T2">Table 2</xref> summarizes the recursive construction of HFTC across taxonomic levels.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Training procedure for the Hierarchical Fungal Taxonomic Classifier (HFTC).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Algorithm 1 training HFTC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<bold>Input</bold>
<break/>&#x2003;Sequences for ITS_10 Dataset<break/>&#x2003;Extracted sequence features<break/>
<bold>Output</bold>
<break/>&#x2003;Species-level taxonomics for fungal ITS sequences<break/>1: Start taxonomic classification at the Kingdom level<break/>2: Construct a sub-classifier to predict the Phylum level<break/>3: <bold>if</bold> any Phylum contains fewer than 1,000 species <bold>then</bold>
<break/>4:&#x2003;Group it as &#x201c;Other Phyla&#x201d; for direct species-level classification<break/>5: <bold>end if</bold>
<break/>6: Continue to the next taxonomic level<break/>7: <bold>while</bold> taxonomic levels remain to be classified <bold>do</bold>
<break/>8:&#x2003;Construct a sub-classifier to predict the next lower taxonomic level<break/>9:&#x2003;<bold>if</bold> the lower-level taxonomic group contains fewer than 1,000 species <bold>then</bold>
<break/>10:&#x2003;&#x2003;Group it as &#x201c;Other&#x201d; for direct species-level classification<break/>11:&#x2003;<bold>end if</bold>
<break/>12:&#x2003;Continue descending through the taxonomic hierarchy<break/>13: <bold>end while</bold>
<break/>14: Perform direct species-level classification on all remaining taxonomic groups<break/>15. <bold>return</bold> Species-level taxonomics for fungal ITS sequences</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-4">
<title>2.4 Model evaluation metrics</title>
<p>We adopt five metrics as foundational indicators to evaluate model&#x2019;s performance: accuracy (ACC), recall, F<sub>1</sub>-score (F<sub>1</sub>), precision, and Matthews correlation coefficient (MCC) (<xref ref-type="bibr" rid="B14">GMJP and o, 2023</xref>; <xref ref-type="bibr" rid="B4">Boughorbel et al., 2017</xref>). They were calculated using <xref ref-type="disp-formula" rid="e1">Equations 1</xref>&#x2013;<xref ref-type="disp-formula" rid="e5">5</xref>:<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mtext>ACC</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>TN</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>TN</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FN</mml:mtext>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:mtext>Recall</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mtext>TP</mml:mtext>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FN</mml:mtext>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
<disp-formula id="e3">
<mml:math id="m3">
<mml:mrow>
<mml:mtext>Precision</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mtext>TP</mml:mtext>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FP</mml:mtext>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
<disp-formula id="e4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:mfrac>
<mml:mtext>TP</mml:mtext>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FN</mml:mtext>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m5">
<mml:mrow>
<mml:mtext>MCC</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>TN</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>FP</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>FN</mml:mtext>
</mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FP</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FN</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>TN</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FP</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>TN</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FN</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where TP represents true positive, FP represents false positive, TN represents true negative, and FN represents false negative.</p>
<p>Conventional metrics in taxonomic classification assess performance at individual levels but may overlook inconsistencies across the hierarchy. To address this limitation, we adopted the hierarchical accuracy (HA) (<xref ref-type="bibr" rid="B42">Tieppo et al., 2022</xref>) metric. HA counts a sample as a true positive only if all its hierarchical predictions are correct, quantifying fully correctly classified samples across the taxonomic path, as presented in <xref ref-type="disp-formula" rid="e6">Equation 6</xref>:<disp-formula id="e6">
<mml:math id="m6">
<mml:mrow>
<mml:mtext>HA</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msup>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2a;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>TN</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FN</mml:mtext>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where TP<sup>&#x2a;</sup> denotes the true positives on the complete taxonomic classification path.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Result and discussion</title>
<sec id="s3-1">
<title>3.1 Division of sub-classifiers in HFTC</title>
<p>HFTC is hierarchically constructed based on fungal phylogenetic taxonomy. In particular, an RF classifier first assigns sequences to phyla at the kingdom level. Phyla with fewer than 1,000 SHs are grouped into &#x201c;Other Phyla&#x201d; for direct species-level classification. For the major phyla, further hierarchical classification is performed. In this study, <italic>Basidiomycota</italic> and <italic>Ascomycota</italic> are the two most abundant phyla; in this study, <italic>Basidiomycota</italic> is divided into four specific classes and one &#x201c;Other Class&#x201d; category, while <italic>Ascomycota</italic> is divided into <italic>Agaricomycetes</italic> and &#x201c;Other Class.&#x201d; At the order level, 11 orders, except <italic>Agaricomycetes</italic>, were trained at the species level due to a significant reduction in species per order. For <italic>Agaricomycetes</italic>, RF classifiers are constructed at five order and four family levels. As a result, we constructed a total of 21 RF-based sub-classifiers. Considering the substantial variation in task complexity among sub-classifiers&#x2014;ranging from dozens to tens of thousands of categories&#x2014;establishing a unified confidence threshold becomes challenging. Given that HFTC&#x2019;s hierarchical architecture effectively mitigates error propagation, we opted against implementing a confidence-based early stopping mechanism. Instead, our approach leverages the highest-confidence predictions from each RF-based sub-classifier as the final output, ensuring robust taxonomic assignments while maintaining computational efficiency. Nonetheless, users are allowed to apply confidence thresholds depending flexibly on task-specific requirements.</p>
</sec>
<sec id="s3-2">
<title>3.2 Strategy and optimization for feature embedding</title>
<sec id="s3-2-1">
<title>3.2.1 Evaluating the performance of various <italic>k</italic>-values in HFTC</title>
<p>To optimize feature representation across the 21 derived hierarchical sub-classifiers, we systematically evaluated the optimal Bi-kmer length <italic>k</italic> for each taxonomic level by five-fold cross-validation. Initially, all sequences were represented as Bi-kmers and then embedded into 200-dimensional numerical vectors using Word2Vec. As the existing methods typically select <italic>k</italic>-values between 7 and 10 (<xref ref-type="bibr" rid="B37">Schloss et al., 2009</xref>; <xref ref-type="bibr" rid="B45">Wang et al., 2007</xref>; <xref ref-type="bibr" rid="B9">Edgar, 2016</xref>; <xref ref-type="bibr" rid="B24">Liimata et al., 2022</xref>; <xref ref-type="bibr" rid="B44">Vu et al., 2020</xref>), we evaluated the <italic>k</italic>-values from 7 to 11 to select the optimal feature representation at each taxonomic level. The accuracy achieved by sub-classifiers using individual <italic>k</italic>-values is presented in <xref ref-type="fig" rid="F2">Figure 2</xref>. The hierarchical classification paths and optimal <italic>k</italic>-values of each sub-classifiers are detailed in <xref ref-type="sec" rid="s11">Supplementary Table S1</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Taxonomic-level optimization of Bi-kmer length. Classification accuracy of 21 sub-classifiers evaluated across Bi-kmer lengths (<italic>k</italic> &#x3d; 7&#x2013;11) on the Test10 dataset. Taxonomic levels are abbreviated as follows: k, kingdom; p, phylum; c, class; o, order; f, family; s, species.</p>
</caption>
<graphic xlink:href="fgene-16-1650244-g002.tif">
<alt-text content-type="machine-generated">Bar chart comparing accuracy across various models named on the x-axis, using colored bars for 7-mer to 11-mer datasets. Most bars show high accuracy around 0.85 to 1.00, with some variation among the models.</alt-text>
</graphic>
</fig>
<p>The results indicate that classifiers across different taxonomic levels exhibit an opposite trend in their optimal k-mer lengths. At higher taxonomic levels (phylum to family), classification accuracy generally peaked at <italic>k</italic> &#x3d; 10. In particular, for phylum-level classification, the sub-classifier achieved the highest accuracy of 98.42% at <italic>k</italic> &#x3d; 10, followed by 98.01% at <italic>k</italic> &#x3d; 11, representing a 1.08% improvement over 97.34% at <italic>k</italic> &#x3d; 7. For two representative class-level classifiers, the highest accuracies were observed at <italic>k</italic> &#x3d; 10 for 98.34% (vs. 98.16% at <italic>k</italic> &#x3d; 11) and <italic>k</italic> &#x3d; 11 for 95.43% (vs. 95.42% at <italic>k</italic> &#x3d; 10). For order- and family-level classifications, <italic>k</italic> &#x3d; 10 again yielded the best accuracy&#x2014;95.32% and 91.93%, respectively&#x2014;surpassing all other tested <italic>k</italic>-values. In contrast, species-level classification exhibited an opposite trend, favoring shorter k-mers. Among the 16 species-level sub-classifiers, 13 achieved peak accuracy at <italic>k</italic> &#x3d; 7. Two exceptional cases achieved optimal performance with <italic>k</italic> &#x3d; 8 for 98.62% (vs. 98.45% at <italic>k</italic> &#x3d; 7) and 96.66% (vs. 96.48% at <italic>k</italic> &#x3d; 7). One sub-classifier showed equal performance at <italic>k</italic> &#x3d; 7 and <italic>k</italic> &#x3d; 8 for 97.80%. As stated above, we identify <italic>k</italic> &#x3d; 7 and <italic>k</italic> &#x3d; 10 as the optimal single <italic>k</italic>-value of k-mer for species-level and higher-level taxonomic classification tasks, respectively.</p>
<p>To systematically evaluate the impact of k-mer combinations on classification performance across different taxonomic levels, we compared the optimal single <italic>k</italic>-values with their adjacent k-mer combinations. <xref ref-type="fig" rid="F3">Figure 3</xref> presents the accuracy results of five classifiers at higher levels in panel A and sixteen sub-classifiers at the species level in panel B.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Comparison of single vs. hybrid k-mer strategies across taxonomic levels. <bold>(A)</bold> Accuracy of single <italic>k</italic> &#x3d; 10 and its hybrid combinations (10 &#x2b; 7, 10 &#x2b; 8, 10 &#x2b; 9, and 10 &#x2b; 11) in five higher-level sub-classifiers. <bold>(B)</bold> Accuracy of single <italic>k</italic> &#x3d; 7 and its hybrid combinations (7 &#x2b; 8, 7 &#x2b; 9, 7 &#x2b; 10, and 7 &#x2b; 11) in 16 species-level sub-classifiers.</p>
</caption>
<graphic xlink:href="fgene-16-1650244-g003.tif">
<alt-text content-type="machine-generated">Bar charts illustrating model accuracy for different configurations. Chart A compares five models with 10-mer variations, showing accuracy mostly above 0.95. Chart B presents multiple models with 7-mer variations, generally achieving similar high accuracy levels, except for Sordariomycetes_o2f, which is lower.</alt-text>
</graphic>
</fig>
<p>In higher-level classification (<xref ref-type="fig" rid="F3">Figure 3A</xref>), using <italic>k</italic> &#x3d; 10 achieved the highest accuracy in two out of five sub-classifiers. In the remaining three cases, <italic>k</italic> &#x3d; 10 ranked second, with accuracy reductions of less than 1% compared to the best hybrid combinations. This indicates that the single <italic>k</italic> &#x3d; 10 setting is sufficient for robust performance at broader taxonomic ranks. At the species level (<xref ref-type="fig" rid="F3">Figure 3B</xref>), 10 of the 16 sub-classifiers yielded peak accuracy with single <italic>k</italic> &#x3d; 7; three cases reached optimal accuracy with a combination of <italic>k</italic> &#x3d; 7 and <italic>k</italic> &#x3d; 8; one with the <italic>k</italic> &#x3d; 7 and <italic>k</italic> &#x3d; 9 combination; and two with the <italic>k</italic> &#x3d; 7 and <italic>k</italic> &#x3d; 11 combination. These results suggest that hybrid k-mer features do not offer a significant advantage over well-chosen single <italic>k</italic>-values.</p>
<p>Overall, the findings demonstrate that adopting the optimal single <italic>k</italic>-value (<italic>k</italic> &#x3d; 10 for higher levels, <italic>k</italic> &#x3d; 7 for species level) provides a favorable trade-off between accuracy and model simplicity, without the need for additional complexity introduced by combining multiple k-mer lengths.</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Biological and statistical rationale for k-mer selection</title>
<p>Biologically, sequences within the same genus often share highly similar overall structures, with distinguishing signals typically confined to subtle local variations such as point mutations and short insertions or deletions (indels) (<xref ref-type="bibr" rid="B28">Mahadani and Ghosh, 2014</xref>). Therefore, species-level classification demands sensitivity to fine-grained, localized sequence differences. Shorter k-mers (e.g., <italic>k</italic> &#x3d; 7 in this study) are better suited to capturing these microvariations, particularly within the hypervariable regions of the ITS sequence, which are the primary source of discriminatory information among closely related fungal taxa (<xref ref-type="bibr" rid="B26">Liu et al., 2025</xref>). Statistically, shorter k-mers also increase the overlap between local motifs, enhancing the resolution of small-scale mutations. This dense representation improves the model&#x2019;s ability to differentiate among species based on minimal but biologically meaningful sequence differences. In contrast, higher-level classifications (e.g., phylum or class) involve greater evolutionary divergence, which manifests as broader conserved motifs or structural variations (<xref ref-type="bibr" rid="B40">Tedersoo et al., 2018</xref>). Statistically, longer k-mers yield sparser but more distinctive representations, reducing feature redundancy and increasing theoretical entropy across the k-mer space (<xref ref-type="bibr" rid="B46">Wang et al., 2018</xref>). This enhances inter-class separability and improves the robustness and accuracy of classification at broader taxonomic ranks.</p>
<p>Experiments highlight the advantage of a hierarchical feature design, where shorter k-mers are better suited for capturing fine-grained sequence variations at the species level, and longer k-mers provide improved resolution for distinguishing broader taxonomic groupings. The ability of HFTC to adaptively select k-mer lengths according to taxonomic level is a key factor underlying its robust and accurate performance across the entire fungal taxonomic hierarchy.</p>
</sec>
<sec id="s3-2-3">
<title>3.2.3 Evaluation of sequence features embedding using Word2Vec</title>
<p>After determining the optimal <italic>k</italic>-values for each taxonomic level, we further evaluated the accuracy of Word2Vec embeddings by comparing them to traditional KFVs and applied both methods to five representative class-to-species sub-classifiers within the phylum <italic>Ascomycota</italic>.</p>
<p>As shown in <xref ref-type="fig" rid="F4">Figure 4</xref>, our method approach achieved an accuracy that was comparable to that of the KFV method. Traditional KFV methods produce extremely high-dimensional and sparse feature spaces (4<sup>
<italic>k</italic>
</sup>), causing memory bottlenecks during model training. In contrast, Word2Vec generates compact, dense representations by learning distributed vector embeddings for k-mers based on their contextual co-occurrence patterns. Importantly, our method reduced the feature dimensionality from 4<sup>7</sup> (e.g., 16,384 for <italic>k</italic> &#x3d; 7) to only 200 by applying average pooling over the Word2Vec-learned vectors of each Bi-kmer in the sequence, highlighting the embedding model&#x2019;s ability to preserve relevant biological information with significantly fewer features.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Accuracy comparison between KFVs and Word2Vec embedding. Accuracy of the five class-to-species sub-classifiers in <italic>Ascomycota</italic> using KFVs and Word2Vec embedding.</p>
</caption>
<graphic xlink:href="fgene-16-1650244-g004.tif">
<alt-text content-type="machine-generated">Bar chart comparing accuracy of KFV and Word2Vec models across five classes: Sordariomycetes, Dothideomycetes, Leotiomycetes, Eurotiomycetes, and Pezizomycetes. KFV and Word2Vec show similar accuracies, with Pezizomycetes being the highest for both.</alt-text>
</graphic>
</fig>
<p>Although averaging simplifies computation and mitigates noise from sequence length variation, it has limitations. In particular, it discards positional information, potentially overlooking structural motifs or taxonomically informative subsequences. Nonetheless, this embedding strategy significantly reduces feature dimensionality, mitigates sparsity, and captures essential compositional and contextual information for downstream classification tasks.</p>
</sec>
</sec>
<sec id="s3-3">
<title>3.3 Model performance evaluation</title>
<sec id="s3-3-1">
<title>3.3.1 Hierarchical sub-classifier validation</title>
<p>The 21 sub-classifiers comprising the HFTC model were systematically evaluated through five-fold cross-validation in both the training and test datasets (Test_10). Three critical aspects are evaluated in <xref ref-type="table" rid="T3">Table 3</xref>(left: training set; right: test set): the number of taxonomic tasks covered (<xref ref-type="bibr" rid="B39">Shoemaker et al., 2017</xref>), the number of sequences (<xref ref-type="bibr" rid="B23">Lennon and Locey, 2020</xref>), and classification accuracy metrics (<xref ref-type="bibr" rid="B15">Hawksworth and L&#xfc;cking, 2017</xref>).</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Information of sub-classifiers in the training and test datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model name</th>
<th align="center">No. of taxa</th>
<th align="center">No. of sequences</th>
<th align="center">Accuracy</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Fungi2p</td>
<td align="center">15/15</td>
<td align="center">251,630/150,270</td>
<td align="center">0.9983/0.9998</td>
</tr>
<tr>
<td align="center">other_p2s</td>
<td align="center">1,312/757</td>
<td align="center">13,111/7,564</td>
<td align="center">0.9679/0.9533</td>
</tr>
<tr>
<td align="center">Ascomycota_p2c</td>
<td align="center">21/19</td>
<td align="center">106,760/60,984</td>
<td align="center">0.9542/0.9973</td>
</tr>
<tr>
<td align="center">Sordariomycetes_c2s</td>
<td align="center">2,868/1,620</td>
<td align="center">28,465/16,041</td>
<td align="center">0.9612/0.9562</td>
</tr>
<tr>
<td align="center">Dothideomycetes_c2s</td>
<td align="center">1,978/1,165</td>
<td align="center">19,655/11,578</td>
<td align="center">0.9709/0.9656</td>
</tr>
<tr>
<td align="center">Leotiomycetes_c2s</td>
<td align="center">1,294/801</td>
<td align="center">12,816/7,968</td>
<td align="center">0.9629/0.9610</td>
</tr>
<tr>
<td align="center">Eurotiomycetes_c2s</td>
<td align="center">2,023/1,173</td>
<td align="center">20,016/11,629</td>
<td align="center">0.9403/0.9236</td>
</tr>
<tr>
<td align="center">Pezizomycetes_c2s</td>
<td align="center">1,148/763</td>
<td align="center">11,443/7,621</td>
<td align="center">0.9807/0.9814</td>
</tr>
<tr>
<td align="center">Ascomycota_p_other_c2s</td>
<td align="center">1,444/616</td>
<td align="center">14,365/6,174</td>
<td align="center">0.9648/0.9639</td>
</tr>
<tr>
<td align="center">Basidiomycota_p2c</td>
<td align="center">14/13</td>
<td align="center">131,759/81,722</td>
<td align="center">0.9861/0.9983</td>
</tr>
<tr>
<td align="center">Agaricomycetes_c2o</td>
<td align="center">21/19</td>
<td align="center">124,875/77,821</td>
<td align="center">0.9532/0.9958</td>
</tr>
<tr>
<td align="center">Agaricales_o2f</td>
<td align="center">41/39</td>
<td align="center">65,887/39,810</td>
<td align="center">0.9193/0.9879</td>
</tr>
<tr>
<td align="center">Cortinariaceae_f2s</td>
<td align="center">1,175/679</td>
<td align="center">11,733/6,772</td>
<td align="center">0.8354/0.8071</td>
</tr>
<tr>
<td align="center">Inocybaceae_f2s</td>
<td align="center">1,377/982</td>
<td align="center">13,761/9,820</td>
<td align="center">0.9826/0.9778</td>
</tr>
<tr>
<td align="center">Agaricales_o_other1_f2s</td>
<td align="center">2,239/1,330</td>
<td align="center">22,267/13,259</td>
<td align="center">0.9778/0.9596</td>
</tr>
<tr>
<td align="center">Agaricales_o_other2_f2s</td>
<td align="center">1,835/2,989</td>
<td align="center">18,126/29,851</td>
<td align="center">0.9765/0.9742</td>
</tr>
<tr>
<td align="center">Russulales_o2s</td>
<td align="center">1,746/1,214</td>
<td align="center">17,402/12,100</td>
<td align="center">0.9429/0.9376</td>
</tr>
<tr>
<td align="center">Sebacinales_o2s</td>
<td align="center">1,030/634</td>
<td align="center">10,300/6,340</td>
<td align="center">0.9520/0.9395</td>
</tr>
<tr>
<td align="center">Cantharellales_o2s</td>
<td align="center">983/628</td>
<td align="center">9,784/6,220</td>
<td align="center">0.9845/0.9848</td>
</tr>
<tr>
<td align="center">Agaricomycetes_c_other_o2s</td>
<td align="center">2,164/1,344</td>
<td align="center">21,502/13,351</td>
<td align="center">0.9780/0.9648</td>
</tr>
<tr>
<td align="center">Basidiomycota_p_other_c2s</td>
<td align="center">689/392</td>
<td align="center">6,884/3,901</td>
<td align="center">0.9640/0.9693</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Notably, the sub-classifiers maintained consistently high accuracy across all hierarchical taxonomic levels, with 17/21 achieving &#x3e;95% classification accuracy (peak performance: <italic>Fungi2p</italic> reached 99.83% and 99.98% in training and test sets, respectively). This robust performance demonstrates the model&#x2019;s dual capability in handling both coarse-grained (phylum-level) and fine-grained (species-level) taxonomic assignments. These results were all validated through 10-fold cross-validation experiments, with most of the standard deviations not exceeding 0.01, indicating stable model performance. The detailed standard deviations and 95% confidence intervals for the accuracy of each of the 21 sub-classifiers are also provided in <xref ref-type="sec" rid="s11">Supplementary Table S1</xref>. Moreover, the minimal differences in accuracy between the training and test datasets highlight a strong generalization capability, indicating that the sub-classifiers are not overfitting and can reliably classify previously unseen data with high precision. However, the Cortinariaceae_f2s classifier for classifying species within the <italic>Cortinariaceae</italic> family of the order <italic>Agaricales</italic> performed poorly, with an accuracy of only 83.54% and 80.71% in training and test sets, respectively. In contrast, the classifier Inocybaceae_f2s for another family within <italic>Agaricales</italic> achieved an accuracy of 98.26%.</p>
<p>To improve this model, we applied GridSearch on the training dataset to find the best performance combination. We tuned two hyperparameters: the number of features considered at each split (max_features) was tuned using a combination of fixed values and dynamic strategies commonly used in tree-based models, specifically {2, 5, 10, &#x2018;log2&#x2019;, &#x2018;sqrt&#x2019;} (<xref ref-type="bibr" rid="B39">Shoemaker et al., 2017</xref>). The dynamic options automatically select the number of features by taking either the base-2 logarithm or the square root of the total number of input features; the number of trees in the forest can be chosen from {50, 100, 200, 500, 800} (<xref ref-type="bibr" rid="B23">Lennon and Locey, 2020</xref>). <xref ref-type="fig" rid="F5">Figure 5</xref> shows heatmaps of ACC, recall, F<sub>1</sub>, and precision across the parameter combinations.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Hyperparameter optimization for Cortinariaceae_f2s via GridSearch. Accuracy, F<sub>1</sub>-score, precision, and recall values of the Cortinariaceae_f2s model constructed using different numbers of features and estimators.</p>
</caption>
<graphic xlink:href="fgene-16-1650244-g005.tif">
<alt-text content-type="machine-generated">Four heatmaps display metrics for model performance based on `n_estimators` and `max_features` parameters. Each heatmap represents Accuracy, Precision, Recall, and F1 Score, with values ranging from roughly 0.7960 to 0.8348. Variations in color intensity indicate performance differences across configurations.</alt-text>
</graphic>
</fig>
<p>When the two hyperparameters approached values of 2 and 800, the model achieved its highest accuracy of 0.8348. However, the model&#x2019;s performance remained suboptimal despite extensive tuning via GridSearch. This indicates that the limited performance of the Cortinariaceae_f2s model is unlikely to be attributable to hyperparameter settings. Default RF parameters proved satisfactory results for most HFTC sub-classifiers; furthermore, tuning was avoided to prevent unnecessary complexity. To explore other potential causes, we examined the upstream and downstream tasks of the Agaricales_o2f and Cortinariaceae_f2g models, with the corresponding classification heatmaps presented in <xref ref-type="fig" rid="F6">Figure 6</xref>.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Hierarchical error propagation analysis in HFTC&#x2019;s taxonomic classification pipeline. <bold>(A)</bold> Heatmaps of upstream tasks: Agaricales_o2f model and <bold>(B)</bold> heatmaps of downstream tasks: Cortinariaceae_f2g model.</p>
</caption>
<graphic xlink:href="fgene-16-1650244-g006.tif">
<alt-text content-type="machine-generated">Panel (A) shows a heatmap with three categories: Cortinariaceae, Inocybaceae, and other, displaying percentage values. Cortinariaceae has 99.95%, Inocybaceae 100%, and other 99.83% in their respective main boxes. Panel (B) is a more detailed heatmap with taxa names on the vertical axis and various percentages per category, with a color gradient from light to dark red indicating the values.</alt-text>
</graphic>
</fig>
<p>The heatmaps reveal that the upstream classifier for the Agaricales order achieves nearly perfect accuracy in assigning sequences to the <italic>Cortinariaceae</italic> family for 99.95%, indicating that most errors in Cortinariaceae_f2s are not caused by upstream misclassification. Within the <italic>Cortinariaceae</italic> family, notable genus-level misclassifications occur, with many genera&#x2014;particularly <italic>Cystinarius</italic>, <italic>Hygronarius</italic>, <italic>Protoglossum</italic>, and <italic>Volvanarius</italic>&#x2014;erroneously predicted as <italic>Cortinarius</italic>. This bias likely stems from the dominance of Cortinarius in the training data and the limited representation of other genera. Additionally, <italic>Cystinarius</italic>, <italic>Hygronarius</italic>, and <italic>Volvanarius</italic>, which only recently split from <italic>Cortinarius</italic> (<xref ref-type="bibr" rid="B24">Liimata et al., 2022</xref>), remain phylogenetically close, leading to similar ITS sequences and reduced resolution in k-mer-based embeddings. These factors elevated false positives and degraded the performance of the Cortinariaceae_f2s classifier. To facilitate detailed analysis of how misclassifications at higher levels (e.g., phylum) affect downstream accuracy, we generated heatmaps of classification results for key taxonomic levels during training, provided in <xref ref-type="sec" rid="s11">Supplementary Image S2</xref>.</p>
</sec>
<sec id="s3-3-2">
<title>3.3.2 Comprehensive validation of the HFTC</title>
<p>After validating the robust performance of individual sub-classifiers, we conducted a comprehensive evaluation of the integrated HFTC system using six metrics across five independent test sets. <xref ref-type="table" rid="T4">Table 4</xref> summarizes the performance of HFTC in the Test10 dataset, while detailed results for the other test sets are provided in <xref ref-type="sec" rid="s11">Supplementary Table S2</xref>.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Prediction performance for the HFTC in Test10.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Level</th>
<th align="center">ACC</th>
<th align="center">HA</th>
<th align="center">Recall</th>
<th align="center">Precision</th>
<th align="center">F<sub>1</sub>
</th>
<th align="center">MCC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Phylum</td>
<td align="center">0.9997</td>
<td align="center">0.9997</td>
<td align="center">0.9997</td>
<td align="center">0.9997</td>
<td align="center">0.9997</td>
<td align="center">0.9994</td>
</tr>
<tr>
<td align="center">Class</td>
<td align="center">0.9943</td>
<td align="center">0.9943</td>
<td align="center">0.9967</td>
<td align="center">0.9943</td>
<td align="center">0.9954</td>
<td align="center">0.9920</td>
</tr>
<tr>
<td align="center">Order</td>
<td align="center">0.9747</td>
<td align="center">0.9744</td>
<td align="center">0.9956</td>
<td align="center">0.9747</td>
<td align="center">0.9845</td>
<td align="center">0.9724</td>
</tr>
<tr>
<td align="center">Family</td>
<td align="center">0.9648</td>
<td align="center">0.9637</td>
<td align="center">0.9648</td>
<td align="center">0.9648</td>
<td align="center">0.9799</td>
<td align="center">0.9646</td>
</tr>
<tr>
<td align="center">Genus</td>
<td align="center">0.9525</td>
<td align="center">0.9510</td>
<td align="center">0.9508</td>
<td align="center">0.9525</td>
<td align="center">0.9544</td>
<td align="center">0.9531</td>
</tr>
<tr>
<td align="center">Species</td>
<td align="center">0.9525</td>
<td align="center">0.9510</td>
<td align="center">0.9508</td>
<td align="center">0.9525</td>
<td align="center">0.9544</td>
<td align="center">0.9531</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Because HFTC achieves direct species-level classification from the family level, bypassing the intermediate genus level, the evaluation metrics at both the genus and species levels remained consistent, each exceeding 95%. Similar performance trends were observed across the remaining four independent test sets, with species-level accuracies of 94.7%, 94.1%, 93.5%, and 93.3%, respectively. In addition, in addressing class imbalance, the MCC surpassed 99.9% at the phylum level, 99.2% at the class level, 97.2% at the order level, 96.5% at the family level, and remained as high as 95.3% at the genus and species levels. These results fully demonstrate that HFTC can maintain robust performance even in the context of highly imbalanced classification tasks involving over 25,000 species. The high consistency between MCC and ACC further validates the model&#x2019;s strong adaptability to imbalanced datasets.</p>
</sec>
</sec>
<sec id="s3-4">
<title>3.4 Comparison with existing predictors</title>
<p>Considering representativeness and availability, five methods, namely, Mothur (<xref ref-type="bibr" rid="B37">Schloss et al., 2009</xref>), RDP (<xref ref-type="bibr" rid="B45">Wang et al., 2007</xref>), Sintax (<xref ref-type="bibr" rid="B9">Edgar, 2016</xref>), QIIME2 (<xref ref-type="bibr" rid="B24">Liimata et al., 2022</xref>), and CNN-Duong (<xref ref-type="bibr" rid="B44">Vu et al., 2020</xref>), were selected for model performance comparison on an independent and identical test dataset Test10. Mothur, Sintax, RDP, and QIIME2 use traditional machine learning, and CNN-Duong uses neural networks. In the experiments, Mothur, RDP, and Sintax were all executed on an AMD Ryzen Threadripper PRO 5955WX 16-Core Processor, using their recommended default parameters: a k-mer size of 8 and a confidence threshold of 0.9 for Sintax. QIIME2 was evaluated under the same environment with a k-mer size of 7 and a confidence threshold of 0.7, as recommended. Similarly, CNN-Duong was executed on GeForce RTX 4090 and evaluated using its default recommended settings, including a k-mer size of 6 and a CNN architecture, comprising two Conv1D layers and two dense layers, with a total of 49.83&#xa0;MB of parameters.</p>
<p>
<xref ref-type="fig" rid="F7">Figure 7</xref> shows that HFTC achieved superior performance across all metrics, with an ACC of 95.25%, an HA of 95.09%, a precision of 95.08%, a recall of 95.25%, an F1-score of 95.80%, and an MCC of 95.31%. The CNN-Duong showed a slightly higher ACC of 95.43% but showed a marked decrease in HA to 91.90%, suggesting potential limitations in handling taxonomic hierarchies. QIIME2 ranked third overall, with both ACC and related metrics at approximately 94.44% and an HA of 93.21%. Notably, QIIME2, Sintax, and RDP are all Na&#xef;ve Bayes-based methods; consistent with our earlier findings that <italic>k</italic> &#x3d; 7 is optimal for species-level classification, QIIME2 (<italic>k</italic> &#x3d; 7) outperformed RDP and Sintax (<italic>k</italic> &#x3d; 8), which both achieved approximately 92% metrics. In addition, Mothur exhibited the poorest performance across all six evaluation metrics, showing an overall accuracy, hierarchical accuracy, precision, recall, F1 score, and MCC of 71.15%, 70.98%, 78.02%, 71.15%, 73.01%, and 74.28%, respectively.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Comparison of six models at the species level in Test10. Benchmarking performance of HFTC against five state-of-the-art classifiers at the species level with six metrics: ACC, HA, precision, recall, F<sub>1</sub>-score, and MCC.</p>
</caption>
<graphic xlink:href="fgene-16-1650244-g007.tif">
<alt-text content-type="machine-generated">Bar chart comparing classification metrics across six methods: HFTC, Mothur, RDP, SINTAX, QIIME2, and CNN-Duong, for ACC, HA, Precision, Recall, F1-Score, and MCC. HFTC consistently scores highest across all metrics, with Mothur scoring lowest.</alt-text>
</graphic>
</fig>
<p>In particular, <xref ref-type="table" rid="T5">Table 5</xref> compares six models in terms of HA, species-level accuracy (ACC), feature vector size, and per-sequence inference time, which reflects the computational efficiency gained through dimensionality reduction. HFTC achieves the highest HA of 95.09% and a competitive ACC of 95.25%, only 0.18% lower than that of the best-performing CNN-Duong model. However, HFTC exhibits a substantial advantage in computational efficiency: its feature vector size is only 200, which is less than 0.3% of the 65,536-dimensional vectors used in traditional k-mer frequency-based models (Mothur, RDP, and Sintax), 1.2% of QIIME2&#x2019;s 16,384, and only 4.8% of CNN-Duong&#x2019;s 4,096. This low-dimensional embedding leads to significantly faster inference. HFTC achieves the fastest inference time among all six models, requiring only 0.37 milliseconds per sequence. It is 35% faster than the second-fastest model, Sintax (0.57&#xa0;ms), and significantly outperforms the others&#x2014;Mothur (0.92&#xa0;ms), CNN-Duong (2.02&#xa0;ms), RDP (12.25&#xa0;ms), and QIIME2 (58.20&#xa0;ms). Notably, HFTC achieves high speed without sacrificing accuracy, showing a good balance between efficiency and performance.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>In-depth comparison of six classifiers on the test10 dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Method</th>
<th align="center">Algorithm</th>
<th align="center">ACC</th>
<th align="center">HA</th>
<th align="center">ACC-HA</th>
<th align="center">Vector size</th>
<th align="center">Inference time (ms)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">HFTC</td>
<td align="center">RF</td>
<td align="center">0.9525</td>
<td align="center">
<bold>0.9509</bold>
</td>
<td align="center">
<bold>1.60%</bold>
</td>
<td align="center">
<bold>200</bold>
</td>
<td align="center">
<bold>0.37</bold>
</td>
</tr>
<tr>
<td align="center">Mothur</td>
<td align="center">KNN</td>
<td align="center">0.7115</td>
<td align="center">0.7098</td>
<td align="center">1.73<bold>%</bold>
</td>
<td align="center">65,536</td>
<td align="center">0.92</td>
</tr>
<tr>
<td align="center">RDP</td>
<td align="center">NB</td>
<td align="center">0.9263</td>
<td align="center">0.9125</td>
<td align="center">13.73<bold>%</bold>
</td>
<td align="center">65,536</td>
<td align="center">12.25</td>
</tr>
<tr>
<td align="center">Sintax</td>
<td align="center">NB</td>
<td align="center">0.9221</td>
<td align="center">0.9048</td>
<td align="center">17.30<bold>%</bold>
</td>
<td align="center">65,536</td>
<td align="center">0.57</td>
</tr>
<tr>
<td align="center">QIIME</td>
<td align="center">NB</td>
<td align="center">0.9444</td>
<td align="center">0.9321</td>
<td align="center">12.36<bold>%</bold>
</td>
<td align="center">16,384</td>
<td align="center">58.50</td>
</tr>
<tr>
<td align="center">CNN-Duong</td>
<td align="center">CNN</td>
<td align="center">
<bold>0.9543</bold>
</td>
<td align="center">0.9193</td>
<td align="center">35.00<bold>%</bold>
</td>
<td align="center">4,096</td>
<td align="center">2.02</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold numbers indicate column-wise optimal values (highest ACC/HA, smallest ACC&#x2010;HA, vector size, and inference time).</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Importantly, HFTC achieves the smallest gap between ACC and HA, with a difference of only 1.60%, indicating its strong ability to produce hierarchically consistent predictions. In contrast, although CNN-Duong attains a slightly higher accuracy of 95.43% at the species level, its HA decreases significantly to 91.93%, resulting in the largest ACC&#x2013;HA gap of 35.00<bold>%</bold>. This performance gap reflects a fundamental limitation of CNN-based models with flat architectures, which do not explicitly capture taxonomic dependencies and are, therefore, more prone to inconsistent predictions across hierarchical levels. In contrast, HFTC uses a hierarchical progressive classifier selection mechanism that underpins its superior hierarchical consistency. Theoretically, this design also carries the potential risk of &#x201c;higher-level misclassifications confining lower-level classifiers to incorrect branches.&#x201d; However, HFTC mitigates this risk through two key features:</p>
<p>First, higher-level classifications in HFTC rely on highly discriminative inter-group features and undergo specialized training, minimizing misclassification rates and reducing the occurrence of incorrect branch guidance. Second, unlike flat models that use a single model to predict all taxonomic ranks simultaneously and yield inconsistent predictions, HFTC decomposes the task into sequential subtasks aligned with the biological taxonomic hierarchy. This inherent structural advantage explains why HFTC maintains such a narrow ACC&#x2013;HA gap, outperforming flat architectures in preserving taxonomic integrity across all levels.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>We presented HFTC, a bioinformatics tool for fungal species-level classification. To support a broad spectrum of fungal taxa, we constructed an ITS reference dataset encompassing over 25,000 species and 4.4 million sequences. HFTC incorporates three key innovations: a bi-kmer comprehensive feature extraction strategy that captures sequence context from both forward and reverse orientations (<xref ref-type="bibr" rid="B39">Shoemaker et al., 2017</xref>); Word2Vec embedding to compress high-dimensional KFV from 4<sup>
<italic>k</italic>
</sup> into only 200-dimensional vectors, balancing computational efficiency with contextual information preservation (<xref ref-type="bibr" rid="B23">Lennon and Locey, 2020</xref>); and a hierarchical classification framework that ensures taxonomic consistency across levels (<xref ref-type="bibr" rid="B15">Hawksworth and L&#xfc;cking, 2017</xref>). These innovations enable HFTC to achieve state-of-the-art performance at the species level, with 95.25% for ACC, 95.31% for MCC, and 95.10% for HA, while maintaining the smallest discrepancy between ACC and HA of 1.60<bold>%</bold>. HFTC shows excellent performance on fungal ITS data, and its architecture is generalizable to other barcoding systems such as 16S rRNA, with appropriate retraining and k-mer optimization.</p>
<p>Despite these strengths, HFTC has limitations that warrant future refinement. First, its performance is partially dependent on the quality of the reference database, highlighting the importance of accurate and comprehensive annotations such as those provided by UNITE. Second, although HFTC demonstrates scalability on large datasets, further optimization may be needed to efficiently handle ultra-large-scale sequencing data in future applications. Third, the k-mer pooling step in HFTC loses positional sequence information, which could limit resolution for closely related taxa with subtle structural variations. One promising direction is to leverage pre-trained language models (PLMs), such as DNABERT (<xref ref-type="bibr" rid="B17">Ji et al., 2021</xref>; <xref ref-type="bibr" rid="B8">Devlin et al., 2019</xref>), to complement HFTC&#x2019;s current framework. These models could provide two key advantages: their ability to generate rich contextual embeddings from large-scale nucleotide corpora may help capture patterns from rare species that HFTC might overlook (<xref ref-type="bibr" rid="B39">Shoemaker et al., 2017</xref>), and their self-attention mechanisms and positional encoding capabilities could help recover the positional information lost during HFTC&#x2019;s average pooling step, thereby enabling better use of local k-mer-based signals (<xref ref-type="bibr" rid="B23">Lennon and Locey, 2020</xref>). Although our study focused on the non-coding ITS region, the same principle could extend to coding markers such as RPB2 or TEF, which also contain taxonomically informative sequence motifs. Additionally, model compression techniques such as knowledge distillation could help reduce computational costs while maintaining accuracy. With these advancements, HFTC could further solidify its role as a cornerstone for scalable and accurate taxonomic identification in microbial research. All source code, datasets, and instructions for the experiments are publicly available at <ext-link ext-link-type="uri" xlink:href="https://github.com/wjjw0731/HFTC/tree/master">https://github.com/wjjw0731/HFTC/tree/master</ext-link>. The repository is documented and fully reproducible.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found at: Zenodo home: <ext-link ext-link-type="uri" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://zenodo.org/uploads/14826761">https://zenodo.org/uploads/14826761</ext-link> DOI:10.5281/zenodo.14826761.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>JW: Software, Conceptualization, Writing &#x2013; original draft, Methodology, Visualization, Validation. SQ: Project administration, Validation, Methodology, Conceptualization, Investigation, Writing &#x2013; review and editing. DX: Visualization, Validation, Software, Writing &#x2013; original draft. YL: Writing &#x2013; original draft, Visualization, Data curation, Validation. CW: Writing &#x2013; review and editing, Resources, Writing &#x2013; original draft, Data curation, Supervision, Software, Funding acquisition, Validation.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported by the National Natural Science Foundation of China (<ext-link ext-link-type="uri" xlink:href="https://www.nsfc.gov.cn/">https://www.nsfc.gov.cn/</ext-link>) under grants nos. 62272065 (to C.W.) and 62002051 (to C.W.) and the Guangxi Natural Science Foundation (<ext-link ext-link-type="uri" xlink:href="http://kjt.gxzf.gov.cn/">http://kjt.gxzf.gov.cn/</ext-link>) under grant no. 2024GXNSFBA010372 (to C.W.).</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2025.1650244/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2025.1650244/full&#x23;supplementary-material</ext-link>.</p>
<supplementary-material xlink:href="Table2.xlsx" id="SM1" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image2.jpg" id="SM2" mimetype="application/jpg" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table1.xlsx" id="SM3" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image1.jpg" id="SM4" mimetype="application/jpg" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altschul</surname>
<given-names>S. F.</given-names>
</name>
<name>
<surname>Gish</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Myers</surname>
<given-names>E. W.</given-names>
</name>
<name>
<surname>Lipman</surname>
<given-names>DJJJomb</given-names>
</name>
</person-group> (<year>1990</year>). <article-title>Basic local alignment search tool</article-title>. <source>J. Mol. Biol.</source> <volume>215</volume> (<issue>3</issue>), <fpage>403</fpage>&#x2013;<lpage>410</lpage>. <pub-id pub-id-type="doi">10.1016/S0022-2836(05)80360-2</pub-id>
<pub-id pub-id-type="pmid">2231712</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Asim</surname>
<given-names>M. N.</given-names>
</name>
<name>
<surname>Malik</surname>
<given-names>M. I.</given-names>
</name>
<name>
<surname>Dengel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ahmed</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>K-mer neural embedding performance analysis using amino acid codons</article-title>,&#x201d; in <source>2020 International Joint Conference on Neural Networks (IJCNN)</source>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Borozan</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Watt</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ferretti</surname>
<given-names>V. J. B.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Integrating alignment-based and alignment-free sequence similarity measures for biological sequence classification</article-title>. <volume>31</volume>(<issue>9</issue>):<fpage>1396</fpage>&#x2013;<lpage>1404</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv006</pub-id>
<pub-id pub-id-type="pmid">25573913</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boughorbel</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jarray</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>El-Anbari</surname>
<given-names>M. J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Optimal classifier for imbalanced data using matthews correlation coefficient metric</article-title>. <source>PLoS One</source> <volume>12</volume> (<issue>6</issue>), <fpage>e0177678</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0177678</pub-id>
<pub-id pub-id-type="pmid">28574989</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bradshaw</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Aime</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Rokas</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Maust</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Moparthi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jellings</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Extensive intragenomic variation in the internal transcribed spacer region of fungi</article-title>. <source>iScience</source> <volume>26</volume> (<issue>8</issue>), <fpage>107317</fpage>. <pub-id pub-id-type="doi">10.1016/j.isci.2023.107317</pub-id>
<pub-id pub-id-type="pmid">37529098</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breiman</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2001</year>). <source>Random forests. Machine Learning</source> <volume>45</volume> (<issue>1</issue>), <fpage>5</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chiu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Crichton</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Korhonen</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pyysalo</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>How to train good word embeddings for biomedical NLP</article-title>,&#x201d; in <source>Proceedings of the 15th workshop on biomedical natural language processing</source>.</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deshpande</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Greenfield</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Charleston</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Porras-Alfaro</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kuske</surname>
<given-names>C. R.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Fungal identification using a Bayesian classifier and the warcup training set of internal transcribed spacer sequences</article-title>. <source>Mycologia</source> <volume>108</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.3852/14-293</pub-id>
<pub-id pub-id-type="pmid">26553774</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Devlin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>M.-W.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Toutanova</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Bert: pre-training of deep bidirectional transformers for language understanding</article-title>,&#x201d; in <source>Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers)</source>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Edgar</surname>
<given-names>R. C. J. B</given-names>
</name>
</person-group>. (<year>2016</year>). <article-title>SINTAX: a simple Non-Bayesian taxonomy classifier for 16S and ITS sequences</article-title>. <fpage>074161</fpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fern&#xe1;ndez-Delgado</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cernadas</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Barro</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Djtjomlr</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Do we need hundreds of classifiers to solve real world classification problems?</article-title>. <volume>15</volume> (<issue>1</issue>):<fpage>3133</fpage>&#x2013;<lpage>3181</lpage>.<pub-id pub-id-type="pmid">24815459</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ficetola</surname>
<given-names>G. F.</given-names>
</name>
<name>
<surname>Coissac</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Zundel</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Riaz</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Shehzad</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Bessi&#xe8;re</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>An <italic>in silico</italic> approach for the evaluation of DNA barcodes</article-title>. <source>BMC Genomics</source> <volume>11</volume> (<issue>1</issue>), <fpage>434</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-11-434</pub-id>
<pub-id pub-id-type="pmid">20637073</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lou</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Analysis of whole-genome as a novel strategy for animal species identification</article-title>. <source>Int. J. Mol. Sci.</source> <volume>25</volume> (<issue>5</issue>), <fpage>2955</fpage>. <pub-id pub-id-type="doi">10.3390/ijms25052955</pub-id>
<pub-id pub-id-type="pmid">38474203</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gmjpo</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Challenges in the real world use of classification accuracy metrics: from recall and precision to the matthews correlation coefficient</article-title>. <source>PLoS ONE</source> <volume>18</volume> (<issue>10</issue>), <fpage>e0291908</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0291908</pub-id>
<pub-id pub-id-type="pmid">37792898</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hawksworth</surname>
<given-names>D. L.</given-names>
</name>
<name>
<surname>L&#xfc;cking</surname>
<given-names>R. J. M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Fungal diversity revisited: 2.2 to 3.8 million species</article-title>. <source>Microbiol. Spectr.</source> <volume>5</volume> (<issue>4</issue>), <fpage>5.4.10</fpage>. <pub-id pub-id-type="doi">10.1128/microbiolspec.funk-0052-2016</pub-id>
<pub-id pub-id-type="pmid">28752818</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jenike</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Campos-Dom&#xed;nguez</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Bodd&#xe9;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cerca</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hodson</surname>
<given-names>C. N.</given-names>
</name>
<name>
<surname>Schatz</surname>
<given-names>M. C.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Guide to k-mer approaches for genomics across the tree of life</article-title>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ji</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Davuluri</surname>
<given-names>R. V.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>DNABERT: pre-trained Bidirectional encoder representations from transformers model for DNA-language in genome</article-title>. <source>Bioinformatics</source> <volume>37</volume> (<issue>15</issue>), <fpage>2112</fpage>&#x2013;<lpage>2120</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab083</pub-id>
<pub-id pub-id-type="pmid">33538820</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ji</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Shang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y. J. B.</given-names>
</name>
</person-group> <article-title>HOTSPOT: hierarchical host prediction for assembled plasmid contigs with transformer</article-title> (<year>2023</year>). <volume>39</volume>(<issue>5</issue>):<fpage>btad283</fpage>, <pub-id pub-id-type="doi">10.1093/bioinformatics/btad283</pub-id>
<pub-id pub-id-type="pmid">37086432</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karsch-Mizrachi</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Takagi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Cochrane</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Insdcjna</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>The international nucleotide sequence database collaboration</article-title>. <source>Nucleic Acids Res.</source> <volume>46</volume> (<issue>D1</issue>), <fpage>D48</fpage>&#x2013;<lpage>D51</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkx1097</pub-id>
<pub-id pub-id-type="pmid">29190397</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>K&#xf5;ljalg</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Nilsson</surname>
<given-names>R. H.</given-names>
</name>
<name>
<surname>Abarenkov</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Tedersoo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Taylor</surname>
<given-names>A. F.</given-names>
</name>
<name>
<surname>Bahram</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <source>Towards a unified paradigm for sequence&#x2010;based identification of fungi</source>. <publisher-name>Wiley Online Library</publisher-name>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>K&#xf5;ljalg</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Nilsson</surname>
<given-names>H. R.</given-names>
</name>
<name>
<surname>Schigel</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Tedersoo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Larsson</surname>
<given-names>K.-H.</given-names>
</name>
<name>
<surname>May</surname>
<given-names>T. W.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>The taxon hypothesis paradigm&#x2014;on the unambiguous detection and communication of taxa</article-title>. <source>Microorganisms</source> <volume>8</volume> (<issue>12</issue>), <fpage>1910</fpage>. <pub-id pub-id-type="doi">10.3390/microorganisms8121910</pub-id>
<pub-id pub-id-type="pmid">33266327</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Le Guillarme</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Thuiller</surname>
<given-names>W. J. M. E.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>TaxoNERD: deep neural models for the recognition of taxonomic entities in the ecological and evolutionary literature</article-title>. <source>Methods Ecol. Evol.</source> <volume>13</volume> (<issue>3</issue>), <fpage>625</fpage>&#x2013;<lpage>641</lpage>. <pub-id pub-id-type="doi">10.1111/2041-210x.13778</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lennon</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Locey</surname>
<given-names>KJJBD</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>More support for Earth&#x2019;s massive microbiome</article-title>. <source>Biol. Direct</source> <volume>15</volume> (<issue>1</issue>), <fpage>5</fpage>. <pub-id pub-id-type="doi">10.1186/s13062-020-00261-8</pub-id>
<pub-id pub-id-type="pmid">32131875</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liimatainen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Pokorny</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Kirk</surname>
<given-names>P. M.</given-names>
</name>
<name>
<surname>Dentinger</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Niskanen</surname>
<given-names>TJFD</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Taming the beast: a revised classification of cortinariaceae based on genomic data</article-title>. <source>Fungal Divers.</source> <volume>112</volume> (<issue>1</issue>), <fpage>89</fpage>&#x2013;<lpage>170</lpage>. <pub-id pub-id-type="doi">10.1007/s13225-022-00499-9</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lindahl</surname>
<given-names>B. D.</given-names>
</name>
<name>
<surname>Nilsson</surname>
<given-names>R. H.</given-names>
</name>
<name>
<surname>Tedersoo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Abarenkov</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Carlsen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kj&#xf8;ller</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Fungal community analysis by high&#x2010;throughput sequencing of amplified markers&#x2013;a user&#x27;s guide</article-title>. <source>New Phytol.</source> <volume>199</volume> (<issue>1</issue>), <fpage>288</fpage>&#x2013;<lpage>299</lpage>. <pub-id pub-id-type="doi">10.1111/nph.12243</pub-id>
<pub-id pub-id-type="pmid">23534863</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>J. J. C.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>FungiLT: a deep learning approach for species-level taxonomic classification of fungal ITS sequences</article-title>. <source>Comput. (Basel).</source> <volume>14</volume> (<issue>3</issue>), <fpage>85</fpage>. <pub-id pub-id-type="doi">10.3390/computers14030085</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mahadani</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ghosh</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Utility of indels for species-level identification of a biologically complex plant group: a study with intergenic spacer in Citrus</article-title>. <source>Mol. Biol. Rep.</source> <volume>41</volume> (<issue>11</issue>), <fpage>7217</fpage>&#x2013;<lpage>7222</lpage>. <pub-id pub-id-type="doi">10.1007/s11033-014-3606-7</pub-id>
<pub-id pub-id-type="pmid">25048292</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mar&#xe7;ais</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Elder</surname>
<given-names>C. S.</given-names>
</name>
<name>
<surname>Kingsford</surname>
<given-names>C. J. B.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>k-nonical space: sketching with reverse complements</article-title>. <volume>40</volume> (<issue>11</issue>). <pub-id pub-id-type="doi">10.1093/bioinformatics/btae629</pub-id>
<pub-id pub-id-type="pmid">39432565</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Neelima</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mehrotra</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). in <source>A comprehensive review on word embedding techniques. 2023 International Conference on Intelligent Systems for Communication, IoT and Security (ICISCoIS)</source>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nilsson</surname>
<given-names>R. H.</given-names>
</name>
<name>
<surname>Kristiansson</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ryberg</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hallenberg</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Larsson</surname>
<given-names>K.-H. J. E.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Intraspecific ITS variability in the kingdom Fungi as expressed in the international sequence databases and its implications for molecular species identification</article-title>. <source>Evol. Bioinform. Online</source> <volume>4</volume>, <fpage>193</fpage>&#x2013;<lpage>201</lpage>. <pub-id pub-id-type="doi">10.4137/ebo.s653</pub-id>
<pub-id pub-id-type="pmid">19204817</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Nilsson</surname>
<given-names>R. H.</given-names>
</name>
<name>
<surname>Taylor</surname>
<given-names>A. F.</given-names>
</name>
<name>
<surname>Adams</surname>
<given-names>R. I.</given-names>
</name>
<name>
<surname>Baschien</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bengtsson-Palme</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cangren</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). &#x201c;<article-title>Taxonomic annotation of public fungal ITS sequences from the built environment - a report from an April 10-11, 2017 workshop (Aberdeen, UK)</article-title>,&#x201d;, <volume>28</volume>. <publisher-loc>Aberdeen, UK</publisher-loc>, <fpage>65</fpage>&#x2013;<lpage>82</lpage>. <pub-id pub-id-type="doi">10.3897/mycokeys.28.20887</pub-id>
<pub-id pub-id-type="pmid">29559822</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nilsson</surname>
<given-names>R. H.</given-names>
</name>
<name>
<surname>Larsson</surname>
<given-names>K.-H.</given-names>
</name>
<name>
<surname>Taylor</surname>
<given-names>A. F. S.</given-names>
</name>
<name>
<surname>Bengtsson-Palme</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jeppesen</surname>
<given-names>T. S.</given-names>
</name>
<name>
<surname>Schigel</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>The UNITE database for molecular identification of fungi: handling dark taxa and parallel taxonomic classifications</article-title>. <source>Nucleic Acids Res.</source> <volume>47</volume> (<issue>D1</issue>), <fpage>D259-D264</fpage>&#x2013;<lpage>D64</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky1022</pub-id>
<pub-id pub-id-type="pmid">30371820</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Raja</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>A. N.</given-names>
</name>
<name>
<surname>Pearce</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Oberlies</surname>
<given-names>N. H. J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Fungal identification using molecular tools: a primer for the natural products research community</article-title>. <source>J. Nat. Prod.</source> <volume>80</volume> (<issue>3</issue>), <fpage>756</fpage>&#x2013;<lpage>770</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jnatprod.6b01085</pub-id>
<pub-id pub-id-type="pmid">28199101</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ratnasingham</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hebert</surname>
<given-names>P. D. J.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Bold: the barcode of life data system (http://www.barcodinglife.org)</article-title>. <source>Mol. Ecol. Notes</source> <volume>7</volume> (<issue>3</issue>), <fpage>355</fpage>&#x2013;<lpage>364</lpage>. <pub-id pub-id-type="doi">10.1111/j.1471-8286.2007.01678.x</pub-id>
<pub-id pub-id-type="pmid">18784790</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saada</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Siga</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Magalh&#xe3;es</surname>
<given-names>M.MMJAS</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Whole-genome alignment: methods, challenges, and future directions</article-title>. <source>Appl. Sci. (Basel).</source> <volume>14</volume> (<issue>11</issue>), <fpage>4837</fpage>. <pub-id pub-id-type="doi">10.3390/app14114837</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schloss</surname>
<given-names>P. D.</given-names>
</name>
<name>
<surname>Westcott</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Ryabin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hall</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Hartmann</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hollister</surname>
<given-names>E. B.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Introducing mothur: open-source, platform-independent, community-supported software for describing and comparing microbial communities</article-title>. <source>Appl. Environ. Microbiol.</source> <volume>75</volume> (<issue>23</issue>), <fpage>7537</fpage>&#x2013;<lpage>7541</lpage>. <pub-id pub-id-type="doi">10.1128/AEM.01541-09</pub-id>
<pub-id pub-id-type="pmid">19801464</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schoch</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Seifert</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Huhndorf</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Robert</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Spouge</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Levesque</surname>
<given-names>C. A.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Nuclear ribosomal internal transcribed spacer (ITS) region as a universal DNA barcode marker for fungi</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>109</volume> (<issue>16</issue>), <fpage>6241</fpage>&#x2013;<lpage>6246</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1117018109</pub-id>
<pub-id pub-id-type="pmid">22454494</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shoemaker</surname>
<given-names>W. R.</given-names>
</name>
<name>
<surname>Locey</surname>
<given-names>K. J.</given-names>
</name>
<name>
<surname>Lennon</surname>
<given-names>JTJNe</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A macroecological theory of microbial biodiversity</article-title>. <source>Nat. Ecol. Evol.</source> <volume>1</volume> (<issue>5</issue>), <fpage>0107</fpage>. <pub-id pub-id-type="doi">10.1038/s41559-017-0107</pub-id>
<pub-id pub-id-type="pmid">28812691</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tedersoo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>S&#xe1;nchez-Ram&#xed;rez</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Koljalg</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Bahram</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>D&#xf6;ring</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schigel</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>High-level classification of the fungi and a tool for evolutionary ecological analyses</article-title>. <volume>90</volume>(<issue>1</issue>):<fpage>135</fpage>&#x2013;<lpage>159</lpage>.</citation>
</ref>
<ref id="B41">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Th</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Anand</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Evaluating distributed word representations for capturing semantics of biomedical concepts</article-title>,&#x201d; in <source>Proceedings of BioNLP</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Muneeb</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sahu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Anand</surname>
<given-names>A.</given-names>
</name>
</person-group> <fpage>158</fpage>&#x2013;<lpage>163</lpage>. <pub-id pub-id-type="doi">10.18653/v1/w15-3820</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tieppo</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Santos</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Barddal</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Nievola</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Hierarchical classification of data streams: a systematic literature review</article-title>. <source>Artif. Intell. Rev.</source> <volume>55</volume> (<issue>4</issue>), <fpage>3243</fpage>&#x2013;<lpage>3282</lpage>. <pub-id pub-id-type="doi">10.1007/s10462-021-10087-z</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<collab>UNITE Community</collab>. (<year>2023</year>). <source>Full UNITE&#x2b;INSD dataset for fungi version 18.07.2023</source>. <publisher-name>Tartu, Estonia: University of Tartu</publisher-name>. <pub-id pub-id-type="doi">10.15156/BIO/2938065</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>van Zyl</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Dunaiski</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tegally</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Baxter</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>de Oliveira</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Xavier</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Alignment-free viral sequence classification at scale</article-title>. <source>BMC Genomics</source> <volume>26</volume> (<issue>1</issue>), <fpage>389</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-025-11554-5</pub-id>
<pub-id pub-id-type="pmid">40251515</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Groenewald</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Verkley</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Convolutional neural networks improve fungal classification</article-title>. <source>Sci. Rep.</source> <volume>10</volume> (<issue>1</issue>), <fpage>12628</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-69245-y</pub-id>
<pub-id pub-id-type="pmid">32724224</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Garrity</surname>
<given-names>G. M.</given-names>
</name>
<name>
<surname>Tiedje</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Cole</surname>
<given-names>JRJA</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Naive Bayesian classifier for rapid assignment of rRNA sequences into the new bacterial taxonomy</article-title>. <source>Appl. Environ. Microbiol.</source> <volume>73</volume> (<issue>16</issue>), <fpage>5261</fpage>&#x2013;<lpage>5267</lpage>. <pub-id pub-id-type="doi">10.1128/AEM.00062-07</pub-id>
<pub-id pub-id-type="pmid">17586664</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>F. J. F.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Identifying group-specific sequences for microbial communities using long k-mer sequence signatures</article-title>. <source>Front. Microbiol.</source> <volume>9</volume>, <fpage>872</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2018.00872</pub-id>
<pub-id pub-id-type="pmid">29774017</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>S. J. B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Its2vec: fungal species identification using sequence embedding and random forest classification</article-title>. <source>Biomed. Res. Int.</source> <volume>2020</volume> (<issue>1</issue>), <fpage>2468789</fpage>. <pub-id pub-id-type="doi">10.1155/2020/2468789</pub-id>
<pub-id pub-id-type="pmid">32566672</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wichmann</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Buschong</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>M&#xfc;ller</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>J&#xfc;nger</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hildebrandt</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hankeln</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>MetaTransformer: deep metagenomic sequencing read classification using self-attention models</article-title>. <source>Nar. Genom. Bioinform.</source> <volume>5</volume> (<issue>3</issue>), <fpage>lqad082</fpage>. <pub-id pub-id-type="doi">10.1093/nargab/lqad082</pub-id>
<pub-id pub-id-type="pmid">37705831</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>M.-L.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Z.-H. J. I.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>A review on multi-label learning algorithms</article-title>. <source>IEEE Trans. Knowl. Data Eng.</source> <volume>26</volume> (<issue>8</issue>), <fpage>1819</fpage>&#x2013;<lpage>1837</lpage>. <pub-id pub-id-type="doi">10.1109/tkde.2013.39</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>HJIJ. M. S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Phylogenetic utility of rRNA ITS2 sequence-structure under functional constraint</article-title>. <source>Int. J. Mol. Sci.</source> <volume>21</volume> (<issue>17</issue>), <fpage>6395</fpage>. <pub-id pub-id-type="doi">10.3390/ijms21176395</pub-id>
<pub-id pub-id-type="pmid">32899108</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hooi</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>J. J. I.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Deep long-tailed learning: a survey</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>45</volume> (<issue>9</issue>), <fpage>10795</fpage>&#x2013;<lpage>10816</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2023.3268118</pub-id>
<pub-id pub-id-type="pmid">37074896</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Y.-H.</given-names>
</name>
<name>
<surname>Pjfig</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A review and tutorial of machine learning methods for microbiome host trait prediction</article-title>. <volume>10</volume>:<fpage>579</fpage>.</citation>
</ref>
</ref-list>
</back>
</article>