<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Bacteriol.</journal-id>
<journal-title>Frontiers in Bacteriology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Bacteriol.</abbrev-journal-title>
<issn pub-type="epub">2813-6144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fbrio.2025.1620906</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Bacteriology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>DL-based organism-level microbial identification via VOCs fingerprints through gas chromatography &#x2013; ion mobility spectrometry</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Kirtsanis</surname>
<given-names>Georgios</given-names>
</name>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3045231/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dolias</surname>
<given-names>Georgios</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/3126733/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Kintzios</surname>
<given-names>Spyridon</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ioannidis</surname>
<given-names>Konstantinos</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/1212329/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Vrochidis</surname>
<given-names>Stefanos</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/1377539/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Kompatsiaris</surname>
<given-names>Ioannis</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/413426/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<institution>Information Technologies Institute, Centre for Research and Technology Hellas</institution>, <addr-line>Thessaloniki</addr-line>,&#xa0;<country>Greece</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1408245/overview">Sohinee Sarkar</ext-link>, Royal Children&#x2019;s Hospital, Australia</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/434213/overview">Apichai Tuanyok</ext-link>, University of Florida, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3067870/overview">Marcelo Gonzalez</ext-link>, Federico Santa Mar&#xed;a Technical University, Chile</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Georgios Kirtsanis, <email xlink:href="mailto:gkirtsanis@iti.gr">gkirtsanis@iti.gr</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>25</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>4</volume>
<elocation-id>1620906</elocation-id>
<history>
<date date-type="received">
<day>30</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>31</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Kirtsanis, Dolias, Kintzios, Ioannidis, Vrochidis and Kompatsiaris.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Kirtsanis, Dolias, Kintzios, Ioannidis, Vrochidis and Kompatsiaris</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Organism-level microbial identification is a well-established topic in literature. Due to biosafety concerns, specifically identifying pathogenic bacteria is of critical importance. This study positions Deep Learning (DL) - based chemometric analysis as a promising strategy for organism-level microbial identification, with potential translational value for rapid diagnostics. Various chemometric methods have been applied to analyze pure and mixed cultures of microorganisms and generate data via Volatile Organic Compounds (VOCs) fingerprints for classification. Although Gas Chromatography - Ion Mobility Spectrometry (GC-IMS) is a promising chemometric technique in this field, limited research has explored its potential for organism-level microbial identification. </p>
</sec>
<sec>
<title>Materials and methods</title>
<p>In this study, GC-IMS prototypes were employed to generate two-dimensional spectral data, which were then used to train supervised classification models. Utilizing a publicly available dataset of four microorganisms, we conduct a series of experiments to perform multi-class classification of pure and mixed cultures. Additionally, we introduce innovative experiments for distinguishing bacteria from fungi and Gram-positive from Gram-negative bacteria. We further investigate the presence and pureness of two pathogenic bacteria, <italic>Escherichia coli</italic> and <italic>Pseudomonas fluorescens</italic>, within the cultures. To achieve this, we apply eight Machine Learning and DL baseline methods, while following a five-fold cross-validation evaluation protocol and presenting a wide set of evaluation metrics to ensure result reproducibility and models&#x2019; generalization. A further evaluation of DL models is also conducted to report the training times and the number of parameters of the proposed DL methods.</p>
</sec>
<sec>
<title>Results</title>
<p>Our key findings highlight a Fully Connected Neural Network (FCNN) with four hidden layers as the most efficient model, consistently achieving the best performance across all tasks in comparison to the other tested models of this study. Additionally, the FCNN model provides fast training and maintains a relatively small number of parameters compared to other DL approaches. </p>
</sec>
<sec>
<title>Discussion</title>
<p>While the dataset&#x2019;s limited size and class imbalance present challenges such as potential overfitting and optimistic bias, the results achieved so far are encouraging and demonstrate the model&#x2019;s strong potential. Future work should aim to expand the dataset across multiple sites and instruments and include clinical validation on real-world samples to further enhance generalizability and ensure translational impact.</p>
</sec>
</abstract>
<kwd-group>
<kwd>biosafety</kwd>
<kwd>organism-level microbial identification</kwd>
<kwd>volatile organic compounds (VOCs)</kwd>
<kwd>chemometrics</kwd>
<kwd>gas chromatography &#x2013; ion mobility spectrometry (GC-IMS)</kwd>
<kwd>Deep Learning (DL)</kwd>
<kwd>machine learning (ML)</kwd>
</kwd-group>
<contract-num rid="cn001">Grant Agreement 101103176</contract-num>
<contract-sponsor id="cn001">European Commission<named-content content-type="fundref-id">10.13039/501100000780</named-content>
</contract-sponsor>
<counts>
<fig-count count="3"/>
<table-count count="12"/>
<equation-count count="7"/>
<ref-count count="47"/>
<page-count count="14"/>
<word-count count="8125"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Bacterial Genetics and AI-enhanced Microbial Engineering</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Bacteria are microscopic, single-celled microorganisms lacking a nuclear membrane (<xref ref-type="bibr" rid="B3">Baron, 1996</xref>). They are metabolically active and divide through binary fission. Despite their seemingly simple structure, bacteria are highly advanced and adaptable organisms capable of causing a wide range of diseases. Pathogenic bacteria, in particular, are associated with specific illnesses such as the plague (<xref ref-type="bibr" rid="B14">Feng et&#xa0;al., 2021</xref>). The diagnosis of bacterial infections and the efficient treatment of infectious diseases are critical for human health (<xref ref-type="bibr" rid="B45">Yang et&#xa0;al., 2024</xref>). Additionally, to minimize the risk of contamination and toxicity, bacterial detection plays a vital role in the quality control of food products such as yogurt, cheese, and beer, as well as in the monitoring of bacteria in crops and silage (<xref ref-type="bibr" rid="B34">Sauer and Kliem, 2010</xref>). A wide range of laboratory techniques can be employed for the taxonomic classification and identification of bacteria (<xref ref-type="bibr" rid="B5">Chauhan et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B47">Zukowska, 2021</xref>). Gram-positive and Gram-negative bacteria possess different cell wall structures, influencing their susceptibility to antibiotics. Consequently, determining the Gram type of bacteria is essential for selecting the most effective antibiotic treatment (<xref ref-type="bibr" rid="B33">Rezaei et&#xa0;al., 2024</xref>). Furthermore, different pathogens necessitate distinct management strategies; for instance, bacterial infections may require immediate antibiotic intervention, whereas fungal infections might need prolonged antifungal therapy (<xref ref-type="bibr" rid="B17">Giuliano et al., 2019</xref>).</p>
<p>Certain strains of <italic>Escherichia coli</italic>, such as <italic>E. coli O157:H7</italic>, are known to cause severe foodborne illnesses, including diarrhea, urinary tract infections, and kidney failure (<xref ref-type="bibr" rid="B44">Yang et&#xa0;al., 2017</xref>), while <italic>E. coli O157:47</italic> is described as a category B biological warfare agent by the Centers for Disease Control and Prevention (CDC; Atlanta, GA, USA) (<xref ref-type="bibr" rid="B30">Pohanka, 2019</xref>), marking its accurate detection as an important concept in the literature. Although generally considered of low clinical significance, <italic>Pseudomonas fluorescens</italic> can either cause opportunistic infections, particularly in immunocompromised patients, including those with advanced cancer (<xref ref-type="bibr" rid="B20">Ishii et&#xa0;al., 2024</xref>) or have a significant impact in agriculture as a major food contaminant (<xref ref-type="bibr" rid="B29">Nunes et&#xa0;al., 2024</xref>). Most of these infections have been bloodstream infections, with few reports of pneumonia. Another bacterium, <italic>Levilactobacillus brevis</italic>, is commonly utilized in the fermentation of foods such as sauerkraut, kimchi, and pickles (<xref ref-type="bibr" rid="B21">Jeon et&#xa0;al., 2024</xref>). Detecting this bacterium ensures the quality and consistency of these fermented products. Regarding fungi, <italic>Saccharomyces cerevisiae</italic> is employed as a probiotic to prevent and treat various gastrointestinal diseases, such as antibiotic-associated diarrhea (<xref ref-type="bibr" rid="B27">Li et&#xa0;al., 2024</xref>). Detecting this yeast in probiotic products ensures they contain the intended beneficial strains. Concretely, the early detection and classification of these bacteria are essential for preventing outbreaks and mitigating biological threats.</p>
<p>Bacteria have been identified using various analytical methods, including molecular biology techniques such as polymerase chain reaction (PCR) for microorganisms, and immunological techniques such as enzyme-linked immunosorbent assays (ELISAs) for both protein toxins and microorganisms (<xref ref-type="bibr" rid="B1">Aboutalebian et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B23">Kim and Kim, 2021</xref>). While these methods are valuable for rapid screening of samples, they possess analytical limitations, including a lack of specificity, which can result in false positives due to cross-reactions with similar molecules. Furthermore, these methods are not suitable for the classification of unknown microbial samples (<xref ref-type="bibr" rid="B11">Duriez et&#xa0;al., 2016</xref>).</p>
<p>Conversely, mass spectrometry (MS) facilitates the unambiguous detection of microorganisms and protein toxins (<xref ref-type="bibr" rid="B11">Duriez et&#xa0;al., 2016</xref>). MS integrates speed, sensitivity, and specificity within a single technique, making it suitable for both targeted and untargeted detection of microorganisms, even in complex samples such as air, water, culture media, bodily fluids, and food (<xref ref-type="bibr" rid="B36">Tait et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B2">Altaee et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B19">Hameed et&#xa0;al., 2018</xref>). MS-based methodologies for identifying microorganisms through Volatile Organic Compounds (VOCs) fingerprints and toxins have continuously advanced with the development of soft ionization techniques, including matrix-assisted laser desorption/ionization (MALDI) and electrospray ionization (ESI), as well as high-resolution and high-mass-accuracy instruments (<xref ref-type="bibr" rid="B12">Dybwad, 2013</xref>; <xref ref-type="bibr" rid="B35">Su et&#xa0;al., 2022</xref>). These advancements have significantly enhanced the capability to accurately identify microorganisms and toxins, thereby ensuring biosafety across various contexts. Matrix-assisted laser desorption/ionization time of-flight mass spectrometry (MALDI-TOF MS) was one of the pioneering approaches for environmental applications. It is now recognized as a rapid, efficient, and reproducible method for species-specific identification of pathogenic microorganisms through the direct analysis of intact bacterial cells (<xref ref-type="bibr" rid="B8">Clark et&#xa0;al., 2013</xref>; <xref ref-type="bibr" rid="B9">Dingle and Butler-Wu, 2013</xref>). Specifically, MALDI-TOF MS, combined with advanced chemometric methods such as unsupervised clustering and classification using Artificial Neural Networks (ANN), has been employed to achieve rapid and reliable identification of bacteria from the genus Yersinia. MS-based methods have also been effectively applied to the identification and specific detection of biological agents by analyzing intact proteins and/or tryptic digests from bacterial cells (<xref ref-type="bibr" rid="B26">Lasch et&#xa0;al., 2010</xref>). However, under these conditions, MALDI-TOF analysis may lack sensitivity and is typically performed following a preliminary bacterial cultivation step, which serves both as a separation and enrichment tool. Additionally, various studies have investigated the capability of Orbitrap-MS prototypes to identify specific pathogenic or non-pathogenic bacterial cultures (<xref ref-type="bibr" rid="B42">Wynne et&#xa0;al., 2010</xref>; <xref ref-type="bibr" rid="B15">Gallien et&#xa0;al., 2012</xref>; <xref ref-type="bibr" rid="B38">Wang et&#xa0;al., 2023</xref>).</p>
<p>On the other hand, Gas Chromatography-Ion Mobility Spectrometry (GC-IMS) is an innovative method that leverages the high separation capacity of GC and the rapid response of IMS (<xref ref-type="bibr" rid="B39">Wang et&#xa0;al., 2020</xref>). The GC-IMS prototype generates two-dimensional data based on the drift times of ions and their retention times. The presence of specific ions in the culture results in distinct peaks at particular drift and retention times, indicating the existence of unique VOCs. Consequently, GC-IMS data comprises highly informative two-dimensional spectra with over 10<sup>6</sup> data points, necessitating pre-processing techniques to extract relevant spatial information (<xref ref-type="bibr" rid="B18">Gu et&#xa0;al., 2021</xref>). GC-IMS has been successfully employed to identify three bacterial species cultured in blood cultures based on their microbial VOC (mVOC) spectra (<xref ref-type="bibr" rid="B10">Drees et&#xa0;al., 2019</xref>). Additionally, bacterial identification using GC-IMS has been documented in the literature for identifying a small set of organisms in both pure and mixed cultures (<xref ref-type="bibr" rid="B28">Lu et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B7">Christmann et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B43">Yan et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B24">Kirtsanis et&#xa0;al., 2025</xref>).</p>
<p>To date, various studies in the literature have combined spectra from GC-IMS prototypes to train supervised classification methods. Bacteria identification, food safety, and food origin are among the most widely explored applications. In (<xref ref-type="bibr" rid="B7">Christmann et&#xa0;al., 2024</xref>), Partial Least Squares Discriminant Analysis (PLS_DA), one of the most commonly used Machine Learning (ML) algorithms in chemometrics, is applied to classify microorganism cultures. By performing both dimensionality reduction and classification, PLS_DA serves as a common baseline for high-dimensional datasets such as GC-IMS. By incorporating target labels into the supervised dimensionality reduction process, it focuses on separating labeled groups. <xref ref-type="bibr" rid="B43">Yan et&#xa0;al. (2024)</xref> applied a 2D CNN-based model, AlexNet, for bacterial culture identification. The use of Deep Learning (DL) models allows for the identification of more complex patterns, while 2D CNNs can precisely leverage the spatial information present in the input data. Additionally, <xref ref-type="bibr" rid="B16">Gerhardt et&#xa0;al. (2019)</xref> demonstrated the effectiveness of combining Principal Component Analysis (PCA) for unsupervised dimensionality reduction with SVM and other ML-based methods to classify different olive oil samples. Similarly, in (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>), a DL-based method was applied to GC-IMS data for olive oil classification. This study showed that a Fully Connected Neural Network (FCNN) outperformed several ML-based models, including Support Vector Machine (SVM), XGBoost, and Logistic Regression (LR). To identify rice varieties and detect adulteration, <xref ref-type="bibr" rid="B22">Ju et&#xa0;al. (2021)</xref> trained a semi-supervised Generative Adversarial Network (GAN) and later replaced the output layer of the discriminator with a softmax classifier, achieving better performance than various ML and DL-based baselines for chemometric tasks. In (<xref ref-type="bibr" rid="B46">Zhao et&#xa0;al., 2024</xref>), an improved GAN based on the diffusion model (DGAN) was used for data generation, followed by a CNN-based model, ResNet50, which outperformed traditional ML baselines in chemometrics.</p>
<p>The main contributions of our presented research are summarized as follows:</p>
<list list-type="bullet">
<list-item>
<p>The integration of DL-based models with chemometric techniques such as GC-IMS offers a pathway toward rapid, culture-based microbial diagnostics. While our current work is exploratory, it represents an important step toward developing clinically applicable solutions for pathogen detection.</p>
</list-item>
<list-item>
<p>Implementation of a pre-processing pipeline to GC-IMS data related to three different bacteria species (<italic>E. coli</italic>, <italic>P. fluorescens</italic> and <italic>L. brevis</italic>) and one fungus (<italic>S. cerevisiae</italic>).</p>
</list-item>
<list-item>
<p>Multi-class classification to identify bacteria and fungi of four pure and ten pure and mixed distinct classes respectively.</p>
</list-item>
<list-item>
<p>Classification of Bacteria &amp; Fungi and Gram-positive &amp; Gram-negative GC-IMS spectra by training ML/DL models on imbalanced datasets.</p>
</list-item>
<list-item>
<p>Identification of pure and mixed GC-IMS spectra based on the Presence and Pureness of the bacteria <italic>E. coli</italic> and <italic>P. fluorescens</italic>.</p>
</list-item>
<list-item>
<p>Implementation of a 5-fold cross-validation protocol to evaluate ML and DL models through various performance metrics.</p>
</list-item>
<list-item>
<p>Further investigation of DL trained models in terms of their training times and trainable parameters.</p>
</list-item>
</list>
<p>The remainder of this paper is structured as follows. The second section details the GC-IMS dataset utilized for training the classification models and outlines the classification methods employed in this study. This section also presents the approach for validating the trained ML and DL models, along with the specifics of the software and hardware used to conduct the experiments. The third section illustrates the experimental results for the classification between bacteria and fungi, as well as Gram-positive and Gram-negative bacteria. Additionally, this section explores multi-class classification of pure and mixed cultures, followed by the identification of specific pathogenic bacteria, such as <italic>Escherichia coli</italic> and <italic>Pseudomonas fluorescens</italic>. The evaluation of  four DL models also consists of investigating their training time and trainable parameters. Finally, the outcomes of the experiments are reported in the last section, accompanied by suggestions on future work.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>GC-IMS data</title>
<p>To evaluate the models&#x2019; performance and their applicability in identifying microbial organisms, we conducted experiments using a publicly available dataset of GC-IMS data from four different organisms (<xref ref-type="bibr" rid="B40">Weller and Christmann, 2023</xref>). These include three bacteria, <italic>Escherichia coli</italic>, <italic>Levilactobacillus brevis</italic>, and <italic>Pseudomonas fluorescens</italic>, and one fungus, <italic>Saccharomyces cerevisiae</italic>. Pure cultures of these organisms were prepared, along with six different mixtures combining each pair. The dataset consists of 10 different classes, with 31 unique cultures analyzed to generate 214 distinct GC-IMS spectra samples. In <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>, we present the number of Cultures and Samples for each class of the presented dataset.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Table of cultures and samples for each different class of the dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Class</th>
<th valign="middle" align="left">Cultures</th>
<th valign="middle" align="left">Samples</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic>
</td>
<td valign="middle" align="center">4</td>
<td valign="middle" align="center">30</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>L. brevis</italic>
</td>
<td valign="middle" align="center">4</td>
<td valign="middle" align="center">28</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>P. fluorescens</italic>
</td>
<td valign="middle" align="center">4</td>
<td valign="middle" align="center">28</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">4</td>
<td valign="middle" align="center">31</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic> and <italic>L. brevis</italic>
</td>
<td valign="middle" align="center">2</td>
<td valign="middle" align="center">14</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic> and <italic>P. fluorescens</italic>
</td>
<td valign="middle" align="center">3</td>
<td valign="middle" align="center">20</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic> and <italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">2</td>
<td valign="middle" align="center">11</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>L. brevis</italic> and <italic>P. fluorescens</italic>
</td>
<td valign="middle" align="center">2</td>
<td valign="middle" align="center">11</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>L. brevis</italic> and <italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">4</td>
<td valign="middle" align="center">27</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>P. fluorescens</italic> and <italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">2</td>
<td valign="middle" align="center">14</td>
</tr>
<tr>
<td valign="middle" align="left">Total</td>
<td valign="middle" align="center">31</td>
<td valign="middle" align="center">214</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>, a representative GC-IMS spectrum can be visualized as a heatmap, where the x-axis indicates the drift time of the separated ions based on IMS and the y-axis represents the retention time derived by GC, corresponding to the separation characteristics of the sample. The color bar indicates the ion intensity, with specific ions (i.e., VOCs fingerprints) forming peaks at particular coordinates, generating rich informational spectra for analysis.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Pre-processing pipeline.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fbrio-04-1620906-g001.tif">
<alt-text content-type="machine-generated">Diagram showing raw-data and pre-processed spectra with a pre-processing pipeline. The raw-data spectrum indicates retention time versus drift time with intensity scale, highlighting a region of interest. The pipeline involves data compression using wavelet transformation, baseline correction with a Tophat white filter, and data points processing including drift time alignment, intensity normalization, and spectra cutting ROI. The pre-processed spectrum displays adjusted data on retention and drift times with a new intensity scale.</alt-text>
</graphic>
</fig>
<p>Given the high dimensionality of the input data (3,150 &#xd7; 6,123), a standard procedure involves applying a pre-processing pipeline (illustrated in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>) to reduce data size, denoise spectra, and extract the most relevant information. To this end, we utilize the open-source Python package <italic>gc-ims-tools</italic> (<xref ref-type="bibr" rid="B6">Christmann et&#xa0;al., 2022</xref>), in accordance to the dataset&#x2019;s initial publication (<xref ref-type="bibr" rid="B7">Christmann et&#xa0;al., 2024</xref>). First, we apply a third-degree wavelet transformation to both the drift and retention time axes, significantly reducing the dimensionality of the spectra to one-tenth of the original data to each direction. Then, a baseline correction algorithm is applied to remove instrumental variations or background noise, using a white top-hat filter of size 15. Finally, to properly prepare the input data for the models, we apply various data points processing techniques. Drift Time Alignment adjusts the drift time axis relative to the Reactant Ion Peak (RIP), which is a standard technique for the task, due to instrumentation variations. Intensities Normalization between zero and one, is a common technique for efficiently training ML and DL models. While, Spectra Cutting is applied on the Region of Interest (ROI), which contains the most valuable information selecting the RIP-relative drift time range between 1.05 and 2.10 and the retention time between 70 and 780 s. The output of the pre-processing pipeline is shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>, where the spectral shape is reduced to (152 &#xd7; 600), resulting in a significant 99.5% reduction in data size.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Classification methods</title>
<p>As outlined in Section 1, GC-IMS has emerged as a promising tool in various tasks for analyzing samples, particularly in food authentication. DL models have also shown great potential in training effective classifiers based on GC-IMS data. However, limited work has addressed the challenge of identifying bacterial cultures directly from data produced by GC-IMS prototypes. To address this, we employ eight different classification methods, widely used in chemometrics, including four ML models, PLS_DA (<xref ref-type="bibr" rid="B7">Christmann et&#xa0;al., 2024</xref>), PCA_LR (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>), PCA_SVM (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>), and XGBoost (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>), and four DL models, FCNN (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>), MLP (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>), CNN1D (<xref ref-type="bibr" rid="B43">Yan et&#xa0;al., 2024</xref>), and CNN2D (<xref ref-type="bibr" rid="B43">Yan et&#xa0;al., 2024</xref>).</p>
<p>Classical ML models are widely employed by researchers to address a variety of problems. Their main advantages include strong generalization on small datasets, fast training times, and relatively few parameters. PLS_DA is a commonly adopted method in chemometrics, providing both data compression and classification of input spectra. Similarly, PCA combined with either SVM or LR is frequently utilized, benefiting from PCA&#x2019;s dimensionality reduction capabilities alongside the efficiency of SVM and LR. Additionally, XGBoost is included in our experiments due to its strong performance across a wide range of ML tasks. To ensure reproducibility of results, a random seed of 42 is set during all trainings. For dimensionality reduction, we retain a number of 50 components, while classification models (SVM, LR, XGBoost) are trained using their default hyper-parameters of their respective packages.</p>
<p>In contrast, DL models feature more parameters, longer training times, and more complex architectures. While they require larger datasets, they effectively capture correlations within input parameters, resulting in improved generalization and higher accuracy. We employ CNN2D to capture the spatial structure of the heatmaps, and CNN1D to assess the performance of convolutional layers on flattened heatmaps. Additionally, we include FCNN and MLP as standard baselines for DL-based methods. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> illustrates the four architectures: CNN2D (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2A</bold>
</xref>), CNN1D (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2B</bold>
</xref>), MLP (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2C</bold>
</xref>), and FCNN (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2C</bold>
</xref>).</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Presentation of four different DL models. <bold>(A)</bold> Architecture of CNN2D model. <bold>(B)</bold> Architecture of CNN1D model. <bold>(C)</bold> Architecture of MLP and FCNN models. Their key difference is the depth of the hidden layers, where on the MLP, we employ a single hidden layer, while on FCNN, we stack four consecutive hidden layers.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fbrio-04-1620906-g002.tif">
<alt-text content-type="machine-generated">Three diagrams labeled A, B, and C illustrate four different neural network classification architectures. Subfigure A represents a Conv 2D network with Leaky ReLU and Max Pooling, followed by flattening and fully connected layers. Subfigure B depicts a similar setup with Conv 1D layers. While Subfigure C features either a FCNN or an MLP with direct flattening, depending on the number of hidden layers.</alt-text>
</graphic>
</fig>
<p>The first model, CNN2D (2A), is adopted to leverage the spatial information within the spectra. The pre-processed spectra serve as input to the architecture, which consists of three hidden 2D convolutional layers with a 3&#xd7;3 kernel size and an increasing number of filters, 32, 64, and 128, respectively. Each convolutional layer is followed by a Leaky ReLU activation and a 2&#xd7;2 Max Pooling operation. The output of the final convolutional layer is flattened and passed through a fully connected hidden layer with Leaky ReLU activation and a Dropout rate of 0.5, resulting in 512 features. Lastly, a final fully connected layer is applied in the extracted features to generate the target classes.</p>
<p>Additionally, we implement a CNN1D (2B) model. After flattening the input spectra, and following the design of CNN2D (2A), we apply three 1D convolutional layers with a kernel size of 9, each followed by a Leaky ReLU activation and a 1D Max Pooling layer of size 4. The number of filters increases progressively (32, 64, 128). As previously, the output of the final convolutional layer is flattened and passed through a fully connected layer with Leaky ReLU activation and a Dropout rate of 0.5, resulting in 512 features. Finally, a fully connected output layer maps these features to the target classes.</p>
<p>Finally, we employ two models: MLP (2C) and FCNN (2C). In both, the input spectra are flattened and passed through fully connected hidden layers, each followed by a Leaky ReLU activation and a Dropout rate of 0.5, resulting in 512 features. Although their architecture is presented in the same way, the MLP model consists of a single hidden layer, whereas the FCNN model includes four hidden layers, resulting in more parameters and higher complexity. As in the previous models, a final fully connected layer processes these features to generate the target class predictions.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Evaluation protocol</title>
<p>The original study that introduced this dataset evaluated models using a fixed train/test split. While this approach is common for large datasets, it has significant disadvantages in the context of small datasets like ours. First, with limited samples, a single split can lead to severe under-representation of certain classes in the test set, causing performance metrics to vary greatly depending on which samples are chosen. Second, the original paper does not provide sufficient details regarding how the split was constructed (e.g., sample or culture stratification, class balance, random seed), making it impossible to reproduce the original results and perform a fair, like-for-like comparison. To this end, and given the small number of samples available in the dataset for many classes (see <xref ref-type="table" rid="T1">
<bold>Tables&#xa0;1</bold>
</xref>, <xref ref-type="table" rid="T2">
<bold>2</bold>
</xref>), creating a fixed test set would result in under-representation of several classes and make final performance metrics highly sensitive to the specific samples selected.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Dataset imbalance report.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Experiment</th>
<th valign="middle" align="left">Minority class</th>
<th valign="middle" align="left">Majority class</th>
<th valign="middle" align="center">Total</th>
<th valign="middle" align="center">Imbalance (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Bacteria &amp; Fungi</td>
<td valign="middle" align="left">31 (Fungi)</td>
<td valign="middle" align="left">86 (Bacteria)</td>
<td valign="middle" align="center">117</td>
<td valign="middle" align="center">73.50</td>
</tr>
<tr>
<td valign="middle" align="left">Gram-positive &amp; Gram-negative</td>
<td valign="middle" align="left">28 (Gram-positive)</td>
<td valign="middle" align="left">58 (Gram-negative)</td>
<td valign="middle" align="center">86</td>
<td valign="middle" align="center">67.44</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic> (+)</td>
<td valign="middle" align="left">75 (Presence)</td>
<td valign="middle" align="left">139 (Absence)</td>
<td valign="middle" align="center">214</td>
<td valign="middle" align="center">64.95</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic> (*)</td>
<td valign="middle" align="left">30 (Pureness)</td>
<td valign="middle" align="left">45 (Mixed)</td>
<td valign="middle" align="center">75</td>
<td valign="middle" align="center">60.00</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>P. fluorescens</italic> (+)</td>
<td valign="middle" align="left">73 (Presence)</td>
<td valign="middle" align="left">141 (Absence)</td>
<td valign="middle" align="center">214</td>
<td valign="middle" align="center">65.88</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>P. fluorescens</italic> (*)</td>
<td valign="middle" align="left">28 (Pureness)</td>
<td valign="middle" align="left">45 (Mixed)</td>
<td valign="middle" align="center">73</td>
<td valign="middle" align="center">61.64</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>An alternative strategy that could prevent potential data leakage caused by culture-specific information is Leave-One-Culture-Out (LOCO) cross-validation, where all spectra from a given culture are held out for validation in each iteration. This would provide a rigorous assessment of generalization to unseen cultures. However, our dataset contains only 31 cultures, with as few as 2&#x2013;4 cultures per class. Under these conditions, LOCO would result in extremely small training sets for some classes (2 cultures in a class, result in 50% split training and validation sets), leading to unstable performance estimates and impractical model training.</p>
<p>To overcome these limitations, we adopt a stratified five-fold cross-validation protocol at the sample level, as illustrated in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>. Cross-Validation is widely recommended for small datasets to improve robustness and reduce variance caused by test set selection, not only in ML and DL in general (<xref ref-type="bibr" rid="B25">Kohavi, 1995</xref>; <xref ref-type="bibr" rid="B32">Refaeilzadeh et&#xa0;al., 2009</xref>; <xref ref-type="bibr" rid="B31">Raschka, 2018</xref>), but also has been widely adopted in chemometrics and bioinformatics tasks for similar reasons (<xref ref-type="bibr" rid="B4">Beleites and Salzer, 2008</xref>; <xref ref-type="bibr" rid="B13">Esbensen and Geladi, 2010</xref>; <xref ref-type="bibr" rid="B41">Westad and Marini, 2015</xref>). In this setup, the dataset is randomly partitioned into five folds while preserving the overall class distribution in each fold. For each iteration, 80% of the samples are used for training and 20% for validation, ensuring that every sample appears exactly once in a validation set across five independent trainings. This approach maximizes the use of available data while maintaining comparability across experiments. To ensure fairness and like-to-like comparison between the proposed methods and the original study&#x2019;s baselines, we include the PLS_DA model in our evaluation using the same hyperparameters. For reproducibility, all experiments are performed with a fixed random seed of 42.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Presentation of five-fold cross-validation evaluation protocol.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fbrio-04-1620906-g003.tif">
<alt-text content-type="machine-generated">Diagram depicting a stratified k-fold cross-validation process with five folds. Data is split with each iteration using a different fold for validation. Training is conducted on remaining folds. Metrics are calculated for each iteration and averaged along with their standard deviation.</alt-text>
</graphic>
</fig>
<p>For DL baseline models, we employ an Adam optimizer, a learning rate of 0.001, a batch size of 8 and cross-entropy loss function (L<italic>
<sub>CE</sub>
</italic>), as shown in <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>. Although hyperparameter fine-tuning could potentially improve the performance of each individual model, it was not applied in this study due to the limited dataset size and the primary objective of comparing baseline architectures rather than optimizing them, this would be out of scope for this study. Therefore, we report results using standard training hyperparameters.</p>
<p>During training, we evaluate through Accuracy (<xref ref-type="disp-formula" rid="eq2">Equation 2</xref>) and F1-score (<xref ref-type="disp-formula" rid="eq3">Equations 3</xref>&#x2013;<xref ref-type="disp-formula" rid="eq5">5</xref>) across the validation set, while Sensitivity (<xref ref-type="disp-formula" rid="eq6">Equation 6</xref>) and Specificity (<xref ref-type="disp-formula" rid="eq7">Equation 7</xref>) are also reported for the validation set for each class <italic>c</italic>, separately. Additionally, training and inference times along with the number of parameters are reported to provide a comprehensive comparison of each model for each experiment. In the following equations, <italic>C</italic>, <italic>N</italic>, <italic>TP</italic>, <italic>TN</italic>, <italic>FP</italic>, and <italic>FN</italic> represent number of classes, number of samples, true positives, true negatives, false positives, and false negatives, respectively. Moreover, <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent the one-hot encoded ground truth label for sample <italic>i</italic> and class <italic>c</italic>, and <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent the predicted probability (from softmax) for sample <italic>i</italic> and class <italic>c</italic>.</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msub>
<mml:mtext>L</mml:mtext>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>C</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>log</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mtext>Accuracy</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mtext>Precision</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mtext>Recall</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mtext>F</mml:mtext>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>score</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext>Precision</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>Recall</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>Precision</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext>Recall</mml:mtext>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>Sensitivity</mml:mtext>
</mml:mrow>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mo>&#x2200;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mo>{</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>Specificity</mml:mtext>
</mml:mrow>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mo>&#x2200;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mo>{</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Software and hardware requirements</title>
<p>For the needs of the given research, we conducted all the experiments on a server equipped with a NVIDIA GeForce RTX 4090 GPU. <italic>CUDA</italic> (12.0) along with <italic>Python</italic> (3.8.20) have been used, while various packages have been employed among <italic>gc-ims-tools</italic> (0.1.7) for the data pre-processing pipeline and PLS_DA training, <italic>scikit-learn</italic> (1.3.2) for the ML models of PCA_SVM and PCA_LR, <italic>xgboost</italic> (2.1.1) for the XGBoost model and <italic>torch</italic> (2.4.1+cu118) for the training of DL models.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results</title>
<sec id="s3_1">
<label>3.1</label>
<title>Tables of experiments</title>
<p>Following, on <xref ref-type="table" rid="T3">
<bold>Tables&#xa0;3</bold>
</xref>, <xref ref-type="table" rid="T4">
<bold>4</bold>
</xref>, we present a list of the eight different experiments conducted on the dataset. More precisely, on <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>, we present two multi-class classification experiments, Pure and Pure &amp; Mixed. As their names suggest, in the first experiment, we train models to classify between the four pure classes, whereas in the second, we train models to classify both pure and mixed cultures, resulting in ten different classes. The numbers indicate the assigned class labels, a standard approach in ML and DL training. Diving deeper into the classification of the pure cultures, we conduct two additional classification experiments, Bacteria &amp; Fungi and Gram-positive &amp; Gram-negative (<xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>). The Bacteria &amp; Fungi experiment is particularly important as it evaluates the models&#x2019; ability to distinguish bacterial from fungal behavior. Meanwhile, the Gram-positive &amp; Gram-negative experiment focuses into bacterial characteristics, analyzing their GC-IMS fingerprints based on their cell wall type. These two experiments demonstrate a dataset imbalance, as reported in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>, which is a common issue in the literature when training ML and DL models. The Gram-positive class contains 28 samples, while the Gram-negative class has more than twice as many, with 58 samples. Similarly, bacterial samples account for a subset of 86 spectra, whereas fungal samples are limited to just 31, approximately one third.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Table of experiments categorized by pure, pure &amp; mixed, bacteria &amp; fungi, and Gram-positive &amp; Gram-negative.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Class</th>
<th valign="middle" align="center">Pure</th>
<th valign="middle" align="center">Pure &amp; Mixed</th>
<th valign="middle" align="center">Bacteria &amp; Fungi</th>
<th valign="middle" align="center">Gram-positive &amp; Gram-negative</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic>
</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">0</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>L. brevis</italic>
</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">1</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>P. fluorescens</italic>
</td>
<td valign="middle" align="center">2</td>
<td valign="middle" align="center">2</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">0</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">3</td>
<td valign="middle" align="center">3</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic> and <italic>L. brevis</italic>
</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">4</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic> and <italic>P. fluorescens</italic>
</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">5</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic> and <italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">6</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>L. brevis</italic> and <italic>P. fluorescens</italic>
</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">7</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>L. brevis</italic> and <italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">8</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>P. fluorescens</italic> and <italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">9</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Each number represents the assigned group for each class.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Table of experiments regarding the presence (+) and pureness (*) of <italic>E. coli</italic> and <italic>P. fluorescens</italic> to specifically identify pathogenic bacterial cultures.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Class</th>
<th valign="middle" align="center">
<italic>E. coli</italic> (+)</th>
<th valign="middle" align="center">
<italic>E. coli</italic> (*)</th>
<th valign="middle" align="center">
<italic>P. fluorescens</italic> (+)</th>
<th valign="middle" align="center">
<italic>P. fluorescens</italic> (*)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic>
</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>L. brevis</italic>
</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>P. fluorescens</italic>
</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">1</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic> and <italic>L. brevis</italic>
</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic> and <italic>P. fluorescens</italic>
</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">0</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic> and <italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>L. brevis</italic> and <italic>P. fluorescens</italic>
</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">0</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>L. brevis</italic> and <italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>P. fluorescens</italic> and <italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">0</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Each number represents the assigned group for each class.</p>
</table-wrap-foot>
</table-wrap>
<p>On the other hand, we explore the possibility of specifically identifying either the Presence (+) or the Pureness (*) of two distinct pathogenic bacteria, <italic>E. coli</italic> and <italic>P. fluorescens</italic>. As presented in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>, to properly evaluate the models&#x2019; ability to detect the presence of a specific bacterium, we assign a value of 1 (one) to all the classes containing the given bacteria and 0 (zero) to all the remaining classes. Subsequently, by training from-scratch the models, we investigate their ability to identify the pureness of the bacteria among all the classes in which it is present. This experiment highlights the models&#x2019; sensitivity on the given bacteria. These two rounds of experiments are conducted for each bacteria, serving as a baseline for identifying pathogenic bacteria using ML and DL methods based on GC-IMS spectra. Similarly, dataset imbalance is also observed in these experiments (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>). For the <italic>E. coli</italic> (+) experiment, the dataset includes 75 samples with presence and 139 with absence. In the <italic>E. coli</italic> (*) experiment, results are reported for 30 pure samples and 45 mixed samples. Likewise, for <italic>P. fluorescens</italic> (+), there are 73 samples with presence and 141 with absence, while the <italic>P. fluorescens</italic> (*) experiment includes 28 pure samples and 45 mixed samples.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Pure and mixed cultures</title>
<p>As discussed earlier and in line with the initial dataset&#x2019;s publication, we experiment with multi-class classification in two different scenarios: Pure and Pure &amp; Mixed cultures. In the Pure experiment, we classify samples into four distinct categories, each representing a pure culture of a specific organism: <italic>E. coli</italic>, <italic>L. brevis</italic>, <italic>P. fluorescens</italic>, and <italic>S. cerevisiae</italic>. In contrast, the Pure &amp; Mixed experiment evaluates the models&#x2019; ability to identify between pure cultures and all possible pairwise mixed cultures in the dataset, resulting in ten distinct classes.</p>
<p>
<xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref> presents the classification results for the models described in Subsection 2.2, reporting the average and standard deviation of Accuracy and F1-Score across five cross-validation folds. The highest-performing model is highlighted in bold, while the second-best is underlined. In both experiments, the FCNN model demonstrates clear superiority, achieving 93.19% average accuracy and 93.04% average F1-score in the Pure experiment, and 92.53% average accuracy and 93.37% average F1-score in the Pure &amp; Mixed experiment, all with a relatively small standard deviation. The CNN2D model consistently ranks second, demonstrating strong performance across all metrics and tasks, showcasing its ability to generalize the information based on their spatial information. The overall performance of DL models can be summarized as an out-performance compared to traditional ML baselines.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Average and standard deviation of accuracy and F1-score for pure and pure &amp; mixed experiments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="left">Model</th>
<th valign="middle" colspan="2" align="center">Pure</th>
<th valign="middle" colspan="2" align="center">Pure &amp; Mixed</th>
</tr>
<tr>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">F1-score</th>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">F1-score</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">XGBoost (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">0.8381 &#xb1; 0.03</td>
<td valign="middle" align="center">0.8378 &#xb1; 0.03</td>
<td valign="middle" align="center">0.6212 &#xb1; 0.07</td>
<td valign="middle" align="center">0.5452 &#xb1; 0.09</td>
</tr>
<tr>
<td valign="middle" align="left">PCA_LR (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">0.8725 &#xb1; 0.04</td>
<td valign="middle" align="center">0.8720 &#xb1; 0.04</td>
<td valign="middle" align="center">0.7476 &#xb1; 0.07</td>
<td valign="middle" align="center">0.7066 &#xb1; 0.09</td>
</tr>
<tr>
<td valign="middle" align="left">PCA_SVM (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">0.8634 &#xb1; 0.05</td>
<td valign="middle" align="center">0.8636 &#xb1; 0.05</td>
<td valign="middle" align="center">0.7991 &#xb1; 0.06</td>
<td valign="middle" align="center">0.7696 &#xb1; 0.07</td>
</tr>
<tr>
<td valign="middle" align="left">PLS_DA (<xref ref-type="bibr" rid="B7">Christmann et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">0.8978 &#xb1; 0.04</td>
<td valign="middle" align="center">0.8978 &#xb1; 0.04</td>
<td valign="middle" align="center">0.8505 &#xb1; 0.06</td>
<td valign="middle" align="center">0.8539 &#xb1; 0.07</td>
</tr>
<tr>
<td valign="middle" align="left">CNN2D (<xref ref-type="bibr" rid="B43">Yan et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">
<underline>0.9239</underline> &#xb1; 0.06</td>
<td valign="middle" align="center">
<underline>0.9224</underline> &#xb1; 0.06</td>
<td valign="middle" align="center">
<underline>0.9205</underline> &#xb1; 0.03</td>
<td valign="middle" align="center">
<underline>0.9282</underline> &#xb1; 0.03</td>
</tr>
<tr>
<td valign="middle" align="left">CNN1D (<xref ref-type="bibr" rid="B43">Yan et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">0.9152 &#xb1; 0.05</td>
<td valign="middle" align="center">0.9137 &#xb1; 0.05</td>
<td valign="middle" align="center">0.8739 &#xb1; 0.05</td>
<td valign="middle" align="center">0.8701 &#xb1; 0.05</td>
</tr>
<tr>
<td valign="middle" align="left">MLP (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">0.9065 &#xb1; 0.05</td>
<td valign="middle" align="center">0.9064 &#xb1; 0.05</td>
<td valign="middle" align="center">
<underline>0.9205</underline> &#xb1; 0.03</td>
<td valign="middle" align="center">0.9231 &#xb1; 0.03</td>
</tr>
<tr>
<td valign="middle" align="left">FCNN (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">
<bold>0.9319</bold> &#xb1; 0.04</td>
<td valign="middle" align="center">
<bold>0.9304</bold> &#xb1; 0.04</td>
<td valign="middle" align="center">
<bold>0.9253</bold> &#xb1; 0.03</td>
<td valign="middle" align="center">
<bold>0.9337</bold> &#xb1; 0.03</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Bold values indicate the best-performing model, while underlined values indicate the second best-performing model.</p>
</table-wrap-foot>
</table-wrap>
<p>Following <xref ref-type="table" rid="T6">
<bold>Table&#xa0;6</bold>
</xref>, we analyze the Selectivity and Specificity of each class in both experiments using the best-performing model, FCNN. We observe that in both cases, the model achieves strong performance across all classes for both metrics. More specifically, introducing pairwise mixed cultures into the training set affects the performance on pure classes. For instance, the model&#x2019;s ability to identify <italic>E. coli</italic> and <italic>L. brevis</italic> significantly decreases, whereas <italic>P. fluorescens</italic> and <italic>S. cerevisiae</italic> maintain or slightly improve their performance. On the other hand, the mixed cultures exhibit a sensitivity variation of up to 10%, while specificity remains consistently high, with differences of less than 1% between values across the different classes.</p>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>Sensitivity and specificity for each class of the best performing model FCNN in pure and pure &amp; mixed experiments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="bottom" rowspan="2" align="left">Class</th>
<th valign="middle" colspan="2" align="center">Pure</th>
<th valign="middle" colspan="2" align="center">Pure &amp; Mixed</th>
</tr>
<tr>
<th valign="middle" align="center">Sensitivity</th>
<th valign="middle" align="center">Specificity</th>
<th valign="middle" align="center">Sensitivity</th>
<th valign="middle" align="center">Specificity</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic>
</td>
<td valign="middle" align="center">0.8933 &#xb1; 0.09</td>
<td valign="middle" align="center">0.9667 &#xb1; 0.04</td>
<td valign="middle" align="center">0.8200 &#xb1; 0.13</td>
<td valign="middle" align="center">0.9787 &#xb1; 0.02</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>L. brevis</italic>
</td>
<td valign="middle" align="center">0.9667 &#xb1; 0.07</td>
<td valign="middle" align="center">0.9889 &#xb1; 0.02</td>
<td valign="middle" align="center">0.9000 &#xb1; 0.08</td>
<td valign="middle" align="center">0.9946 &#xb1; 0.01</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>P. fluorescens</italic>
</td>
<td valign="middle" align="center">0.9267 &#xb1; 0.09</td>
<td valign="middle" align="center">0.9666 &#xb1; 0.03</td>
<td valign="middle" align="center">0.9333 &#xb1; 0.13</td>
<td valign="middle" align="center">0.9731 &#xb1; 0.00</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">0.9381 &#xb1; 0.08</td>
<td valign="middle" align="center">0.9882 &#xb1; 0.02</td>
<td valign="middle" align="center">0.9381 &#xb1; 0.08</td>
<td valign="middle" align="center">0.9890 &#xb1; 0.01</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic> and <italic>L. brevis</italic>
</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">1.0000 &#xb1; 0.00</td>
<td valign="middle" align="center">0.9949 &#xb1; 0.01</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic> and <italic>P. fluorescens</italic>
</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">0.9500 &#xb1; 0.10</td>
<td valign="middle" align="center">0.9947 &#xb1; 0.01</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>E. coli</italic> and <italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">1.0000 &#xb1; 0.00</td>
<td valign="middle" align="center">0.9951 &#xb1; 0.01</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>L. brevis</italic> and <italic>P. fluorescens</italic>
</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">0.9000 &#xb1; 0.20</td>
<td valign="middle" align="center">1.0000 &#xb1; 0.00</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>L. brevis</italic> and <italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">0.9267 &#xb1; 0.09</td>
<td valign="middle" align="center">0.9947 &#xb1; 0.01</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>P. fluorescens</italic> and <italic>S. cerevisiae</italic>
</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">1.0000 &#xb1; 0.00</td>
<td valign="middle" align="center">1.0000 &#xb1; 0.00</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Bacteria &amp; Fungi and Gram-positive &amp; Gram-negative</title>
<p>The classification between Bacteria &amp; Fungi is a key experiment in our work, as it highlights the distinct correlations associated with bacterial compared to fungal cultures. In this experiment, we conducted a new training of the models based on their ability to classify between the two categories. The main challenge lies in effectively grouping all bacterial samples and identifying their correlations in comparison to fungi cultures.</p>
<p>As shown in <xref ref-type="table" rid="T7">
<bold>Table&#xa0;7</bold>
</xref>, both MLP and FCNN models achieve top performance, reaching an average accuracy and f1-score of 98.30% and 97.63%, respectively. These are followed by the other two DL baselines, CNN1D and CNN2D, while the ML baselines also demonstrate promising performance. Similarly, in <xref ref-type="table" rid="T8">
<bold>Table&#xa0;8</bold>
</xref>, we observe that for class zero (bacteria), the top-performing models achieve 100% sensitivity and 93.33% specificity, with an inverse pattern observed for class one (fungi). </p>
<table-wrap id="T7" position="float">
<label>Table&#xa0;7</label>
<caption>
<p>Average and standard deviation of Accuracy and F1-score for Bacteria &amp; Fungi and Gram-positive &amp; Gram-negative experiments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="bottom" rowspan="2" align="left">Model</th>
<th valign="middle" colspan="2" align="center">Bacteria &amp; Fungi</th>
<th valign="middle" colspan="2" align="center">Gram-positive &amp; Gram-negative</th>
</tr>
<tr>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">F1-score</th>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">F1-score</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">XGBoost (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">0.9485 &#xb1; 0.04</td>
<td valign="middle" align="center">0.9243 &#xb1; 0.07</td>
<td valign="middle" align="center">0.8954 &#xb1; 0.06</td>
<td valign="middle" align="center">0.8743 &#xb1; 0.07</td>
</tr>
<tr>
<td valign="middle" align="left">PCA_LR (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">0.9572 &#xb1; 0.03</td>
<td valign="middle" align="center">0.9402 &#xb1; 0.04</td>
<td valign="middle" align="center">0.9307 &#xb1; 0.04</td>
<td valign="middle" align="center">0.9181 &#xb1; 0.05</td>
</tr>
<tr>
<td valign="middle" align="left">PCA_SVM (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">0.9319 &#xb1; 0.05</td>
<td valign="middle" align="center">0.9007 &#xb1; 0.08</td>
<td valign="middle" align="center">0.9065 &#xb1; 0.08</td>
<td valign="middle" align="center">0.8986 &#xb1; 0.09</td>
</tr>
<tr>
<td valign="middle" align="left">PLS_DA (<xref ref-type="bibr" rid="B7">Christmann et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">0.9659 &#xb1; 0.03</td>
<td valign="middle" align="center">0.9539 &#xb1; 0.04</td>
<td valign="middle" align="center">
<underline>0.9654</underline> &#xb1; 0.03</td>
<td valign="middle" align="center">
<underline>0.9610</underline> &#xb1; 0.03</td>
</tr>
<tr>
<td valign="middle" align="left">CNN2D (<xref ref-type="bibr" rid="B43">Yan et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">
<underline>0.9743</underline> &#xb1; 0.02</td>
<td valign="middle" align="center">
<underline>0.9643</underline> &#xb1; 0.03</td>
<td valign="middle" align="center">0.9301 &#xb1; 0.02</td>
<td valign="middle" align="center">0.9192 &#xb1; 0.03</td>
</tr>
<tr>
<td valign="middle" align="left">CNN1D (<xref ref-type="bibr" rid="B43">Yan et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">
<underline>0.9743</underline> &#xb1; 0.02</td>
<td valign="middle" align="center">
<underline>0.9643</underline> &#xb1; 0.03</td>
<td valign="middle" align="center">0.8255 &#xb1; 0.08</td>
<td valign="middle" align="center">0.8150 &#xb1; 0.08</td>
</tr>
<tr>
<td valign="middle" align="left">MLP (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">
<bold>0.9830</bold> &#xb1; 0.02</td>
<td valign="middle" align="center">
<bold>0.9763</bold> &#xb1; 0.03</td>
<td valign="middle" align="center">
<bold>0.9771</bold> &#xb1; 0.03</td>
<td valign="middle" align="center">
<bold>0.9735</bold> &#xb1; 0.03</td>
</tr>
<tr>
<td valign="middle" align="left">FCNN (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">
<bold>0.9830</bold> &#xb1; 0.02</td>
<td valign="middle" align="center">
<bold>0.9763</bold> &#xb1; 0.03</td>
<td valign="middle" align="center">
<bold>0.9771</bold> &#xb1; 0.03</td>
<td valign="middle" align="center">
<bold>0.9735</bold> &#xb1; 0.03</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Bold values indicate the best-performing model, while underlined values indicate the second best-performing model.</p>
</table-wrap-foot>
</table-wrap>
<p>Following the same setup, we conducted another innovative experiment, aiming to further analyze the behavior of a more precise bacterial categorization, namely Gram-positive &amp; Gram-negative. To this end, we assigned class zero to Gram-negative bacteria, including <italic>E. coli</italic> and <italic>P. fluorescens</italic>, and class one to Gram-positive bacteria such as <italic>L. brevis</italic>.</p>
<p>In this task, as shown in <xref ref-type="table" rid="T7">
<bold>Table&#xa0;7</bold>
</xref>, MLP and FCNN again emerge as the top-performing models, achieving 97.71% accuracy and 97.35% f1-score. The ML baseline PLS_DA follows as the second-best performer in both metrics, while the remaining ML and DL baselines demonstrate significant lower performance. For the best-performing models, FCNN and MLP, sensitivity and specificity alternate between 100% and 93.33% across the two classes, as presented in <xref ref-type="table" rid="T8">
<bold>Table&#xa0;8</bold>
</xref>.</p>
<table-wrap id="T8" position="float">
<label>Table&#xa0;8</label>
<caption>
<p>Sensitivity and Specificity for each class of the best performing models MLP and FCNN in Bacteria (zero) &amp; Fungi (one) and Gram-positive (one) &amp; Gram-negative (zero) experiments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Experiment</th>
<th valign="middle" align="left">Class</th>
<th valign="middle" align="left">Sensitivity</th>
<th valign="middle" align="left">Specificity</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="2" align="left">Bacteria &amp; Fungi</td>
<td valign="middle" align="left">0</td>
<td valign="middle" align="left">1.0000 &#xb1; 0.00</td>
<td valign="middle" align="left">0.9333 &#xb1; 0.08</td>
</tr>
<tr>
<td valign="middle" align="left">1</td>
<td valign="middle" align="left">0.9333 &#xb1; 0.08</td>
<td valign="middle" align="left">1.0000 &#xb1; 0.00</td>
</tr>
<tr>
<td valign="middle" rowspan="2" align="left">Gram-positive &amp; Gram-negative</td>
<td valign="middle" align="left">0</td>
<td valign="middle" align="left">1.0000 &#xb1; 0.00</td>
<td valign="middle" align="left">0.9333 &#xb1; 0.08</td>
</tr>
<tr>
<td valign="middle" align="left">1</td>
<td valign="middle" align="left">0.9333 &#xb1; 0.08</td>
<td valign="middle" align="left">1.0000 &#xb1; 0.00</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Each number represents the assigned group for each class.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Identification of pathogenic bacterial cultures</title>
<p>Finally, we evaluate the performance of the proposed models in the task of specifically identifying pathogenic bacteria, such as <italic>E. coli</italic>, as a highly pathogenic bacterium and <italic>P. fluorescens</italic> as a low pathogenic bacterium. To this end, our initial experiments involve ten different pure and mixed cultures to evaluate the models&#x2019; ability to detect the Presence (+) of a specific bacterium. We assign class zero to cultures that do not contain the specific bacterium and class one to those where it is present. Furthermore, among the cultures where the bacterium is present, we further classify the Pureness (*) of the culture in comparison to mixed ones, to dive deeper into the identification of pathogenic bacterial cultures.</p>
<p>As presented in <xref ref-type="table" rid="T9">
<bold>Tables&#xa0;9</bold>
</xref>, <xref ref-type="table" rid="T10">
<bold>10</bold>
</xref>, we observe that FCNN is the best-performing model for identifying the presence of <italic>E. coli</italic>, achieving 92.50% accuracy and an F1-score of 91.40%, closely followed by MLP with 92.04% accuracy and 90.86% F1-score. The remaining baselines exhibit lower performance. Sensitivity and specificity alternate between 96.45% and 84.76%, demonstrating that the model is highly specific in detecting the presence of pathogenic bacteria.</p>
<table-wrap id="T9" position="float">
<label>Table&#xa0;9</label>
<caption>
<p>Average and standard deviation of Accuracy and F1-score for Presence (+) and Pureness (*) of <italic>E. coli</italic> experiments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="bottom" rowspan="2" align="left">Model</th>
<th valign="middle" colspan="2" align="center">
<italic>E. coli</italic> (+)</th>
<th valign="middle" colspan="2" align="center">
<italic>E. coli</italic> (*)</th>
</tr>
<tr>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">F1-score</th>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">F1-score</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">PCA_SVM (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">0.8130 &#xb1; 0.07</td>
<td valign="middle" align="center">0.7810 &#xb1; 0.08</td>
<td valign="middle" align="center">
<underline>0.9867</underline> &#xb1; 0.03</td>
<td valign="middle" align="center">
<underline>0.9864</underline> &#xb1; 0.03</td>
</tr>
<tr>
<td valign="middle" align="left">PCA_LR (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">0.8315 &#xb1; 0.05</td>
<td valign="middle" align="center">0.8053 &#xb1; 0.06</td>
<td valign="middle" align="center">
<underline>0.9867</underline> &#xb1; 0.03</td>
<td valign="middle" align="center">
<underline>0.9864</underline> &#xb1; 0.03</td>
</tr>
<tr>
<td valign="middle" align="left">XGBoost (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">0.8548 &#xb1; 0.08</td>
<td valign="middle" align="center">0.8309 &#xb1; 0.09</td>
<td valign="middle" align="center">0.9305 &#xb1; 0.09</td>
<td valign="middle" align="center">0.9291 &#xb1; 0.09</td>
</tr>
<tr>
<td valign="middle" align="left">PLS_DA (<xref ref-type="bibr" rid="B7">Christmann et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">0.8878 &#xb1; 0.07</td>
<td valign="middle" align="center">0.8776 &#xb1; 0.07</td>
<td valign="middle" align="center">
<underline>0.9867</underline> &#xb1; 0.03</td>
<td valign="middle" align="center">
<underline>0.9864</underline> &#xb1; 0.03</td>
</tr>
<tr>
<td valign="middle" align="left">CNN2D (<xref ref-type="bibr" rid="B43">Yan et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">0.8926 &#xb1; 0.04</td>
<td valign="middle" align="center">0.8766 &#xb1; 0.05</td>
<td valign="middle" align="center">
<underline>0.9867</underline> &#xb1; 0.03</td>
<td valign="middle" align="center">
<underline>0.9864</underline> &#xb1; 0.03</td>
</tr>
<tr>
<td valign="middle" align="left">CNN1D (<xref ref-type="bibr" rid="B43">Yan et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">0.8643 &#xb1; 0.05</td>
<td valign="middle" align="center">0.8469 &#xb1; 0.05</td>
<td valign="middle" align="center">
<bold>1.0000</bold> &#xb1; 0.00</td>
<td valign="middle" align="center">
<bold>1.0000</bold> &#xb1; 0.00</td>
</tr>
<tr>
<td valign="middle" align="left">MLP (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">
<underline>0.9204</underline> &#xb1; 0.06</td>
<td valign="middle" align="center">
<underline>0.9086</underline> &#xb1; 0.06</td>
<td valign="middle" align="center">
<underline>0.9867</underline> &#xb1; 0.03</td>
<td valign="middle" align="center">
<underline>0.9864</underline> &#xb1; 0.03</td>
</tr>
<tr>
<td valign="middle" align="left">FCNN (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">
<bold>0.9250</bold> &#xb1; 0.05</td>
<td valign="middle" align="center">
<bold>0.9140 +</bold> 0.06</td>
<td valign="middle" align="center">
<bold>1.0000</bold> &#xb1; 0.00</td>
<td valign="middle" align="center">
<bold>1.0000</bold> &#xb1; 0.00</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Bold values indicate the best-performing model, while underlined values indicate the second best-performing model.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T10" position="float">
<label>Table&#xa0;10</label>
<caption>
<p>Sensitivity and Specificity for each class of the best performing model FCNN in Presence (+) and Pureness (*) of <italic>E. coli</italic> and <italic>P. fluorescens</italic> experiments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Experiment</th>
<th valign="middle" align="center">Class</th>
<th valign="middle" align="center">Sensitivity</th>
<th valign="middle" align="center">Specificity</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="2" align="left">
<italic>E. coli</italic> (+)</td>
<td valign="middle" align="left">0</td>
<td valign="middle" align="center">0.9645 &#xb1; 0.04</td>
<td valign="middle" align="center">0.8476 &#xb1; 0.09</td>
</tr>
<tr>
<td valign="middle" align="left">1</td>
<td valign="middle" align="center">0.8476 &#xb1; 0.09</td>
<td valign="middle" align="center">0.9645 &#xb1; 0.04</td>
</tr>
<tr>
<td valign="middle" rowspan="2" align="left">
<italic>E. coli</italic> (*)</td>
<td valign="middle" align="left">0</td>
<td valign="middle" align="center">1.0000 &#xb1; 0.00</td>
<td valign="middle" align="center">1.0000 &#xb1; 0.00</td>
</tr>
<tr>
<td valign="middle" align="left">1</td>
<td valign="middle" align="center">1.0000 &#xb1; 0.00</td>
<td valign="middle" align="center">1.0000 &#xb1; 0.00</td>
</tr>
<tr>
<td valign="middle" rowspan="2" align="left">
<italic>P. fluorescens</italic> (+)</td>
<td valign="middle" align="left">0</td>
<td valign="middle" align="center">0.9162 &#xb1; 0.07</td>
<td valign="middle" align="center">0.9431 &#xb1; 0.03</td>
</tr>
<tr>
<td valign="middle" align="left">1</td>
<td valign="middle" align="center">0.9431 &#xb1; 0.03</td>
<td valign="middle" align="center">0.9162 &#xb1; 0.07</td>
</tr>
<tr>
<td valign="middle" rowspan="2" align="left">
<italic>P. fluorescens</italic> (*)</td>
<td valign="middle" align="left">0</td>
<td valign="middle" align="center">1.0000 &#xb1; 0.00</td>
<td valign="middle" align="center">0.9667 &#xb1; 0.07</td>
</tr>
<tr>
<td valign="middle" align="left">1</td>
<td valign="middle" align="center">0.9667 &#xb1; 0.07</td>
<td valign="middle" align="center">1.0000 &#xb1; 0.00</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Each number represents the assigned group for each class.</p>
</table-wrap-foot>
</table-wrap>
<p>On the other hand, when distinguishing between pure and mixed cultures of the pathogenic bacterium <italic>E. coli</italic>, both FCNN and CNN1D achieved perfect performance, with 100% accuracy, F1-score, sensitivity, and specificity. The remaining baseline models also demonstrated strong performance, further supporting the reliability of the classification. These results highlight the models&#x2019; ability to accurately determine the pureness of specific pathogenic bacterial cultures. However, potential dataset-specific or culture-dependent noise should be considered when interpreting these perfect scores. Future work should involve evaluating these classes on larger and more diverse datasets to assess the consistency and generalization of the models.</p>
<p>In alignment with previous experiments, we investigate the classification performance of the models based on the Presence (+) and Pureness (*) of another pathogenic bacterium, <italic>P. fluorescens</italic>. <xref ref-type="table" rid="T11">
<bold>Table&#xa0;11</bold>
</xref> once again highlights the superiority of the FCNN model in both tasks. Specifically, FCNN achieves 93.44% accuracy and an f1-score of 92.70% in the <italic>P. fluorescens</italic> (+) experiment, followed by MLP and CNN2D, while CNN1D and PLS_DA also demonstrate promising results. Similarly, in the <italic>P. fluorescens</italic> (*) experiment, FCNN outperforms the other baselines, achieving 98.67% and 98.56% of accuracy and f1-score, respectively. CNN2D follows closely in performance, whereas the remaining baselines exhibit significantly lower results in comparison.</p>
<table-wrap id="T11" position="float">
<label>Table&#xa0;11</label>
<caption>
<p>Average and standard deviation of Accuracy and F1-score for Presence (+) and Pureness (*) of <italic>P. fluorescens</italic> experiments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="left">Model</th>
<th valign="middle" colspan="2" align="center">
<italic>P. fluorescens</italic> (+)</th>
<th valign="middle" colspan="2" align="center">
<italic>P. fluorescens</italic> (*)</th>
</tr>
<tr>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">F1-score</th>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">F1-score</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">PCA_SVM (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">0.8318 &#xb1; 0.04</td>
<td valign="middle" align="center">0.7970 &#xb1; 0.04</td>
<td valign="middle" align="center">0.8762 &#xb1; 0.07</td>
<td valign="middle" align="center">0.8509 &#xb1; 0.09</td>
</tr>
<tr>
<td valign="middle" align="left">PCA_LR (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">0.8412 &#xb1; 0.04</td>
<td valign="middle" align="center">0.8054 &#xb1; 0.05</td>
<td valign="middle" align="center">0.8762 &#xb1; 0.07</td>
<td valign="middle" align="center">0.8523 &#xb1; 0.09</td>
</tr>
<tr>
<td valign="middle" align="left">XGBoost (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">0.8178 &#xb1; 0.02</td>
<td valign="middle" align="center">0.7754 &#xb1; 0.02</td>
<td valign="middle" align="center">0.9438 &#xb1; 0.05</td>
<td valign="middle" align="center">0.9381 &#xb1; 0.06</td>
</tr>
<tr>
<td valign="middle" align="left">PLS_DA (<xref ref-type="bibr" rid="B7">Christmann et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">0.9155 &#xb1; 0.04</td>
<td valign="middle" align="center">0.9053 &#xb1; 0.05</td>
<td valign="middle" align="center">0.9591 &#xb1; 0.05</td>
<td valign="middle" align="center">0.9550 &#xb1; 0.06</td>
</tr>
<tr>
<td valign="middle" align="left">CNN2D (<xref ref-type="bibr" rid="B43">Yan et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">
<underline>0.9298</underline> &#xb1; 0.02</td>
<td valign="middle" align="center">0.9193 &#xb1; 0.03</td>
<td valign="middle" align="center">
<underline>0.9733</underline> &#xb1; 0.05</td>
<td valign="middle" align="center">
<underline>0.9700</underline> &#xb1; 0.06</td>
</tr>
<tr>
<td valign="middle" align="left">CNN1D (<xref ref-type="bibr" rid="B43">Yan et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">0.9157 &#xb1; 0.04</td>
<td valign="middle" align="center">0.9012 &#xb1; 0.05</td>
<td valign="middle" align="center">0.9600 &#xb1; 0.08</td>
<td valign="middle" align="center">0.9569 &#xb1; 0.09</td>
</tr>
<tr>
<td valign="middle" align="left">MLP (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">
<underline>0.9298</underline> &#xb1; 0.03</td>
<td valign="middle" align="center">
<underline>0.9208</underline> &#xb1; 0.04</td>
<td valign="middle" align="center">0.9591 &#xb1; 0.05</td>
<td valign="middle" align="center">0.9550 &#xb1; 0.06</td>
</tr>
<tr>
<td valign="middle" align="left">FCNN (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="center">
<bold>0.9344</bold> &#xb1; 0.03</td>
<td valign="middle" align="center">
<bold>0.9270</bold> &#xb1; 0.03</td>
<td valign="middle" align="center">
<bold>0.9867</bold> &#xb1; 0.03</td>
<td valign="middle" align="center">
<bold>0.9856</bold> &#xb1; 0.03</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Bold values indicate the best-performing model, while underlined values indicate the second best-performing model.</p>
</table-wrap-foot>
</table-wrap>
<p>As shown in <xref ref-type="table" rid="T10">
<bold>Table&#xa0;10</bold>
</xref>, the sensitivity and specificity of the best-performing model, FCNN, are smoother compared to previous experiments. The values for the two classes in each experiment alternate between 91.62% and 94.31% for the presence of the pathogenic bacterium and between 100% and 96.67% for the pureness of the culture. These findings demonstrate that the models are highly specific in detecting the pureness of the pathogenic bacterium <italic>P. fluorescens</italic>.</p>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>Evaluation of DL models&#x2019; training time and parameters</title>
<p>To further evaluate the proposed DL baseline models, we further report two key evaluation aspects of DL research. First, we measured the training time for each of the four models: CNN1D, CNN2D, MLP, and FCNN. For each experiment and each iteration of the five-fold cross-validation evaluation method, we recorded the overall training time, compute the average and standard deviation in seconds, and presented the results in <xref ref-type="table" rid="T12">
<bold>Table&#xa0;12</bold>
</xref>.</p>
<table-wrap id="T12" position="float">
<label>Table&#xa0;12</label>
<caption>
<p>Average and standard deviation of training time (s) across all the experiments and number of trainable parameters in respect to the target classes for each DL model.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Model</th>
<th valign="middle" align="left">Training time (s)</th>
<th valign="middle" align="left">Trainable parameters</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">CNN1D (<xref ref-type="bibr" rid="B43">Yan et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="left">58.38 &#xb1; 71.34</td>
<td valign="middle" align="left">93,285,376 + 513 &#xd7; classes</td>
</tr>
<tr>
<td valign="middle" align="left">CNN2D (<xref ref-type="bibr" rid="B43">Yan et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="left">
<underline>43.11 &#xb1; 29.55</underline>
</td>
<td valign="middle" align="left">81,423,360 + 513 &#xd7; classes</td>
</tr>
<tr>
<td valign="middle" align="left">MLP (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="left">48.85 &#xb1; 78.16</td>
<td valign="middle" align="left">
<bold>46,694,912 + 513</bold> &#xd7; classes</td>
</tr>
<tr>
<td valign="middle" align="left">FCNN (<xref ref-type="bibr" rid="B37">Vega-M&#xe1;rquez et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="left">
<bold>24.54 &#xb1; 12.41</bold>
</td>
<td valign="middle" align="left">
<underline>47,482,880 +</underline> 513 &#xd7; classes</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Bold values indicate the best-performing model, while underlined values indicate the second best-performing model.</p>
</table-wrap-foot>
</table-wrap>
<p>Notably, FCNN, apart from being the best-performing model across all experiments, exhibited the fastest training time, averaging 24.54 seconds. This is nearly half the time required by the second-fastest model, CNN2D, which averaged 43.11 seconds. MLP followed with an average training time of 48.85 seconds but showed a high standard deviation of 78.16 seconds, and CNN1D had the longest training time, averaging 58.38 &#xb1; 71.34 seconds.</p>
<p>Additionally, <xref ref-type="table" rid="T12">
<bold>Table&#xa0;12</bold>
</xref> reports the overall number of trainable parameters relative to the number of target classes, as the last hidden layer&#x2019;s parameters depend on the output classes, as illustrated in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>. MLP has the fewest trainable parameters at 46,694,912, closely followed by FCNN with 47,482,880 parameters, presenting only a 1.68% increase. CNN2D nearly doubles this amount with 81,423,360 parameters, while CNN1D has the highest number, requiring 93,285,376 trainable parameters.</p>
<p>These results highlight that FCNN not only achieves the highest accuracy across all experiments but also is trained fastest and is the second most compact model in terms of trainable parameters. MLP and CNN2D perform competitively depending on the experiment, while CNN1D consistently ranks as the least effective model among the four DL baselines, showcasing that CNN models are quite computationally insufficient in the presented tasks.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>This study explored the ability of ML-based and DL-based supervised classification methods in identifying organism-level microbial cultures through their representative VOCs fingerprints. We investigated pure and mixed cultures of four different microorganisms as multi-class classification problems. Additionally, we introduced two new experiments, one identifying between Bacteria and Fungi, while the other distinguishing Gram-positive from Gram-negative bacteria. Finally, we presented the results on identifying two pathogenic bacteria, <italic>Escherichia coli</italic> (highly pathogenic) and <italic>Pseudomonas fluorescens</italic> (low pathogenic), by training models to classify their presence and pureness in various cultures.</p>
<p>To properly evaluate on those experiments, we designed a five-fold cross-validation evaluation protocol for eight different models (PLS_DA, PCA_SVM, PCA_LR, XGBoost, CNN1D, CNN2D, MLP, FCNN), while reporting a wide collection of evaluation metrics. A further evaluation of DL models is conducted to analyze training time and trainable parameters. Based on the reported results, FCNN outperforms the other experimented baselines by achieving the best performance among the models evaluated in this study in terms of overall performance metrics and training time across all experiments, while having a slightly higher parameters count (less than 2%), compared to the lightest model, MLP.</p>
<p>In future work, we plan to incorporate imbalance-aware techniques such as class-weighted losses or focal loss for deep learning models, class weights for machine learning models, and expand the evaluation with metrics like macro-F1, PR-AUC, and 95% confidence intervals computed via bootstrapping across folds to improve and evaluate the models under imbalanced conditions. We also aim to integrate model interpretability methods, such as Grad-CAM or saliency maps for CNN architectures and SHAP for FCNN models, to highlight informative retention-time and drift-time regions, thereby linking predictive features to underlying chemical patterns and ensuring biological plausibility, through various explainable AI techniques.</p>
<p>Finally, due to the significant limitations of the dataset, such as the small number of samples (214 in total) and the imbalance across different experiments, there is a substantial risk of overfitting, which may lead to inflated performance metrics. This limitation makes it difficult to draw definitive conclusions about the generalization capability of our models. While we employed k-fold cross-validation to maximize the use of the limited data in both training and validation sets, we acknowledge that this approach carries a risk of optimistic bias, potential data leakage, or overfitting to instrumentation-specific noise. Therefore, it is important to emphasize that this work represents an early-stage investigation under controlled laboratory conditions and does not constitute clinical validation. To establish real-world applicability, future studies should also include clinically relevant samples processed under different culture media and across multiple GC-IMS instruments and laboratories. A key next step will be the creation of a large-scale, multi-site dataset that incorporates diverse instruments, operators, and sample preparation protocols. Such an effort will be essential to evaluate the robustness, transferability, and generalization of the proposed models.</p>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material. Further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>GK: Writing &#x2013; original draft, Writing &#x2013; review &amp; editing, Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Software, Validation, Visualization. GD: Conceptualization, Funding acquisition, Project administration, Resources, Supervision, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. SK: Funding acquisition, Project administration, Writing &#x2013; review &amp; editing. KI: Funding acquisition, Project administration, Resources, Supervision, Writing &#x2013; review &amp; editing. SV: Funding acquisition, Project administration, Writing &#x2013; review &amp; editing. IK: Funding acquisition, Project administration, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research and/or publication of this article. Funded by the European Union through Grant Agreement 101103176.</p>
</sec>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The author(s) declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec id="s9" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec id="s10" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Author disclaimer</title>
<p>Views and opinions expressed are however those of the author(s) only and do not necessarily reflect those of the European Union or the European Commission. Neither the European Union nor the granting authority can be held responsible for them.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aboutalebian</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ahmadikia</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Fakhim</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Chabavizadeh</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Okhovat</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Nikaeen</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Direct detection and identification of the most common bacteria and fungi causing otitis externa by a stepwise multiplex pcr</article-title>. <source>Front. Cell. Infect. Microbiol.</source> <volume>11</volume>, <elocation-id>644060</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fcimb.2021.644060</pub-id>, PMID: <pub-id pub-id-type="pmid">33842390</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altaee</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Kadhim</surname> <given-names>M. J.</given-names>
</name>
<name>
<surname>Hameed</surname> <given-names>I. H.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Characterization of metabolites produced by e. coli and analysis of its chemical compounds using gc-ms</article-title>. <source>Int. J. Curr. Pharm. Rev. Res.</source> <volume>7</volume>, <fpage>13</fpage>&#x2013;<lpage>19</lpage>.</citation></ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baron</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Introduction to bacteriology</article-title>. <source>BMJ</source> <volume>2</volume>, <fpage>245</fpage>&#x2013;<lpage>245</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1136/bmj.2.5351.245</pub-id>
</citation></ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beleites</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Salzer</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Assessing and improving the stability of chemometric models in small sample size situations</article-title>. <source>Analytical. Bioanal. Chem.</source> <volume>390</volume>, <fpage>1261</fpage>&#x2013;<lpage>1271</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00216-007-1818-6</pub-id>, PMID: <pub-id pub-id-type="pmid">18228011</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chauhan</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Jindal</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Biochemical and molecular methods for bacterial identification</article-title>,&#x201d; in <source>Microbiological methods for environment, food and pharmaceutical analysis</source> <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>, <fpage>425</fpage>&#x2013;<lpage>468</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-030-52024-3_10</pub-id>
</citation></ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Christmann</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Rohn</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Weller</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>gc-ims-tools&#x2013;a new python package for chemometric analysis of gc&#x2013;ims data</article-title>. <source>Food Chem.</source> <volume>394</volume>, <fpage>133476</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.foodchem.2022.133476</pub-id>, PMID: <pub-id pub-id-type="pmid">35717914</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Christmann</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Weber</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Rohn</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Weller</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Nontargeted volatile metabolite screening and microbial contamination detection in fermentation processes by headspace gc-ims</article-title>. <source>Analytical. Chem.</source> <volume>96</volume>, <fpage>3794</fpage>&#x2013;<lpage>3801</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/acs.analchem.3c04857</pub-id>, PMID: <pub-id pub-id-type="pmid">38386844</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Clark</surname> <given-names>C. G.</given-names>
</name>
<name>
<surname>Kruczkiewicz</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Guan</surname> <given-names>C.</given-names>
</name>
<name>
<surname>McCorrister</surname> <given-names>S. J.</given-names>
</name>
<name>
<surname>Chong</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Wylie</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2013</year>). <article-title>Evaluation of maldi-tof mass spectroscopy methods for determination of escherichia coli pathotypes</article-title>. <source>J. Microbiol. Methods</source> <volume>94</volume>, <fpage>180</fpage>&#x2013;<lpage>191</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.mimet.2013.06.020</pub-id>, PMID: <pub-id pub-id-type="pmid">23816532</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dingle</surname> <given-names>T. C.</given-names>
</name>
<name>
<surname>Butler-Wu</surname> <given-names>S. M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Maldi-tof mass spectrometry for microorganism identification</article-title>. <source>Clinics Lab. Med.</source> <volume>33</volume>, <fpage>589</fpage>&#x2013;<lpage>609</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cll.2013.03.001</pub-id>, PMID: <pub-id pub-id-type="pmid">23931840</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Drees</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Vautz</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Liedtke</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Rosin</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Althoff</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Lippmann</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>Gc-ims headspace analyses allow early recognition of bacterial growth and rapid pathogen differentiation in standard blood cultures</article-title>. <source>Appl. Microbiol. Biotechnol.</source> <volume>103</volume>, <fpage>9091</fpage>&#x2013;<lpage>9101</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00253-019-10181-x</pub-id>, PMID: <pub-id pub-id-type="pmid">31664484</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Duriez</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Armengaud</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Fenaille</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Ezan</surname> <given-names>E.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Mass spectrometry for the detection of bioterrorism agents: from environmental to clinical applications</article-title>. <source>J. Mass. Spectromet.</source> <volume>51</volume>, <fpage>183</fpage>&#x2013;<lpage>199</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/jms.3747</pub-id>, PMID: <pub-id pub-id-type="pmid">26956386</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Dybwad</surname> <given-names>M.</given-names>
</name>
<name>
<surname>van der Laaken</surname> <given-names>A. L.</given-names>
</name>
<name>
<surname>Blatny</surname> <given-names>J. M.</given-names>
</name>
<name>
<surname>Paauw</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Rapid identification of bacillus anthracis spores in suspicious powder samples by using matrix-assisted laser desorption ionization&#x2013;time of flight mass spectrometry (MALDI-TOF MS)</article-title>. <source>Applied and Environmental Microbiology</source>, <volume>79</volume> (<issue>17</issue>), <fpage>5372</fpage>&#x2013;<lpage>5383</lpage>. <publisher-loc>Washington</publisher-loc>: <publisher-loc>American Society for Microbiology (ASM)</publisher-loc>., PMID: <pub-id pub-id-type="pmid">23811517</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Esbensen</surname> <given-names>K. H.</given-names>
</name>
<name>
<surname>Geladi</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Principles of proper validation: use and abuse of re-sampling for validation</article-title>. <source>J. Chemometr.</source> <volume>24</volume>, <fpage>168</fpage>&#x2013;<lpage>187</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/cem.1310</pub-id>
</citation></ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Ding</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>P.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Effective discrimination of yersinia pestis and yersinia pseudotuberculosis by maldi-tof ms using multivariate analysis</article-title>. <source>Talanta</source> <volume>234</volume>, <fpage>122640</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.talanta.2021.122640</pub-id>, PMID: <pub-id pub-id-type="pmid">34364449</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gallien</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Duriez</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Crone</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Kellmann</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Moehring</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Domon</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Targeted proteomic quantification on quadrupole-orbitrap mass spectrometer</article-title>. <source>Mol. Cell. Proteomics</source> <volume>11</volume>, <fpage>1709</fpage>&#x2013;<lpage>1723</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1074/mcp.O112.019802</pub-id>, PMID: <pub-id pub-id-type="pmid">22962056</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gerhardt</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Schwolow</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Rohn</surname> <given-names>S.</given-names>
</name>
<name>
<surname>P&#xe9;rez-Cacho</surname> <given-names>P. R.</given-names>
</name>
<name>
<surname>Gal&#xe1;n-Soldevilla</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Arce</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>Quality assessment of olive oils based on temperature-ramped hs-gc-ims and sensory evaluation: Comparison of different processing approaches by lda, knn, and svm</article-title>. <source>Food Chem.</source> <volume>278</volume>, <fpage>720</fpage>&#x2013;<lpage>728</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.foodchem.2018.11.095</pub-id>, PMID: <pub-id pub-id-type="pmid">30583434</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Giuliano</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Patel</surname> <given-names>C. R.</given-names>
</name>
<name>
<surname>Kale-Pradhan</surname> <given-names>P. B.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Kale-Pradhan PB. A guide to bacterial culture identification and results interpretation</article-title>. <source>Pharm. Ther.</source> <volume>44</volume>, <fpage>192</fpage>., PMID: <pub-id pub-id-type="pmid">30930604</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Du</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Recent development of hs-gc-ims technology in rapid and non-destructive detection of quality and contamination in agri-food products</article-title>. <source>TrAC. Trends Analytical. Chem.</source> <volume>144</volume>, <fpage>116435</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.trac.2021.116435</pub-id>
</citation></ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hameed</surname> <given-names>R. H.</given-names>
</name>
<name>
<surname>Abbas</surname> <given-names>F. M.</given-names>
</name>
<name>
<surname>Hameed</surname> <given-names>I. H.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Analysis of secondary metabolites released by pseudomonas fluorescens using gc-ms technique and determination of its anti-fungal activity</article-title>. <source>Indian J. Public Health Res. Dev.</source> <volume>9</volume>, <fpage>449</fpage>&#x2013;<lpage>455</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.5958/0976-5506.2018.00485.0</pub-id>
</citation></ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ishii</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Kushima</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Koide</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Kinoshita</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Pseudomonas fluorescens pneumonia</article-title>. <source>Int. J. Infect. Dis.</source> <volume>140</volume>, <fpage>92</fpage>&#x2013;<lpage>94</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ijid.2024.01.007</pub-id>, PMID: <pub-id pub-id-type="pmid">38218379</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jeon</surname> <given-names>J. H.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>J. S.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>Z. H.</given-names>
</name>
<name>
<surname>Jung</surname> <given-names>J. Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Complete genome sequence of levilactobacillus brevis nsmj23, makgeolli isolate with antimicrobial activity</article-title>. <source>Microbiol. Resour. Announcements.</source> <volume>13</volume>, <fpage>e01060</fpage>&#x2013;<lpage>e01023</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1128/mra.01060-23</pub-id>, PMID: <pub-id pub-id-type="pmid">38179912</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ju</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Lian</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Ge</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Identification of rice varieties and adulteration using gas chromatography-ion mobility spectrometry</article-title>. <source>IEEE Access</source> <volume>9</volume>, <fpage>18222</fpage>&#x2013;<lpage>18234</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/Access.6287639</pub-id>
</citation></ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname> <given-names>S. O.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>S. S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Bacterial pathogen detection by conventional culture-based and recent alternative (polymerase chain reaction, isothermal amplification, enzyme linked immunosorbent assay, bacteriophage amplification, and gold nanoparticle aggregation) methods in food samples: A review</article-title>. <source>J. Food Saf.</source> <volume>41</volume>, <fpage>e12870</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/jfs.12870</pub-id>
</citation></ref>
<ref id="B24">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kirtsanis</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Dolias</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Kintzios</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ioannidis</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Vrochidis</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Kompatsiaris</surname> <given-names>I.</given-names>
</name>
</person-group> (<year>2025</year>). &#x201c;<article-title>Cnn-based deep autoencoders for limited gas chromatography-ion mobility spectrometry data</article-title>,&#x201d; in <conf-name>2025 IEEE International Instrumentation and Measurement Technology Conference (I2MTC)</conf-name>. <fpage>1</fpage>&#x2013;<lpage>6</lpage> (<publisher-name>IEEE</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1109/I2MTC62753.2025.11079137</pub-id>
</citation></ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kohavi</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>1995</year>). <source>A study of cross-validation and bootstrap for accuracy estimation and model selection</source> Vol. 14 (<publisher-loc>Montreal, Canada</publisher-loc>: <publisher-name>Ijcai</publisher-name>) <volume>14</volume> (<issue>2</issue>), <fpage>1137</fpage>&#x2013;<lpage>1145</lpage>.</citation></ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lasch</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Drevinek</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Nattermann</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Grunow</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Stammler</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Dieckmann</surname> <given-names>R.</given-names>
</name>
<etal/>
</person-group>. (<year>2010</year>). <article-title>Characterization of yersinia using maldi-tof mass spectrometry and chemometrics</article-title>. <source>Analytical. Chem.</source> <volume>82</volume>, <fpage>8464</fpage>&#x2013;<lpage>8475</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/ac101036s</pub-id>, PMID: <pub-id pub-id-type="pmid">20866090</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Fang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Luan</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>An improved yolov3 model for detection of invasive saccharomyces cerevisiae infections</article-title>,&#x201d; in <source>Multimedia Tools and Applications</source> <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>, <fpage>1</fpage>&#x2013;<lpage>18</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11042-024-19649-z</pub-id>
</citation></ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zeng</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>B.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Use of gc-ims for detection of volatile organic compounds to identify mixed bacterial culture medium</article-title>. <source>Amb. Express.</source> <volume>12</volume>, <fpage>31</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13568-022-01367-0</pub-id>, PMID: <pub-id pub-id-type="pmid">35244795</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Nunes</surname> <given-names>A. L.</given-names>
</name>
<name>
<surname>Perna</surname> <given-names>O. F.</given-names>
</name>
<name>
<surname>Queiroz</surname> <given-names>M. S.</given-names>
</name>
<name>
<surname>Zaro</surname> <given-names>G. C.</given-names>
</name>
<name>
<surname>de Lima</surname> <given-names>J. D.</given-names>
</name>
<name>
<surname>da Silva</surname> <given-names>G. J.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Role of pseudomonas fluorescens secondary metabolites in agroecosystem applications</article-title>,&#x201d; in <source>Bacterial secondary Metabolites</source> (<publisher-loc>Amsterdam</publisher-loc>: <publisher-name>Elsevier</publisher-name>), <fpage>211</fpage>&#x2013;<lpage>220</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/B978-0-323-95251-4.00008-9</pub-id>
</citation></ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pohanka</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Current trends in the biosensors for biological warfare agents assay</article-title>. <source>Materials</source> <volume>12</volume>, <fpage>2303</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/ma12142303</pub-id>, PMID: <pub-id pub-id-type="pmid">31323857</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Raschka</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Model evaluation, model selection, and algorithm selection in machine learning</article-title>. <source>arXiv. preprint. arXiv:1811.12808</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1811.12808</pub-id>
</citation></ref>
<ref id="B32">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Refaeilzadeh</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>Cross-validation</article-title>,&#x201d; in <source>Encyclopedia of database systems</source> (<publisher-loc>Boston, MA</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>532</fpage>&#x2013;<lpage>538</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-0-387-39940-9_565</pub-id>
</citation></ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rezaei</surname> <given-names>F. Y.</given-names>
</name>
<name>
<surname>Pircheraghi</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Nikbin</surname> <given-names>V. S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Antibacterial activity, cell wall damage, and cytotoxicity of zinc oxide nanospheres, nanorods, and nanoflowers</article-title>. <source>ACS Appl. Nano. Mater.</source> <volume>7</volume>, <fpage>15242</fpage>&#x2013;<lpage>15254</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/acsanm.4c02046</pub-id>
</citation></ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sauer</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Kliem</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Mass spectrometry tools for the classification and identification of bacteria</article-title>. <source>Nat. Rev. Microbiol.</source> <volume>8</volume>, <fpage>74</fpage>&#x2013;<lpage>82</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nrmicro2243</pub-id>, PMID: <pub-id pub-id-type="pmid">20010952</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Su</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>Z. H.</given-names>
</name>
<name>
<surname>Chiou</surname> <given-names>S. F.</given-names>
</name>
<name>
<surname>Shiea</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>D. C.</given-names>
</name>
<name>
<surname>Tseng</surname> <given-names>S. P.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Rapid characterization of bacterial lipids with ambient ionization mass spectrometry for species differentiation</article-title>. <source>Molecules</source> <volume>27</volume>, <fpage>2772</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/molecules27092772</pub-id>, PMID: <pub-id pub-id-type="pmid">35566120</pub-id></citation></ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tait</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Perry</surname> <given-names>J. D.</given-names>
</name>
<name>
<surname>Stanforth</surname> <given-names>S. P.</given-names>
</name>
<name>
<surname>Dean</surname> <given-names>J. R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Identification of volatile organic compounds produced by bacteria using hs-spme-gc&#x2013;ms</article-title>. <source>J. Chromatogr. Sci.</source> <volume>52</volume>, <fpage>363</fpage>&#x2013;<lpage>373</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/chromsci/bmt042</pub-id>, PMID: <pub-id pub-id-type="pmid">23661670</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vega-M&#xe1;rquez</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Nepomuceno-Chamorro</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Jurado-Campos</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Rubio-Escudero</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Deep learning techniques to improve the performance of olive oil classification</article-title>. <source>Front. Chem.</source> <volume>7</volume>, <elocation-id>929</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fchem.2019.00929</pub-id>, PMID: <pub-id pub-id-type="pmid">32010673</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>R. Y.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>Y. Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Gu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Pattern recognition analysis of metabolites in escherichia coli based on esi-orbitrap mass spectrometry</article-title>. <source>Chem. Biodivers.</source> <volume>20</volume>, <fpage>e202201153</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/cbdv.202201153</pub-id>, PMID: <pub-id pub-id-type="pmid">37081715</pub-id></citation></ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Recent progress in food flavor analysis using gas chromatography&#x2013;ion mobility spectrometry (gc&#x2013;ims)</article-title>. <source>Food Chem.</source> <volume>315</volume>, <fpage>126158</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.foodchem.2019.126158</pub-id>, PMID: <pub-id pub-id-type="pmid">32014672</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weller</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Christmann</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Hs-gc-ims data of fermentations of different organisms</article-title>. <source>Mendeley Data</source>, <fpage>V1</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.17632/v9gxkpdp3c.1</pub-id>
</citation></ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Westad</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Marini</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Validation of chemometric models&#x2013;a tutorial</article-title>. <source>Analytica. Chim. Acta</source> <volume>893</volume>, <fpage>14</fpage>&#x2013;<lpage>24</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.aca.2015.06.056</pub-id>, PMID: <pub-id pub-id-type="pmid">26398418</pub-id></citation></ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wynne</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Edwards</surname> <given-names>N. J.</given-names>
</name>
<name>
<surname>Fenselau</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Phyloproteomic classification of unsequenced organisms by top-down identification of bacterial proteins using caplc-ms/ms on an orbitrap</article-title>. <source>Proteomics</source> <volume>10</volume>, <fpage>3631</fpage>&#x2013;<lpage>3643</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/pmic.201000172</pub-id>, PMID: <pub-id pub-id-type="pmid">20845332</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Zeng</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>B.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Rapid bacterial identification through volatile organic compound analysis and deep learning</article-title>. <source>BMC Bioinf.</source> <volume>25</volume>, <fpage>347</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12859-024-05967-4</pub-id>, PMID: <pub-id pub-id-type="pmid">39506632</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>S. C.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>C. H.</given-names>
</name>
<name>
<surname>Aljuffali</surname> <given-names>I. A.</given-names>
</name>
<name>
<surname>Fang</surname> <given-names>J. Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Current pathogenic escherichia coli foodborne outbreak cases and therapy development</article-title>. <source>Arch. Microbiol.</source> <volume>199</volume>, <fpage>811</fpage>&#x2013;<lpage>825</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00203-017-1393-y</pub-id>, PMID: <pub-id pub-id-type="pmid">28597303</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Su</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Ying</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Smart nanozymes for diagnosis of bacterial infection: The next frontier from laboratory to bedside testing</article-title>. <source>ACS Appl. Mater. Interfaces.</source> <volume>16</volume>, <fpage>44361</fpage>&#x2013;<lpage>44375</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/acsami.4c07043</pub-id>, PMID: <pub-id pub-id-type="pmid">39162136</pub-id></citation></ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Lian</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Recognition of rice species based on gas chromatography-ion mobility spectrometry and deep learning</article-title>. <source>Agriculture</source> <volume>14</volume>, <fpage>1552</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agriculture14091552</pub-id>
</citation></ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zukowska</surname> <given-names>M. E.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Advanced methods of bacteriological identification in a clinical microbiology laboratory</article-title>. <source>J. Pre-Clin. Clin. Res.</source> <volume>15</volume> (<issue>2</issue>), <fpage>68</fpage>&#x2013;<lpage>72</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.26444/jpccr/134646</pub-id>
</citation></ref>
</ref-list>
</back>
</article>