<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1651917</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2025.1651917</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>aiGeneR 3.0: an enhanced deep network model for resistant strain identification and multi-drug resistance prediction in <italic>Escherichia coli</italic> causing urinary tract infection using next-generation sequencing data</article-title>
<alt-title alt-title-type="left-running-head">Nayak et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2025.1651917">10.3389/fgene.2025.1651917</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Nayak</surname>
<given-names>Debasish Swapnesh Kumar</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1745910/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Pati</surname>
<given-names>Abhilash</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2607150/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Panigrahi</surname>
<given-names>Amrutanshu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2683664/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Khan</surname>
<given-names>Mudassir</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2743111/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Alabdullah</surname>
<given-names>Bayan</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sahoo</surname>
<given-names>Santanu Kumar</given-names>
</name>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3240322/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sahu</surname>
<given-names>Bibhuprasad</given-names>
</name>
<xref ref-type="aff" rid="aff8">
<sup>8</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2916486/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Almjally</surname>
<given-names>Abrar</given-names>
</name>
<xref ref-type="aff" rid="aff9">
<sup>9</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Mallik</surname>
<given-names>Saurav</given-names>
</name>
<xref ref-type="aff" rid="aff10">
<sup>10</sup>
</xref>
<xref ref-type="aff" rid="aff11">
<sup>11</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/635395/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Swarnkar</surname>
<given-names>Tripti</given-names>
</name>
<xref ref-type="aff" rid="aff12">
<sup>12</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3238278/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Computer Science &#x26; Engineering, Siksha &#x201c;O&#x201d; Anusandhan (Deemed to be University)</institution>, <addr-line>Bhubaneswar</addr-line>, <addr-line>Odisha</addr-line>, <country>India</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Computer Science and Engineering, Centurion University of Technology and Management</institution>, <addr-line>Bhubaneswar</addr-line>, <addr-line>Odisha</addr-line>, <country>India</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Computer Science and Engineering, Indian Institute of Technology</institution>, <addr-line>Bhilai</addr-line>, <addr-line>Chhattisgarh</addr-line>, <country>India</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Computer Science, College of Computer Science, Applied College Tanumah, King Khalid University</institution>, <addr-line>Abha</addr-line>, <country>Saudi Arabia</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Jadara University Research Center, Jadara University</institution>, <addr-line>Irbid</addr-line>, <country>Jordan</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Department of Information Systems, College of Computer and Information Sciences, Princess Nourah Bint Abdulrahman University</institution>, <addr-line>Riyadh</addr-line>, <country>Saudi Arabia</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>Department of Electronics and Communication Engineering, Siksha &#x201c;O&#x201d; Anusandhan (Deemed to be University)</institution>, <addr-line>Bhubaneswar</addr-line>, <addr-line>Odisha</addr-line>, <country>India</country>
</aff>
<aff id="aff8">
<sup>8</sup>
<institution>Symbiosis Institute of Technology, Hyderabad Campus, Symbiosis International University</institution>, <addr-line>Pune</addr-line>, <country>India</country>
</aff>
<aff id="aff9">
<sup>9</sup>
<institution>College of Computer and Information Sciences, Imam Mohammad Ibn Saud Islamic University (IMSIU)</institution>, <addr-line>Riyadh</addr-line>, <country>Saudi Arabia</country>
</aff>
<aff id="aff10">
<sup>10</sup>
<institution>Department of Environmental Health, Harvard T H Chan School of Public Health</institution>, <addr-line>Boston</addr-line>, <addr-line>MA</addr-line>, <country>United States</country>
</aff>
<aff id="aff11">
<sup>11</sup>
<institution>Department of Pharmacology &#x26; Toxicology, University of Arizona</institution>, <addr-line>Tucson</addr-line>, <addr-line>AZ</addr-line>, <country>United States</country>
</aff>
<aff id="aff12">
<sup>12</sup>
<institution>Department of Computer Application, National Institute of Technology</institution>, <addr-line>Raipur</addr-line>, <country>India</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/601790/overview">Francesco Asnicar</ext-link>, University of Trento, Italy</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2962898/overview">Vinothkumar Kolluru</ext-link>, Stevens Institute of Technology, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3138028/overview">Volkan Alparslan</ext-link>, Kocaeli University Faculty of Medicine, T&#xfc;rkiye</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Saurav Mallik, <email>sauravmtech2@gmail.com</email>, <email>smallik@arizona.edu</email>, <email>smallik@hsph.harvard.edu</email>; Mudassir Khan, <email>mudassirkhan12@gmail.com</email>, <email>mkmiyob@kku.edu.sa</email>; Abhilash Pati, <email>er.abhilash.pati@gmail.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>23</day>
<month>10</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1651917</elocation-id>
<history>
<date date-type="received">
<day>22</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>10</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Nayak, Pati, Panigrahi, Khan, Alabdullah, Sahoo, Sahu, Almjally, Mallik and Swarnkar.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Nayak, Pati, Panigrahi, Khan, Alabdullah, Sahoo, Sahu, Almjally, Mallik and Swarnkar</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Infectious diseases pose a global health threat, with antimicrobial resistance (AMR) exacerbating the issue. Considering <italic>Escherichia coli</italic> (<italic>E. coli</italic>) is frequently linked to urinary tract infections, researching antibiotic resistance genes in this context is essential for identifying and combating the growing problem of drug resistance.</p>
</sec>
<sec>
<title>Objective</title>
<p>Machine learning (ML), particularly deep learning (DL), has proven effective in rapidly detecting strains for infection prevention and reducing mortality rates. We proposed aiGeneR 3.0, a simplified and effective DL model employing a long-short-term memory mechanism for identifying multi-drug resistant and resistant strains in <italic>E. coli</italic>. The aiGeneR 3.0 paradigm for identifying and classifying antibiotic resistance is a tandem link of quality control incorporated with DL models. Cross-validation was adopted to measure the ROC-AUC, F1-score, accuracy, precision, sensitivity, specificity, and overall classification performance of aiGeneR 3.0. We hypothesized that the aiGeneR 3.0 would be more effective than other baseline DL models for antibiotic resistance detection with an effective computational cost. We assess how well our model can be memorized and generalized.</p>
</sec>
<sec>
<title>Results</title>
<p>Our aiGeneR 3.0 can handle imbalances and small datasets, offering higher classification accuracy (93%) with a simple model architecture. The multi-drug resistance prediction ability of aiGeneR 3.0 has a prediction accuracy of 98%. aiGeneR 3.0 uses deep networks (LSTM) with next-generation sequencing (NGS) data, making it suitable for novel antibiotics and growing resistance identification in the future.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>This work uniquely integrates SNP-level insights with DL, offering potential clinical utility in guiding antibiotic stewardship. It also enables a robust, generalized, and memorized model for future use in AMR analysis.</p>
</sec>
</abstract>
<kwd-group>
<kwd>deep learning</kwd>
<kwd>machine learning</kwd>
<kwd>next-generation sequencing</kwd>
<kwd>antimicrobial resistance</kwd>
<kwd>antibiotic resistance genes</kwd>
</kwd-group>
<contract-sponsor id="cn001">Princess Nourah Bint Abdulrahman University<named-content content-type="fundref-id">10.13039/501100004242</named-content>
</contract-sponsor>
<counts>
<page-count count="22"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Pharmacogenetics and Pharmacogenomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>One of the biggest concerns for global public healthcare is the issue of diseases brought on by bacteria that are resistant to antibiotics, often known as antimicrobial resistance (AMR). According to estimates from the World Health Organization (WHO), there were over 700,000 fatalities from drug-resistant illnesses in 2019, and that number might increase to 10 million deaths by 2050 (<xref ref-type="bibr" rid="B46">Sharma et al., 2024</xref>; <xref ref-type="bibr" rid="B6">Chandra et al., 2021</xref>). Identifying antibiotic resistance genes (ARGs) is important for discovering the patterns of AMR and plays a key role in personalized treatment and drug discovery.</p>
<p>Urinary tract infections (UTIs) are among the many infectious diseases that pose a serious threat to world health (<xref ref-type="bibr" rid="B51">Tan and Chlebicki, 2016</xref>). <italic>Escherichia coli</italic> (<italic>E. coli</italic>) bacteria are the main cause of UTIs, which affect millions of people each year (<xref ref-type="bibr" rid="B55">Vasudevan, 2014</xref>). If left untreated, many infections that affect the urinary system carry the potential to cause consequences like kidney damage. The problem is heightened by the advent of antibiotic resistance in <italic>E. Coli</italic> strains (<xref ref-type="bibr" rid="B53">Totsika et al., 2012</xref>), which restricts available treatments and calls into question accepted ideas of antimicrobial stewardship (<xref ref-type="bibr" rid="B38">Niranjan and Malini, 2014</xref>). <italic>E. coli</italic> is the primary cause of UTIs and provides a statistical analysis of other bacteria that can cause UTIs, as shown in a short research conducted in the northern region of India. <italic>E. coli</italic> (76.60%) was the most common gram-negative bacterium among the 47 positive isolates out of a total of 83 positive samples (<xref ref-type="bibr" rid="B10">Das et al., 2018</xref>). As shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, <italic>E. coli</italic> is the main cause of UTI in more than 53% of the cases, which is significant and needs to be addressed in the AMR pattern, antibiotic-resistant strains, and ARGs in <italic>E. coli</italic> for further effective drug development.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Most UTI cases are with selected bacteria.</p>
</caption>
<graphic xlink:href="fgene-16-1651917-g001.tif">
<alt-text content-type="machine-generated">Pie chart showing bacterial distribution: Escherichia coli at 56% (red), Klebsiella pneumoniae at 29% (blue), Pseudomonas aeruginosa at 10% (purple), and Proteus mirabilis at 5% (yellow).</alt-text>
</graphic>
</fig>
<p>The robust identification of antibiotic resistance determinants and their curation in specialized databases has been made possible by the growing availability and affordability of whole-genome sequencing data from clinical strains. Computational techniques can then search these resources for known causative genes, given the sequence from a new strain (<xref ref-type="bibr" rid="B31">McArthur et al., 2013</xref>; <xref ref-type="bibr" rid="B58">Zankari et al., 2012</xref>; <xref ref-type="bibr" rid="B49">Stoesser et al., 2013</xref>). By detecting mutations, examining entire genomes, and pinpointing particular resistance genes, the genetic study of <italic>E. coli</italic> shows antibiotic resistance patterns. To effectively tackle the global challenge of antibiotic resistance, this technique aids in understanding the genetic basis of resistance, tracking its spread, and predicting emerging patterns. This information guides focused treatments for antibiotic stewardship (<xref ref-type="bibr" rid="B54">Truswell et al., 2023</xref>; <xref ref-type="bibr" rid="B57">Wilson et al., 2016</xref>; <xref ref-type="bibr" rid="B30">Malekian et al., 2022</xref>). Antibiotic resistance, particularly in bacteria like <italic>E. coli</italic>, makes urinary tract infections (UTIs) a serious concern to human health. UTIs are among the most prevalent bacterial illnesses worldwide, with millions of cases occurring yearly. UTIs can cause serious side effects such as kidney infections, sepsis, and long-term harm to the urinary system if they are not addressed. The danger is exacerbated by introducing strains resistant to antibiotics, particularly in <italic>E. coli</italic> (<xref ref-type="bibr" rid="B44">Reza Asadi et al., 2019</xref>; <xref ref-type="bibr" rid="B20">Jafri et al., 2014</xref>; <xref ref-type="bibr" rid="B5">Bryce et al., 2016</xref>).</p>
<p>Novel approaches are needed to address the growing epidemic of antibiotic resistance in urinary tract infections (UTIs). Through genetic insights, DNA data advances several sectors, including disease diagnosis, customized medicine, AMR analysis, and microbial diversity study (<xref ref-type="bibr" rid="B45">Satam et al., 2023</xref>). Due to its capacity to handle high-dimensional data, identify intricate correlations, and integrate various data sources, deep learning (DL) performs very well when evaluating DNA sequencing data for the identification of antibiotic resistance (<xref ref-type="bibr" rid="B29">Lueftinger et al., 2021</xref>; <xref ref-type="bibr" rid="B47">Shi et al., 2019</xref>). DL is a revolutionary method that reduces the need for wasteful antibiotic treatment by providing precision medicine through the identification of distinct resistance profiles (<xref ref-type="bibr" rid="B5">Bryce et al., 2016</xref>; <xref ref-type="bibr" rid="B52">Taylor et al., 2018</xref>). Real-time decision assistance is empowered by DL, allowing for quick and knowledgeable antibiotic selection decisions. Additionally, it makes it easier to identify new resistance trends early on, which supports preventative measures (<xref ref-type="bibr" rid="B42">Ren et al., 2022a</xref>; <xref ref-type="bibr" rid="B43">Ren et al., 2022b</xref>). This work holds the potential to completely transform the way that UTIs are managed and the identification of resistance patterns in <italic>E. coli</italic> utilizing the next-generation WGS data, providing efficient solutions to the ever-changing problem of antibiotic resistance. We proposed our aiGeneR 3.0 model, which can identify the multidrug resistance genes in <italic>E. coli</italic>. In our work, we deal with a highly imbalanced and small dataset to assess the efficacy of our aiGeneR 3.0 model. We also compare the performance of aiGeneR 3.0 with well-accepted state-of-the-art ML and DL models. The generalization of our model boosts the adaptability and robustness. The simplified architecture and less computational time are the major advantages of our aiGeneR 3.0 model. We hypothesized that the aiGeneR 3.0 can reduce the cost and time for multi-drug resistance identification utilizing the WGS data. The dataset (NGS single-nucleotide polymorphism (SNP) WGS) utilized in this work is small and imbalanced; still, our aiGeneR 3.0 performs exceptionally well; the ROC value achieved during the deployment phase has already proven this. The detailed architecture of our study is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The overall architecture of our study.</p>
</caption>
<graphic xlink:href="fgene-16-1651917-g002.tif">
<alt-text content-type="machine-generated">Flowchart of aiGeneR 3.0 illustrating the process for predicting E. coli resistance. E. coli WGS data undergoes imputation, normalization, and encoding, preparing train and test data. Models like SVM and CNN are applied, resulting in predictive models indicating susceptibility or resistance. Hypotheses include improved accuracy, handling dataset imbalances, and analyzing AMR patterns.</alt-text>
</graphic>
</fig>
<p>The following describes the paper&#x2019;s structure and major contributions. The relevant work for classifying and identifying <italic>E. coli</italic> antibiotic resistance is included in <xref ref-type="sec" rid="s2">Section 2</xref> to set up our research pipeline. We go over the aiGeneR 3.0 content and overall design in <xref ref-type="sec" rid="s3">Section 3</xref>. The AI models and the experimental technique are presented in <xref ref-type="sec" rid="s4">Section 4</xref>. <xref ref-type="sec" rid="s5">Section 5</xref> has the experimental results presentation. The validation and discussion of our aiGeneR 3.0 outcome are conducted in <xref ref-type="sec" rid="s6">Section 6</xref> and <xref ref-type="sec" rid="s7">Section 7</xref>, respectively. We benchmarked our aiGeneR 3.0 in <xref ref-type="sec" rid="s8">Section 8</xref>, and <xref ref-type="sec" rid="s9">Section 9</xref> held the conclusion.</p>
</sec>
<sec id="s2">
<title>2 Literature surveys</title>
<p>Researchers <xref ref-type="bibr" rid="B34">Moradigaravand et al. (2018)</xref> used gradient-boosted decision trees to achieve a 91% success rate in predicting antibiotic resistance in 1,681 <italic>Escherichia coli</italic> strains. Researchers found that using population structure and gene content greatly improved prediction accuracy. Based on these findings, machine learning (ML) shows promise as a clinical tool for identifying antibiotic resistance. Introduced by <xref ref-type="bibr" rid="B2">Arango-Argoty et al. (2018)</xref>, the DeepARG-SS model outperformed conventional approaches with a recall of 91% and an accuracy of 97% over 30 antibiotic categories. Applying the DeepARG-LS model to the MEGARes database confirmed its great recall and accuracy. When used in conjunction with the DeepARG-DB database, these models allow for more precise gene identification by producing predictions of antibiotic resistance genes. The difficulties and limits of using ML to forecast antibiotic resistance were addressed by <xref ref-type="bibr" rid="B4">Boolchandani et al. (2019)</xref>. In order to improve the accuracy of predictions, the study highlighted the necessity for extensive databases that connect resistance genes to test results. The significance of continuously improving computational methods to combat antibiotic resistance was highlighted by recognizing Resfams, Resfinder, and CARD as effective techniques for finding resistance genes. Among the multi-label classification models used by <xref ref-type="bibr" rid="B42">Ren et al. (2022a)</xref> to forecast <italic>E. coli</italic> multi-drug resistance, the ECC model proved to be the most accurate. In order to have a whole picture of resistance, the study stressed the significance of non-chromosomal genetic variables. Researchers <xref ref-type="bibr" rid="B18">Gunasekaran et al. (2021)</xref> used DL methods to classify DNA sequences, successfully determining the origins of viruses and DNA mutations with a high degree of accuracy. This research proved that DL could be useful for a variety of genetic analysis, drug discovery, and viral identification tasks. The accuracy of antimicrobial resistance predictions for underrepresented groups was greatly enhanced by the deep transfer learning model put forth by <xref ref-type="bibr" rid="B43">Ren et al. (2022b)</xref> while dealing with tiny, imbalanced datasets. Rapid diagnosis and focused therapies could both benefit from this strategy.</p>
<p>Over the last decade, various tools, quality control pipelines, and AI models have been gaining attention in AMR analysis. The AMR mechanism is too complex and requires trained manpower to access the laboratory test for the identification of resistance patterns, resistance strains, and multi-drug resistant percentages. In addition to this, the procedure of resistant strain identification is associated with massive cost and time. However, AI models have been found to perform well compared to traditional approaches for resistant strain identification. It was also found that existing AI studies for resistant strain identification lack a comparative analysis that includes ensemble and simplified model architectures. In addition to this, the computational cost associated with the resistant strain identification is still an open issue. This study aims to bridge this gap and provide evidence for the superiority of ensemble-based DL, transfer learning, and solo simplified architecture-based DL models regarding prediction accuracy and computational cost. Additionally, it is observed from the literature that researchers used transfer learning (TL) on a small dataset to identify the resistant strains. We aim to achieve a more effective outcome with less model complexity and computational time. Interventional studies involving animals or humans, as well as other studies that require ethical approval, must list the authority that provided approval and the corresponding ethical approval code.</p>
<p>In this section, where applicable, authors are required to disclose details of how generative artificial intelligence (GenAI) has been used in this paper (e.g., to generate text, data, or graphics, or to assist in study design, data collection, analysis, or interpretation). The use of GenAI for superficial text editing (e.g., grammar, spelling, punctuation, and formatting) does not need to be declared.</p>
</sec>
<sec sec-type="materials|methods" id="s3">
<title>3 Materials and methods</title>
<p>The methods, resources, and materials used in this study to accomplish the study&#x2019;s goals are described in this section. This section seeks to present a clear and thorough explanation of the experimental design, data collection, and data analysis methods.</p>
<sec id="s3-1">
<title>3.1 Dataset</title>
<p>The <italic>E. coli</italic> WGS dataset utilized in this study is openly available and was collected from <xref ref-type="bibr" rid="B15">GitHub (2025)</xref>, and <xref ref-type="bibr" rid="B34">Moradigaravand et al. (2018)</xref>. Both these datasets have the susceptible and resistant information of the WGS of the <italic>E. coli</italic> K-12 strain. Due to the common association between these mutations and increased antibiotic resistance in both environmental and clinical contexts, the double-mutated <italic>E. coli</italic> genome dataset was chosen for its practical and clinical importance. The genetic variety of resistant strains is reflected in this dataset, which captures changes linked to resistance to several classes of antibiotics. This provides valuable insights into the complicated processes of resistance. Previous studies have mostly concentrated on single mutations or resistance genes specific to individual isolates; this work fills a significant void by shifting the focus to double mutations.</p>
</sec>
<sec id="s3-2">
<title>3.2 Dataset collection</title>
<p>We employed two datasets of <italic>E. coli</italic> in this study, which included WGS, SNP, and resistance-susceptible data for four antibiotics: gentamicin (GEN), cefotaxime (CTX), ciprofloxacin (CIP), and ceftazidime (CTZ), as shown in <xref ref-type="table" rid="T1">Table 1</xref>. The first dataset has 809 <italic>E. Coli</italic> strains, which are generated by <xref ref-type="bibr" rid="B43">Ren et al. (2022b)</xref>. Clinical samples from both humans and animals were used to get the isolates. Using the VITEK<sup>&#xae;</sup> 2 system (bioM&#xe9;rieux, Nurtingen, Germany), antimicrobial susceptibility testing was carried out, and results were evaluated by EUCAST criteria. The proportion of isolates resistant to CTX, GEN, CTZ, and CIP is 23%, 44%, 34%, and 45% in that sequence.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Strain distribution to all the studied antibiotics.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Antibiotics</th>
<th align="center">GEN</th>
<th align="center">CTZ</th>
<th align="center">CTX</th>
<th align="center">CIP</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">&#x23; Susceptible</td>
<td align="center">188</td>
<td align="center">276</td>
<td align="center">358</td>
<td align="center">366</td>
</tr>
<tr>
<td align="center">&#x23; Resistance</td>
<td align="center">621</td>
<td align="center">533</td>
<td align="center">451</td>
<td align="center">443</td>
</tr>
<tr>
<td align="center">Total</td>
<td align="center">809</td>
<td align="center">809</td>
<td align="center">809</td>
<td align="center">809</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>It is observed from <xref ref-type="fig" rid="F3">Figure 3</xref> that the dataset utilized in our work has a high imbalance ratio of resistance-susceptible strains for GEN and CTZ antibiotics, with a slight improvement in the case of inconsistency for CTX and CIP antibiotics. There is a significant imbalance ratio in susceptible (S): resistant (R) of 1:3 and 1:2 in the case of GEN and CTZ antibiotics, respectively. While considering all four antibiotics, the ratio of S: R is 1:2 (1,188:2048).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Susceptible and resistant strains for all four antibiotics.</p>
</caption>
<graphic xlink:href="fgene-16-1651917-g003.tif">
<alt-text content-type="machine-generated">Bar chart comparing the number of bacterial strains susceptible and resistant to four antibiotics: GEN, CTZ, CTX, and CIP. GEN shows 188 susceptible and 621 resistant strains. CTZ has 276 susceptible and 533 resistant strains. CTX displays 358 susceptible and 451 resistant strains. CIP has 366 susceptible and 443 resistant strains. Blue bars represent susceptible strains, and orange bars represent resistant strains.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3-3">
<title>3.3 Quality control</title>
<p>Data quality is the key to various AI model performances (<xref ref-type="bibr" rid="B35">Nayak et al., 2022</xref>; <xref ref-type="bibr" rid="B36">Nayak et al., 2023</xref>; <xref ref-type="bibr" rid="B50">Swain et al., 2023</xref>). Ren, Y. et al. developed the dataset (<xref ref-type="bibr" rid="B15">GitHub, 2025</xref>) to preprocess the raw WGS data; it uses BWA-MEM, and clean reads were mapped to the <italic>E. coli</italic> reference genome (<italic>E. coli</italic> K-12 strain, MG1655) after low-quality reads were filtered using fastp (v0.23.2) (<xref ref-type="bibr" rid="B7">Chen et al., 2018</xref>). By extracting reference and variant alleles and combining isolates according to reference allele positions, single-nucleotide polymorphisms (SNPs) were found using bcftools (v1.14) (<xref ref-type="bibr" rid="B9">Danecek et al., 2011</xref>; <xref ref-type="bibr" rid="B26">Li and Durbin, 2009</xref>). Preserving alleles that were found to be variations in more than half of the samples and creating an SNP matrix. One-hot encoding transformed the matrix into a binary format for further ML analysis.</p>
</sec>
<sec id="s3-4">
<title>3.4 Data preparation</title>
<p>This phase is the most crucial and contributes the most toward the model&#x2019;s performance (<xref ref-type="bibr" rid="B37">Nayak et al., 2024</xref>; <xref ref-type="bibr" rid="B33">Mohanty et al., 2023</xref>). We utilized the dataset developed by <xref ref-type="bibr" rid="B15">GitHub (2025)</xref> for our study. Hence, we restructured the dataset to meet our study objective. The one-hot encoding in the original data ranges from 1-4, while in our study, we modified this to 0.25-1. In addition, we aim to study the effect of this one-hot encoding on computational costs.</p>
</sec>
<sec id="s3-5">
<title>3.5 Proposed model aiGeneR 3.0</title>
<p>To identify <italic>E. coli</italic> strains that have gained resistance, we created the state-of-the-art aiGeneR 3.0 model; this model is based on DL and ML. The approach we have created is multi-staged and uses modern techniques to boost accuracy and robustness. Beginning with processed Next-Generation Sequencing (NGS) WGS data (<xref ref-type="bibr" rid="B15">GitHub, 2025</xref>), which offers a solid foundation for comprehensive inquiry, we use it as our primary source. To prepare the dataset, we use the quality control (QC) pipeline that we developed before (<xref ref-type="bibr" rid="B35">Nayak et al., 2022</xref>). In the final phase, highly-trained deep neural networks (LSTM) and Linear regression (LR) are used to reliably identify susceptible and resistant bacteria and to forecast the likelihood of multi-drug resistance in any given strain that shows resistance to any of the four antibiotics under investigation. The design and execution of aiGeneR 3.0 are depicted in <xref ref-type="fig" rid="F4">Figure 4</xref>. To describe the association between gene regulation variables and resistance extent, we used Linear Regression with least-squares optimization to reduce prediction errors and identify resistance-associated markers. Linear Regression provided a baseline prediction framework for deep learning model comparison, making the aiGeneR 3.0 pipeline robust in multidrug resistance categorization. Ultimately, a comprehensive evaluation of its efficacy using a predetermined set of assessment measures ensures the model&#x2019;s reliability. Biological confirmation also adds credence to its real-world utility. The aiGeneR 3.0 model is an all-inclusive and potent tool that could change the game when finding <italic>E. coli</italic> antibiotic resistance.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Overall architecture of our proposed aiGeneR 3.0 model.</p>
</caption>
<graphic xlink:href="fgene-16-1651917-g004.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a process for predicting multi-drug resistance using E. coli gene data. It includes steps: quality control, data partition, linear regression, LSTM for learning models, and prediction of normal or infected samples. The process ends with performance measurement.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3-6">
<title>3.6 Algorithm: the proposed model aiGeneR 3.0</title>
<p>
<list list-type="simple">
<list-item>
<p>Step 1 Read the data</p>
<list list-type="simple">
<list-item>
<p>&#x2003;&#x2003;Gather <italic>E. coli</italic> whole-genome sequencing data using NGS.</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Use fastp for sequencing reads and quality assurance (<xref ref-type="bibr" rid="B7">Chen et al., 2018</xref>).</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Align the filtered reads using BWA-mem.</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Adopt Bfctools for calling variants (<xref ref-type="bibr" rid="B9">Danecek et al., 2011</xref>).</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Sort and filter the aligned reads using Samtools (<xref ref-type="bibr" rid="B26">Li and Durbin, 2009</xref>)</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Let ED be the processed dataset containing SNPs. <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
</list>
</list-item>
<list-item>
<p>Step 2 Preparing the data</p>
<list list-type="simple">
<list-item>
<p>&#x2003;&#x2003;Align up (A) the data and eliminate duplicates.</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>n</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>E</mml:mi>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;To remove duplicates, <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>d</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
</list>
</list-item>
<list-item>
<p>Step 3 Data Engineering</p>
<list list-type="simple">
<list-item>
<p>&#x2003;&#x2003;Use a one-hot encoding method (OH<sub>E</sub>) as <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mi>E</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>O</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>H</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Decide on normalized values between 0.25 and 1 as follows: Equation, <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.25</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>0.75</mml:mn>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>X</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mi>E</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="italic">Min</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">Max</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="italic">Min</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula> , where <italic>OH</italic> is the normalized one-hot encoding data.</p>
</list-item>
</list>
</list-item>
<list-item>
<p>Step 4 Split the train and test</p>
<list list-type="simple">
<list-item>
<p>&#x2003;&#x2003;Split the data into sets for training and testing.</p>
</list-item>
</list>
</list-item>
<list-item>
<p>Step 5 aiGeneR 3.0 application (train data)</p>
<list list-type="simple">
<list-item>
<p>&#x2003;&#x2003;Customize the model by <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>G</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>R</mml:mi>
<mml:mn>3.0</mml:mn>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Utilizing the training set <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>G</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>R</mml:mi>
<mml:mn>3.0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>G</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>R</mml:mi>
<mml:mn>3.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, Training the aiGeneR 3.0 model.</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Acquire the predictive model. <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>G</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>R</mml:mi>
<mml:mn>3.0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>G</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>R</mml:mi>
<mml:mn>3.0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Find out what percentages of the various types are resistant to antibiotics with the Equation <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, Where <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the indicator function, which has values <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
</list>
</list-item>
<list-item>
<p>Step 6 Multi-drug resistant prediction and identification of resistant strains (test data)</p>
<list list-type="simple">
<list-item>
<p>&#x2003;&#x2003;Determine which strains are resistant in the test data by following the Equation.</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>G</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>R</mml:mi>
<mml:mn>3.0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Estimate the resistance of strains to multiple drugs as per the following Equation, <inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Where <inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> determines multi-drug resistant by comparing the predicted resistance probability <inline-formula id="inf15">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> against a threshold <inline-formula id="inf16">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The rule for identification follows the rule below, <inline-formula id="inf17">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>g</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:msub>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2265;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>g</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Achieve outcomes with <inline-formula id="inf18">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>y</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> as susceptible-resistant strains.</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<inline-formula id="inf19">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
</mml:mstyle>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>, determine the percentage of bacterial strains that are resilient to antibiotics.</p>
</list-item>
</list>
</list-item>
<list-item>
<p>Step 7 Assessment of the Model</p>
<list list-type="simple">
<list-item>
<p>&#x2003;&#x2003;Analyze aiGeneR 3.0&#x2019;s effectiveness.</p>
</list-item>
</list>
</list-item>
</list>
</p>
</sec>
<sec id="s3-7">
<title>3.7 Architecture and parameter</title>
<p>The primary step in using aiGeneR 3.0 is careful data preprocessing. The genomic sequences of different strains of bacteria are encoded into numerical formats that are suitable for input into neural networks. Enabling further calculation usually involves converting categorical genetic data into a numerical format using methods like one-hot encoding (<xref ref-type="bibr" rid="B8">Dahouda and Joe, 2021</xref>). The proposed architecture consists of a total of eight layers, and four types of layers are utilized, out of which three are dense, two are dropout, and one each for regularization, flattening, and softmax make up the model&#x2019;s architecture. The first, second, and third dense layers contain 64, 64, and 32 neurons, respectively. Comparably, our work employs many values for the dropout and regularization layers. The different regularization values experimented with in our work are 0.01, 0.001, and 0.0001, and the dropout rates are 0.25, 0.5, 0.7, and 0.9. Our proposed local architecture of the LSTM model with other added layers utilizing the random search is shown in <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Local architecture of the LSTM deployed in our study.</p>
</caption>
<graphic xlink:href="fgene-16-1651917-g005.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a neural network model. It starts with input genome data flowing through a series of layers: Dense, Regularization, Dropout, LSTM, Dropout, Dense, Flatten, Dense, and Softmax. It results in the classification of strains as resistant or susceptible.</alt-text>
</graphic>
</fig>
<p>The main objective of this study is to examine and compare the effectiveness of our proposed aiGeneR 3.0 model with different parameters to achieve the best classification accuracy for identifying the resistant strains utilizing the WGS <italic>E. coli</italic> NGS data. Thus, we adopt several changes during the implementation phase to the parameters of aiGeneR 3.0. We discuss some key phases of our experiments in the following sub-sections and the updation in several parameters of our implemented model, among which a few are shown in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>All the phases of model implementations have different parameter configurations.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Phases</th>
<th align="center">Train-test</th>
<th align="center">Regularization</th>
<th align="center">Dropout</th>
<th align="center">K-fold</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">1</td>
<td align="center">70:30</td>
<td align="center">0.01</td>
<td align="center">0.25</td>
<td align="center">3</td>
</tr>
<tr>
<td align="center">2</td>
<td align="center">
<bold>80:20</bold>
</td>
<td align="center">
<bold>0.001</bold>
</td>
<td align="center">
<bold>0.5</bold>
</td>
<td align="center">
<bold>5</bold>
</td>
</tr>
<tr>
<td align="center">3</td>
<td align="center">90:10</td>
<td align="center">0.0001</td>
<td align="center">0.7</td>
<td align="center">10</td>
</tr>
<tr>
<td align="center">4</td>
<td align="center">70:30</td>
<td align="center">0.001</td>
<td align="center">0.5</td>
<td align="center">5</td>
</tr>
<tr>
<td align="center">5</td>
<td align="center">80:20</td>
<td align="center">0.01</td>
<td align="center">0.25</td>
<td align="center">5</td>
</tr>
<tr>
<td align="center">6</td>
<td align="center">80:20</td>
<td align="center">0.0001</td>
<td align="center">0.5</td>
<td align="center">5</td>
</tr>
<tr>
<td align="center">7</td>
<td align="center">80:20</td>
<td align="center">0.001</td>
<td align="center">0.7</td>
<td align="center">10</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>&#x2a;Bold showing the best result.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec id="s4">
<title>4 Experimental setup and implementation</title>
<p>The proposed aiGeneR 3.0 is a complete package of DL and ML models for identifying resistant strains and predicting multidrug resistance in strains. The architecture of aiGeneR 3.0 is simple and less complex than that of previously proposed DL models for resistant strain identification. In our experimental setup, we implemented several versions of aiGeneR 3.0 with different model parameters and finally proposed the architecture that consumes less computational time and produces the most significant results. In this section, we discuss a few of the several implementation versions of aiGeneR 3.0 with different hyperparameters.</p>
<sec id="s4-1">
<title>4.1 Experiment I: high learning rate with smaller training data</title>
<p>During the initial development phase, we are refining the aiGeneR 3.0 model, which employs an LSTM architecture, for our analysis. A softmax layer was incorporated to facilitate classification tasks. A learning rate of 0.01 and a dropout rate of 0.25 were utilized to optimize the training process and mitigate overfitting. To conduct a comprehensive assessment of the model&#x2019;s performance, we partitioned the dataset into two distinct subsets: the training set and the testing set. The ratio of the train-test split was 70:30. Furthermore, a K-fold cross-validation technique was employed with K &#x3d; 3 to assess the model&#x2019;s ability to generalize. Through iterative training and evaluation on various subsets of the dataset, we successfully enhanced the accuracy and dependability of the model in identifying resistant strains in the data.</p>
</sec>
<sec id="s4-2">
<title>4.2 Experiment II: moderate learning rate with increasing training data</title>
<p>In this implementation phase, we continued our research by iteratively improving the aiGeneR 3.0 model by changing several critical hyperparameters. We adjusted the learning rate to 0.001 to address overfitting and raised the dropout rate to 0.5. These changes should promote more regularization. To keep the assessment process consistent, we partitioned the dataset at an 80:20 train-test split ratio. We also used a K-fold cross-validation method with K &#x3d; 5 to strengthen our model evaluation and thoroughly examine its generalizability capabilities, which improved the validation procedure. The model&#x2019;s training dynamics were fine-tuned using these improvements so that it could better use features from the NGS data to identify resistant bacteria.</p>
</sec>
<sec id="s4-3">
<title>4.3 Experiment III: low learning rate with maximum training data</title>
<p>During this experiment phase, we kept tweaking the hyperparameters of the aiGeneR 3.0 model to make it even better. Now, we are trying to find the sweet spot by gradually adjusting the model weights during training with a learning rate of 0.0001. We raised the dropout rate to 0.7 to improve model regularization and alleviate overfitting worries; this should lead to more diverse and resilient learned representations. To keep things uniform throughout the assessment, we kept the train-test split ratio at 90:10. To further evaluate the model&#x2019;s efficacy across different data subsets, we also used a more stringent K-fold cross-validation method with K &#x3d; 10. This broadened the scope of our validation strategy.</p>
<p>We consistently obtained the best performance metrics with an 8:2 train-test ratio throughout all stages of our model, aiGeneR 3.0, as shown in <xref ref-type="table" rid="T2">Table 2</xref>. This ratio consistently produced the best outcomes, even with changing parameter combinations during the different phases. After training on 80% of the data and testing on the remaining 20%, our model showed exceptional accuracy, precision, sensitivity, and specificity. This strategy ensured that generalization and model complexity were balanced, enabling reliable performance on several splits of the datasets. Furthermore, at each step, regularization strategies, dropout rates, and K-fold cross-validation were methodically investigated to improve the performance of our models. Notably, the 8:2 train-test ratio was a stable base for attaining optimal outcomes across all of our implementations, even though changes to the parameters affected the model&#x2019;s behavior.</p>
<p>In addition to the above experiments, we also implemented our aiGeneR 3.0 in several other phases, with the model parameters fine-tuned. We also take the different train-test splits to the above experiments and add various other possible dropout rates. However, we observe different model matrices with each of these implementation phases of our aiGeneR 3.0 and consider the best performance, which is described in sections 5 (results) and 6 (discussions).</p>
</sec>
</sec>
<sec id="s5">
<title>5 Performance evaluation</title>
<p>This section presents a thorough performance evaluation of aiGeneR 3.0 and discusses the various evaluation processes adopted in our study. Our study employs a distinct blend of methodologies: power analysis, empirical analysis, and evaluation of model generalization. Empirical analysis assesses the practical value of the model in real-world situations, while power analysis evaluates its ability to detect meaningful effects. The analysis of model generalization focuses on its ability to acquire knowledge from training data and adjust to various unseen datasets. This comprehensive evaluation technique will unveil the intricate complexities of aiGeneR 3.0, providing insights into its effectiveness and robustness.</p>
<sec id="s5-1">
<title>5.1 Power analysis</title>
<p>Power analysis is a statistical method employed to ascertain the minimal sample size necessary for a study to attain a specified degree of statistical power (<xref ref-type="bibr" rid="B37">Nayak et al., 2024</xref>; <xref ref-type="bibr" rid="B22">Jamthikar et al., 2020</xref>). Power analysis is essential in the realm of deep learning models as it allows for the estimation of the required sample size to effectively detect significant impacts or disparities in the model&#x2019;s performance while maintaining a desired level of confidence.</p>
<p>We conducted a power analysis to determine the minimum sample size needed to calculate a population proportion with precision and accuracy. The experiments were conducted utilizing the methodology described in (<xref ref-type="bibr" rid="B21">Jamthikar et al., 2019</xref>; <xref ref-type="bibr" rid="B48">Skandha et al., 2020</xref>). The formula for sample size calculation, represented by the symbol Sn, is shown in <xref ref-type="disp-formula" rid="e1">Equation 1</xref>:<disp-formula id="e1">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">z</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2010;</mml:mo>
<mml:mover accent="true">
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:msup>
<mml:mtext>MoE</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>In this context, MoE represents the margin of error, <inline-formula id="inf20">
<mml:math id="m21">
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> denotes the estimated proportion of the feature in the population, and z&#x2a; refers to the Z-score linked to the relevant confidence level. The MoE2 was calculated by using half of the width of the confidence interval. We selected a ratio of 0.5 and a confidence level of 95% for our experiment. The power analysis is conducted using MedCalc (<xref ref-type="bibr" rid="B32">Medcalc, 2025</xref>) and demonstrates that the study has a sample size (809) that exceeds the required amount to achieve the intended degree of statistical power and correct classification. The minimum sample size for the dataset utilized is 271 (68 are susceptible and 203 are resistant), which is smaller than the available data.</p>
</sec>
<sec id="s5-2">
<title>5.2 Empirical analysis</title>
<p>The confusion matrix, a real-to-anticipated-class matrix with multiple evaluation standards, is the primary target of performance parameters. TP and FP stand for true positives and false positives, respectively, in the confusion matrix. Similarly, TN and FN represent true negatives and false negatives, respectively. There are four types of predictions: TP, which accurately predicts that samples with resistance will be resistant; TN, which accurately predicts that samples without resistance will be susceptible; FP, which inaccurately predicts that susceptible samples will be resistant; and FN, which inaccurately predicts that resistant samples will be susceptible.</p>
<p>Measures such as accuracy (Acc), precision (Pre), specificity (Spe), sensitivity (Sen), F1-score (F1), Matthews Correlation Coefficient (MCC), and area under the curve (AUC) are some of the classification performance measures that were studied in this study. The total number of input samples divided by the number of valid predictions is the &#x201c;processor, ranging from 1 to 4 (dataset A), while the second dataset consists of one-hot encoding. We call the recall the percentage of positive observations that were projected to be positive compared to the total number of positive observations. F1 is the weighted mean of precision and recall. All the model metrics are calculated based on the following equations (<xref ref-type="disp-formula" rid="e2">Equations 2</xref>&#x2013;<xref ref-type="disp-formula" rid="e7">7</xref>).<disp-formula id="e2">
<mml:math id="m22">
<mml:mrow>
<mml:mtext>Accuracy&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>Acc</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
<disp-formula id="e3">
<mml:math id="m23">
<mml:mrow>
<mml:mtext>Precision&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>Pre</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
<disp-formula id="e4">
<mml:math id="m24">
<mml:mrow>
<mml:mtext>Sensitivity&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>Sen</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m25">
<mml:mrow>
<mml:mtext>Specificity&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>Spe</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
<disp-formula id="e6">
<mml:math id="m26">
<mml:mrow>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mtext>&#x2009;Score</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#x2a;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mfrac>
<mml:mrow>
<mml:mtext>Precision</mml:mtext>
<mml:mo>&#x2a;</mml:mo>
<mml:mtext>Recall</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>Precision</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>Recall</mml:mtext>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="e7">
<mml:math id="m27">
<mml:mrow>
<mml:mtext>MCC</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
</sec>
</sec>
<sec sec-type="results" id="s6">
<title>6 Results</title>
<p>The Anaconda environment and Jupyter Notebook are used to carry out the model architecture design and parameter configuration. The learning models are implemented in Python (version 3.7) (<xref ref-type="bibr" rid="B41">Python, 2025</xref>). Here, we present the findings from the exploratory data analysis, together with a discussion of the results obtained using the suggested methodology. Our study optimized K-Nearest Neighbors (KNN), Decision Tree (DT), Support Vector Machine (SVM), VGG-19, 1-Dimensional Convolutional Neural Network (1-D CNN), and ResNet-50 to ensure resilient performance. Grid search found the best k for KNN, balancing classification accuracy and computing economy. DT used a Gini impurity-based criterion with a maximum depth to avoid overfitting. SVM utilized an RBF kernel with optimized hyperparameters C and &#x3b3; by cross-validation. Transfer learning adjusted VGG-19 and ResNet-50 architectures were adapted to gene expression data, where both models were trained from scratch with customized input layers and trained using the Adam optimizer. Convolutional, pooling, and dense layers were added to the 1-D CNN architecture with sequential data pattern kernel sizes and activation functions. All models were hyperparameter-tuned and evaluated for optimal performance.</p>
<p>We used a grid search to optimize the learning rate, batch size, hidden layers, and neuron counts hyperparameters for aiGeneR 3.0; we then tested each combination using 5-fold cross-validation to make sure it generalized well and did not overfit. The input data was meticulously preprocessed to remove duplicates, remove samples with too many missing values, and impute missing entries using nearest-neighbor averaging. The magnitude-driven biases in deep learning models were mitigated by scaling numerical features to 0.25&#x2013;1. To classify resistance, we used a 0.5 threshold to turn projected probabilities into binary calls, and we fine-tuned for unbalanced medicines using ROC to maximize F1-score and minority-class detection. Stable training, repeatable performance, and accurate resistance strain prediction were all achieved by means of this integrated technique.</p>
<p>The proposed aiGeneR 3.0 architecture was constructed using two different machines. The main machine, also known as machine-1, is a workstation running Ubuntu 20.04 that has an Intel Core i7 CPU, 32&#xa0;GB of RAM, and 1&#xa0;TB of solid-state drive storage, among other characteristics. The second machine, called Machine-2, is equipped with an Intel Core i5 processor, ranging from 1 to 4 (dataset A), while the second dataset consists of one-hot encoding, utilizing two different datasets. The first dataset consists of the one-hot encoding, ranging from 1 to 4 (dataset A), while the second dataset consists of one-hot encoding, ranging from 0.25 to 1 (dataset B). The comparison of the two systems&#x2019; computing time performance using the implemented model is presented in <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Computational time taken by all the studied models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Models</th>
<th colspan="2" align="center">Dataset A (&#xb5;s)</th>
<th colspan="2" align="center">Dataset B (0.25&#x2013;1) (&#xb5;s)</th>
</tr>
<tr>
<th align="center">Machine-1</th>
<th align="center">Machine-2</th>
<th align="center">Machine-1</th>
<th align="center">Machine-2</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">SVM &#x2b; RBF</td>
<td align="center">250</td>
<td align="center">420</td>
<td align="center">90</td>
<td align="center">114</td>
</tr>
<tr>
<td align="center">DT</td>
<td align="center">241</td>
<td align="center">495</td>
<td align="center">120</td>
<td align="center">151</td>
</tr>
<tr>
<td align="center">KNN</td>
<td align="center">306</td>
<td align="center">570</td>
<td align="center">122</td>
<td align="center">142</td>
</tr>
<tr>
<td align="center">VGG-19</td>
<td align="center">320</td>
<td align="center">640</td>
<td align="center">120</td>
<td align="center">148</td>
</tr>
<tr>
<td align="center">1D-CNN</td>
<td align="center">250</td>
<td align="center">426</td>
<td align="center">95</td>
<td align="center">113</td>
</tr>
<tr>
<td align="center">ResNet-50</td>
<td align="center">296</td>
<td align="center">580</td>
<td align="center">123</td>
<td align="center">146</td>
</tr>
<tr>
<td align="center">CNN TL</td>
<td align="center">300</td>
<td align="center">620</td>
<td align="center">126</td>
<td align="center">146</td>
</tr>
<tr>
<td align="center">CNN Ensemble</td>
<td align="center">380</td>
<td align="center">592</td>
<td align="center">117</td>
<td align="center">141</td>
</tr>
<tr>
<td align="center">aiGeneR 3.0</td>
<td align="center">207</td>
<td align="center">370</td>
<td align="center">
<bold>87</bold>
</td>
<td align="center">
<bold>97</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values show the best performance result in our study.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>From the above table, we observed notable variations in the computation times of different deployed models when they were assessed during both the training and testing stages, including aiGeneR 3.0. Notably, our suggested aiGeneR 3.0 model leads other studied models in terms of efficiency for the two datasets, consuming just 207 &#xb5;s for machine-1 and 87&#xa0;&#xb5;s for machine-2. Due to its better hardware, machine-1 constantly shows faster computational times than machine-2; yet, aiGeneR 3.0 is the most effective model, with quick processing times that boost output and facilitate quick decision-making. On the other hand, other models like SVM &#x2b; RBF, DT, KNN, VGG-19, 1D-CNN, ResNet-50, CNN TL, and Ensemble approaches have significantly longer training and testing times on both the datasets studied. All things considered, aiGeneR 3.0&#x2019;s effectiveness highlights how quickly it can train and assess models, which shows its potential for quick learning capacity. We evaluate our aiGeneR 3.0 with a previously developed TL model. <xref ref-type="bibr" rid="B43">Ren et al. (2022b)</xref> and found that it consumes a remarkably less computational time of 31% and 45% in machine-1 for dataset A, and similarly takes 40% and 38% less in machine-2 for dataset B, as seen in <xref ref-type="table" rid="T3">Table 3</xref>. In addition to this, it can be seen from the table that the one-hot encoding approach adopted in our study (dataset B) shows a remarkably lower computational time compared to dataset A (<xref ref-type="bibr" rid="B43">Ren et al., 2022b</xref>) for producing the classification result for all the studied models.</p>
<p>This section quantifies and thoroughly examines the accuracy of the proposed framework, aiGeneR 3.0. Regarding its simple bending model architecture and predictive abilities, aiGeneR 3.0 performs admirably in various tasks, including prediction and classification. The pipeline of aiGeneR 3.0 is the adaptation of the LR and LSTM algorithms. Through an in-depth evaluation of its accuracy, we aim to gain insight into how well aiGeneR 3.0 works when it comes to resistance strain classification with limited computational capabilities and unbalanced data. This section discusses the outcome of our work in the following manner,<list list-type="simple">
<list-item>
<p>A. The ability of our aiGeneR 3.0 model to identify resistant strains by utilizing a single antibiotic.</p>
</list-item>
<list-item>
<p>B. The performance outcome of aiGeneR 3.0 on all four antibiotics taken together to identify the resistant strains.</p>
</list-item>
<list-item>
<p>C. Comparison of all the studied AI models</p>
</list-item>
<list-item>
<p>D. The aiGeneR 3.0 and multi-drug resistance prediction</p>
</list-item>
</list>
</p>
<p>A: We assess the performance of all our studied models on four different antibiotics. The antibiotics considered for our work are CTX, GEN, CTZ, and CIP. We observed better model metrics while we deployed our proposed model on the CIP dataset, as this dataset is quite balanced compared to other datasets. The detailed model metrics for the CIP dataset of implementations are shown in <xref ref-type="table" rid="T4">Table 4</xref> below.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Performance metrics of all the studied models on the CIP dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">Acc (%)</th>
<th align="center">Pre (%)</th>
<th align="center">Sen (%)</th>
<th align="center">Spe (%)</th>
<th align="center">F1 (%)</th>
<th align="center">MCC (%)</th>
<th align="center">AUC (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">SVM &#x2b; RBF</td>
<td align="center">86</td>
<td align="center">87</td>
<td align="center">86</td>
<td align="center">94</td>
<td align="center">86</td>
<td align="center">81</td>
<td align="center">94</td>
</tr>
<tr>
<td align="center">DT</td>
<td align="center">83</td>
<td align="center">84</td>
<td align="center">83</td>
<td align="center">78</td>
<td align="center">83</td>
<td align="center">61</td>
<td align="center">94</td>
</tr>
<tr>
<td align="center">KNN</td>
<td align="center">83</td>
<td align="center">83</td>
<td align="center">83</td>
<td align="center">82</td>
<td align="center">83</td>
<td align="center">65</td>
<td align="center">94</td>
</tr>
<tr>
<td align="center">ResNet-50</td>
<td align="center">90</td>
<td align="center">90</td>
<td align="center">90</td>
<td align="center">90</td>
<td align="center">90</td>
<td align="center">80</td>
<td align="center">96</td>
</tr>
<tr>
<td align="center">VGG-19</td>
<td align="center">82</td>
<td align="center">84</td>
<td align="center">83</td>
<td align="center">91</td>
<td align="center">83</td>
<td align="center">75</td>
<td align="center">90</td>
</tr>
<tr>
<td align="center">1D-CNN</td>
<td align="center">86</td>
<td align="center">88</td>
<td align="center">83</td>
<td align="center">88</td>
<td align="center">86</td>
<td align="center">76</td>
<td align="center">91</td>
</tr>
<tr>
<td align="center">CNN TL</td>
<td align="center">91</td>
<td align="center">91</td>
<td align="center">89</td>
<td align="center">91</td>
<td align="center">91</td>
<td align="center">82</td>
<td align="center">97</td>
</tr>
<tr>
<td align="center">CNN Ensemble</td>
<td align="center">92</td>
<td align="center">92</td>
<td align="center">90</td>
<td align="center">92</td>
<td align="center">91</td>
<td align="center">84</td>
<td align="center">97</td>
</tr>
<tr>
<td align="center">aiGeneR 3.0</td>
<td align="center">
<bold>93</bold>
</td>
<td align="center">
<bold>96</bold>
</td>
<td align="center">
<bold>90</bold>
</td>
<td align="center">
<bold>95</bold>
</td>
<td align="center">
<bold>92</bold>
</td>
<td align="center">
<bold>90</bold>
</td>
<td align="center">
<bold>99</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values show the best performance result in our study.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The identification of resistant strains by our proposed aiGeneR 3.0, utilizing the CIP dataset, has an accuracy of 93%, which is higher than that of all the studied models. In addition to this, our proposed approach achieves higher sensitivity and specificity of 90% and 95%, respectively, as shown in <xref ref-type="fig" rid="F6">Figure 6</xref>.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Performance metrics of all the deployed models on the CIP dataset.</p>
</caption>
<graphic xlink:href="fgene-16-1651917-g006.tif">
<alt-text content-type="machine-generated">Bar chart comparing performance metrics of various models: SVM+RBF, DT, KNN, ResNet-50, VGG-19, 1D-CNN, Transfer Learning, CNN Ensemble, aiGeneR 2.0. Metrics include accuracy, precision, sensitivity, specificity, and F1 score, with values ranging from 78% to 96%. Models generally perform above 80%.</alt-text>
</graphic>
</fig>
<p>We also evaluate our proposed model on the CTX, CTZ, and GEN antibiotics datasets. aiGeneR 3.0 achieves the highest classification accuracy of 82%, 88%, and 80% for CTX, CTZ, and GEN data, respectively. It is observed that the GEN dataset is highly imbalanced and contains a susceptible-to-resistant ratio of 4:1, and notably, our aiGeneR 3.0 reaches the highest classification accuracy of 80% among all the studied models. In addition to this, aiGeneR 3.0 sustains good specificity and sensitivity values for all three antibiotics, which shows its potential to classify the resistant strains with a very minimal false negative rate. The model metrics for CTX, GEN, and CTZ are summarized in <xref ref-type="table" rid="T5">Table 5</xref> below. However, among all the studied models, CNN ensemble, CNN TL, and 1D CNN perform better compared to other models in terms of classification accuracy.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Performance metrics of all the studied models on the CTX, CTZ, and GEN datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Model</th>
<th colspan="5" align="center">CTX</th>
<th colspan="5" align="center">CTZ</th>
<th colspan="5" align="center">GEN</th>
</tr>
<tr>
<th align="center">A (%)</th>
<th align="center">P (%)</th>
<th align="center">Se (%)</th>
<th align="center">Sp (%)</th>
<th align="center">F (%)</th>
<th align="center">A (%)</th>
<th align="center">P (%)</th>
<th align="center">Se (%)</th>
<th align="center">Sp (%)</th>
<th align="center">F (%)</th>
<th align="center">A (%)</th>
<th align="center">P (%)</th>
<th align="center">Se (%)</th>
<th align="center">Sp (%)</th>
<th align="center">F (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">SVM &#x2b; RBF</td>
<td align="center">77</td>
<td align="center">77</td>
<td align="center">77</td>
<td align="center">78</td>
<td align="center">77</td>
<td align="center">82</td>
<td align="center">82</td>
<td align="center">81</td>
<td align="center">84</td>
<td align="center">82</td>
<td align="center">74</td>
<td align="center">62</td>
<td align="center">75</td>
<td align="center">20</td>
<td align="center">85</td>
</tr>
<tr>
<td align="center">DT</td>
<td align="center">67</td>
<td align="center">67</td>
<td align="center">68</td>
<td align="center">67</td>
<td align="center">67</td>
<td align="center">74</td>
<td align="center">75</td>
<td align="center">78</td>
<td align="center">68</td>
<td align="center">75</td>
<td align="center">66</td>
<td align="center">73</td>
<td align="center">80</td>
<td align="center">34</td>
<td align="center">76</td>
</tr>
<tr>
<td align="center">KNN</td>
<td align="center">74</td>
<td align="center">73</td>
<td align="center">75</td>
<td align="center">73</td>
<td align="center">74</td>
<td align="center">79</td>
<td align="center">79</td>
<td align="center">82</td>
<td align="center">73</td>
<td align="center">79</td>
<td align="center">72</td>
<td align="center">79</td>
<td align="center">83</td>
<td align="center">44</td>
<td align="center">81</td>
</tr>
<tr>
<td align="center">ResNet-50</td>
<td align="center">75</td>
<td align="center">75</td>
<td align="center">75</td>
<td align="center">75</td>
<td align="center">75</td>
<td align="center">78</td>
<td align="center">79</td>
<td align="center">78</td>
<td align="center">79</td>
<td align="center">78</td>
<td align="center">63</td>
<td align="center">60</td>
<td align="center">88</td>
<td align="center">37</td>
<td align="center">71</td>
</tr>
<tr>
<td align="center">VGG-19</td>
<td align="center">72</td>
<td align="center">71</td>
<td align="center">72</td>
<td align="center">71</td>
<td align="center">72</td>
<td align="center">67</td>
<td align="center">70</td>
<td align="center">65</td>
<td align="center">76</td>
<td align="center">62</td>
<td align="center">75</td>
<td align="center">86</td>
<td align="center">77</td>
<td align="center">46</td>
<td align="center">85</td>
</tr>
<tr>
<td align="center">1D-CNN</td>
<td align="center">73</td>
<td align="center">73</td>
<td align="center">71</td>
<td align="center">75</td>
<td align="center">74</td>
<td align="center">82</td>
<td align="center">82</td>
<td align="center">87</td>
<td align="center">75</td>
<td align="center">84</td>
<td align="center">79</td>
<td align="center">80</td>
<td align="center">78</td>
<td align="center">50</td>
<td align="center">81</td>
</tr>
<tr>
<td align="center">CNN TL</td>
<td align="center">79</td>
<td align="center">81</td>
<td align="center">80</td>
<td align="center">78</td>
<td align="center">80</td>
<td align="center">83</td>
<td align="center">84</td>
<td align="center">88</td>
<td align="center">79</td>
<td align="center">79</td>
<td align="center">79</td>
<td align="center">79</td>
<td align="center">77</td>
<td align="center">47</td>
<td align="center">85</td>
</tr>
<tr>
<td align="center">CNN Ensemble</td>
<td align="center">80</td>
<td align="center">81</td>
<td align="center">79</td>
<td align="center">80</td>
<td align="center">80</td>
<td align="center">83</td>
<td align="center">83</td>
<td align="center">89</td>
<td align="center">79</td>
<td align="center">81</td>
<td align="center">80</td>
<td align="center">85</td>
<td align="center">80</td>
<td align="center">60</td>
<td align="center">86</td>
</tr>
<tr>
<td align="center">aiGeneR 3.0</td>
<td align="center">
<bold>82</bold>
</td>
<td align="center">
<bold>83</bold>
</td>
<td align="center">
<bold>93</bold>
</td>
<td align="center">
<bold>82</bold>
</td>
<td align="center">
<bold>83</bold>
</td>
<td align="center">
<bold>88</bold>
</td>
<td align="center">
<bold>90</bold>
</td>
<td align="center">
<bold>90</bold>
</td>
<td align="center">
<bold>84</bold>
</td>
<td align="center">
<bold>90</bold>
</td>
<td align="center">
<bold>80</bold>
</td>
<td align="center">
<bold>91</bold>
</td>
<td align="center">
<bold>84</bold>
</td>
<td align="center">
<bold>62</bold>
</td>
<td align="center">
<bold>87</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values show the best performance result in our study.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The performance of aiGeneR 3.0, while we are utilizing the CTX, CTZ, and GEN antibiotics, excels in terms of classification accuracy, sensitivity, and specificity. In addition to this, we obtained a notable sensitivity and specificity while deploying our aiGeneR 3.0 on these three datasets. This result showcases the potential of aiGeneR 3.0 to identify the resistant strains in <italic>E. coli</italic> and can further be tested with other bacterial agents causing antibiotic resistance. The performance of the top-4 models on CTX, CTZ, and GEN datasets based on the accuracy (A), precision (P), sensitivity (Se), specificity (Sp), and F score (F) is visualized in <xref ref-type="fig" rid="F7">Figure 7</xref>.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Performance metrics of the top-4 studied models on <bold>(a)</bold> CTX, <bold>(b)</bold> CTZ, and <bold>(c)</bold> GEN datasets.</p>
</caption>
<graphic xlink:href="fgene-16-1651917-g007.tif">
<alt-text content-type="machine-generated">Three bar charts labeled (a) GEN, (b) CTX, and (c) CTZ compare performance metrics of models: 1D-CNN, CNN TL, CNN Ensemble, and aiGener 3.0. Six metrics are shown: Accuracy (blue), Precision (green), Sensitivity (red), Specificity (cyan), F1 Score (magenta). aiGener 3.0 often achieves the highest scores across metrics, notably 93% in Precision on chart (b).</alt-text>
</graphic>
</fig>
<p>B: We evaluate the efficacy of our proposed aiGeneR 3.0 to predict the resistance strains by taking all four antibiotics. This pipeline is designed by taking all the strains of the dataset along with all four antibiotics. We refined the dataset by keeping the original susceptible strains and updating the strain resistance to more than two antibiotics as multidrug resistance.</p>
<p>Based on our evaluations with a dataset that included all antibiotics, a learning rate of 0.01, a dropout rate of 0.5, and k-fold cross-validation with k &#x3d; 5, we found that the aiGeneR 3.0 model performed significantly better than other models. The model metrics for all the studied models are shown in <xref ref-type="table" rid="T6">Table 6</xref> below and can be visualized in <xref ref-type="fig" rid="F8">Figure 8</xref>. With an impressive 92% accuracy, 92% precision, 91% sensitivity, and 95% specificity, the model accurately identified resistant strains while reducing false positives and negatives. Additionally, aiGeneR 3.0 demonstrated excellent discriminative ability in differentiating between susceptible and resistant strains with an impressive AUC value of 0.99. These results highlight the efficacy of our model architecture and training methodology, confirming that it is suitable for precise antibiotic resistance prediction and indicating that it may prove to be a helpful tool for improving therapeutic strategies in clinical settings. In addition to this, the classification accuracy of our proposed aiGeneR 3.0 model is 3% higher than that of the previously studied CNN TL model (<xref ref-type="bibr" rid="B43">Ren et al., 2022b</xref>). On the highly unbalanced all-antibiotic dataset, the highest Matthews Correlation Coefficient (MCC) obtained by aiGeneR 3.0 is 87%, while the lowest MCC acquired by DT is 28%. Because the CIP dataset is more balanced than the all-antibiotic datasets, the MCC on this dataset is excellent across all models that have been assessed.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Model metrics of all the studied models utilizing the dataset with cases having resistance/susceptibility to all four antibiotics.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">Acc (%)</th>
<th align="center">Pre (%)</th>
<th align="center">Sen (%)</th>
<th align="center">Spe (%)</th>
<th align="center">F1 (%)</th>
<th align="center">MCC (%)</th>
<th align="center">AUC</th>
<th align="center">Brier score</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">SVM &#x2b; RBF</td>
<td align="center">86</td>
<td align="center">86</td>
<td align="center">86</td>
<td align="center">73</td>
<td align="center">86</td>
<td align="center">56</td>
<td align="center">0.92</td>
<td align="center">0.040</td>
</tr>
<tr>
<td align="center">DT</td>
<td align="center">75</td>
<td align="center">75</td>
<td align="center">75</td>
<td align="center">59</td>
<td align="center">75</td>
<td align="center">28</td>
<td align="center">0.72</td>
<td align="center">0.140</td>
</tr>
<tr>
<td align="center">KNN</td>
<td align="center">86</td>
<td align="center">86</td>
<td align="center">86</td>
<td align="center">80</td>
<td align="center">86</td>
<td align="center">65</td>
<td align="center">0.89</td>
<td align="center">0.055</td>
</tr>
<tr>
<td align="center">ResNet-50</td>
<td align="center">87</td>
<td align="center">87</td>
<td align="center">86</td>
<td align="center">73</td>
<td align="center">86</td>
<td align="center">57</td>
<td align="center">0.90</td>
<td align="center">0.049</td>
</tr>
<tr>
<td align="center">VGG-19</td>
<td align="center">82</td>
<td align="center">82</td>
<td align="center">81</td>
<td align="center">76</td>
<td align="center">82</td>
<td align="center">57</td>
<td align="center">0.92</td>
<td align="center">0.047</td>
</tr>
<tr>
<td align="center">aiGeneR 1.0</td>
<td align="center">89</td>
<td align="center">88</td>
<td align="center">86</td>
<td align="center">82</td>
<td align="center">89</td>
<td align="center">70</td>
<td align="center">0.93</td>
<td align="center">0.035</td>
</tr>
<tr>
<td align="center">1D-CNN</td>
<td align="center">88</td>
<td align="center">88</td>
<td align="center">86</td>
<td align="center">85</td>
<td align="center">88</td>
<td align="center">73</td>
<td align="center">0.98</td>
<td align="center">0.010</td>
</tr>
<tr>
<td align="center">CNN Ensemble</td>
<td align="center">89</td>
<td align="center">89</td>
<td align="center">86</td>
<td align="center">87</td>
<td align="center">89</td>
<td align="center">76</td>
<td align="center">0.97</td>
<td align="center">0.015</td>
</tr>
<tr>
<td align="center">CNN TL</td>
<td align="center">91</td>
<td align="center">91</td>
<td align="center">91</td>
<td align="center">93</td>
<td align="center">90</td>
<td align="center">84</td>
<td align="center">0.97</td>
<td align="center">0.015</td>
</tr>
<tr>
<td align="center">aiGeneR 3.0</td>
<td align="center">
<bold>92</bold>
</td>
<td align="center">
<bold>92</bold>
</td>
<td align="center">
<bold>91</bold>
</td>
<td align="center">
<bold>95</bold>
</td>
<td align="center">
<bold>91</bold>
</td>
<td align="center">
<bold>87</bold>
</td>
<td align="center">
<bold>0.99</bold>
</td>
<td align="center">
<bold>0.005</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values show the best performance result in our study.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Performance of the top 5 models on all the antibiotics data.</p>
</caption>
<graphic xlink:href="fgene-16-1651917-g008.tif">
<alt-text content-type="machine-generated">Bar chart comparing five models: 1D-CNN, CNN Ensemble, aiGeneR 1.0, TL, and aiGeneR 3.0 across six metrics: Acc, Pre, Sen, Spe, F1, and AUC. aiGeneR 3.0 consistently scores highest, especially in AUC with 99%.</alt-text>
</graphic>
</fig>
<p>C: Comparison of studied models. The studied NGS <italic>E. coli</italic> WGS includes 810 strains and 14,972 SNPs. Our study made use of all 14,972 SNPs with data standardization. With a ratio of 8:2, 648 samples were used for training, and 162 samples were used for testing. The complexity and processing demand of each strategy were evaluated as we explored different models for resistant strain identification using NGS <italic>E. coli</italic> WGS. DT has the potential to overfit as the depth increases, while SVM with RBF kernels is computationally demanding and produces higher classification accuracy compared to DT and KNN, as shown in <xref ref-type="table" rid="T4">Table 4</xref>. When it comes to prediction, KNN requires more memory and has more computational complexity (<xref ref-type="bibr" rid="B25">Kuang and Zhao, 2009</xref>)Thus, we observed a higher computational time for KNN in <xref ref-type="table" rid="T3">Table 3</xref>. The CNNs like ResNet-50 and VGG-19 deployed in our study have complex architecture and consume more memory and computational cost, as shown in <xref ref-type="table" rid="T3">Table 3</xref>. The level of complexity in aiGeneR 1.0 is moderate (<xref ref-type="bibr" rid="B37">Nayak et al., 2024</xref>). Despite its simplicity, the 1D-CNN still requires a lot of resources. CNN Ensemble adds complexity by combining different models (<xref ref-type="bibr" rid="B59">Zhang et al., 2020</xref>) and consumes the highest computational time compared to all the studied AI models, as shown in <xref ref-type="table" rid="T3">Table 3</xref>. Despite keeping complexity high, CNN TL shortens training time. The proposed aiGeneR 3.0 strikes the perfect balance between processing time and significant classification accuracy, especially due to its streamlined LSTM architecture.</p>
<p>The overall performance of all the models is assessed in terms of classification accuracy, precision, sensitivity, and specificity, as discussed in the performance evaluation section. In addition to this, we consider computational time to be one of the major performance parameters used to evaluate all the studied models. We observe that our proposed model, aiGeneR 3.0, achieves higher performance metrics compared to all other studied models. There is a slight increasing trend in the classification accuracy of aiGeneR 3.0 compared to the previously deployed CNN TL model. <xref ref-type="bibr" rid="B43">Ren et al. (2022b)</xref> with a remarkable AUC of 0.99. The most powerful aspect of our aiGeneR 3.0 model is the computational cost; it consumes very little computational power compared to all other studied models. The overall architecture and one-hot encoding techniques adopted by aiGeneR 3.0 make it robust and computationally cost-effective. The multi-drug resistant prediction is one of the significant contributions of aiGeneR 3.0 compared to previous works (<xref ref-type="bibr" rid="B42">Ren et al., 2022a</xref>), and it will be discussed in the next section. Additionally, the computational time taken by our aiGeneR 3.0 is much lower compared to all other studied models. The average learning time taken by aiGeneR 3.0 is 86&#xb5;s less taken from all the studied models together and 83&#xb5;s less compared to the previously studied TL model.</p>
<sec id="s6-1">
<title>6.1 The aiGeneR 3.0 and multi-drug resistant prediction</title>
<p>Our proposed aiGeneR 3.0 model&#x2019;s experimental results show potential in predicting multi-drug resistant (MDR) in <italic>E. coli</italic> strains. We considered the strain that resists more than two antibiotics to be in the multidrug-resistant category. We used the prediction model of logistic regression (<xref ref-type="bibr" rid="B56">Vermeiren et al., 2007</xref>) to estimate the percentage of bacteria resistant to four frequently given antibiotics: CIP, CTX, CTZ, and GEN. The percentage of resistance to each antibiotic is determined by counting the number of antibiotics that the strain is resistant to; the range is 0.25 for resistance to one antibiotic and 1 for resistance to all four antibiotics. The performance of our deployed resistant prediction model achieves a prediction accuracy of 98%, and the other model metrics for all the studied models are shown in <xref ref-type="table" rid="T7">Table 7</xref>.</p>
<table-wrap id="T7" position="float">
<label>TABLE 7</label>
<caption>
<p>Predictive model metrics for MDR (all studied models).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">MSE (Train)</th>
<th align="center">R<sup>2</sup>
</th>
<th align="center">RMSE</th>
<th align="center">MSE (new data)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">SVM &#x2b; RBF</td>
<td align="center">0.00349</td>
<td align="center">0.920</td>
<td align="center">0.059</td>
<td align="center">0.00380</td>
</tr>
<tr>
<td align="center">DT</td>
<td align="center">0.00680</td>
<td align="center">0.800</td>
<td align="center">0.082</td>
<td align="center">0.00700</td>
</tr>
<tr>
<td align="center">KNN</td>
<td align="center">0.00300</td>
<td align="center">0.930</td>
<td align="center">0.055</td>
<td align="center">0.00320</td>
</tr>
<tr>
<td align="center">ResNet-50</td>
<td align="center">0.00288</td>
<td align="center">0.940</td>
<td align="center">0.053</td>
<td align="center">0.00300</td>
</tr>
<tr>
<td align="center">VGG-19</td>
<td align="center">0.00379</td>
<td align="center">0.910</td>
<td align="center">0.062</td>
<td align="center">0.00400</td>
</tr>
<tr>
<td align="center">aiGeneR 1.0</td>
<td align="center">0.00141</td>
<td align="center">0.960</td>
<td align="center">0.037</td>
<td align="center">0.00160</td>
</tr>
<tr>
<td align="center">1D-CNN</td>
<td align="center">0.00090</td>
<td align="center">0.980</td>
<td align="center">0.030</td>
<td align="center">0.00100</td>
</tr>
<tr>
<td align="center">CNN Ensemble</td>
<td align="center">0.00081</td>
<td align="center">0.985</td>
<td align="center">0.028</td>
<td align="center">0.00090</td>
</tr>
<tr>
<td align="center">CNN TL</td>
<td align="center">0.00070</td>
<td align="center">0.990</td>
<td align="center">0.027</td>
<td align="center">0.00080</td>
</tr>
<tr>
<td align="center">aiGeneR 3.0</td>
<td align="center">
<bold>0.00054</bold>
</td>
<td align="center">
<bold>0.994</bold>
</td>
<td align="center">
<bold>0.023</bold>
</td>
<td align="center">
<bold>0.00051</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values show the best performance result in our study.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The experimental result for MDR prediction witnessed a 98% accuracy rate; our model demonstrated exceptional predictive performance and resilience in detecting variations in MDR. The model&#x2019;s lowest mean squared error (MSE) during training (0.00054) was found during the model performance evaluation, demonstrating how closely the predicted resistance percentages matched the actual values. Moreover, the high R-squared value of 0.9940 indicates that our model may explain a considerable amount of variability in the resistance percentages across strains. The model&#x2019;s accuracy for predicting levels of resistance is further demonstrated by the root mean squared error (RMSE) of 0.02327.</p>
<p>Our model&#x2019;s active predictive capacity was tested using fresh data, and its MSE of 0.00051 confirmed its generalizability and dependability in practical settings. Together, these findings highlight the precision and effectiveness of our suggested aiGeneR 3.0 model in identifying multi-drug resistance in <italic>E. coli</italic> strains, providing crucial data for directing antibiotic treatment procedures and battling antibiotic resistance. As we obtained the best version of our proposed aiGeneR 3.0 with a moderate learning rate, with an increase in training data (80%), we intended to keep this for multi-drug identification. In addition to this, the model parameters like learning rate, CV, train-test split, and dropout rate of aiGeneR 3.0 are 0.001, 5, 8:2, and 0.5, respectively.</p>
</sec>
</sec>
<sec id="s7">
<title>7 Validation</title>
<p>Accurate evaluation of model performance is a crucial component in building reliable and efficient prediction models. The ability to assess how well a model performs is an important indicator of its suitability for solving practical issues in many fields, including ML and scientific inquiry (<xref ref-type="bibr" rid="B3">Bellazzi and Zupan, 2008</xref>). In this section, we take a close look at our proposed models and evaluate them thoroughly, taking into account many criteria so that users can understand their strengths and weaknesses. We examine numerous critical aspects to evaluate the model&#x2019;s performance in different contexts. Each section delves into a different facet of the model&#x2019;s performance and thoroughly analyzes its efficacy.</p>
<sec id="s7-1">
<title>7.1 Effect of Training size</title>
<p>The comparison of classification accuracy on test data and all conceivable train-test splits on the used dataset is shown in <xref ref-type="table" rid="T8">Table 8</xref>. According to the PE (<xref ref-type="sec" rid="s4">section 4</xref>), the objective is to monitor the effect of data size on the model&#x2019;s performance. As the proportion of training data increases, the accuracy of the aiGeneR 3.0 classification model rapidly increases. It is observed that the model achieves its highest accuracy of 92% when the train-to-test split ratio is 80:20, as shown in <xref ref-type="table" rid="T8">Table 8</xref>.</p>
<table-wrap id="T8" position="float">
<label>TABLE 8</label>
<caption>
<p>Minimum unseen cases and samples are required for the generalization of individual models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">&#x23; Unseen samples</th>
<th align="center">&#x23; Cases</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">SVM &#x2b; RBF</td>
<td align="center">80</td>
<td align="center">648</td>
</tr>
<tr>
<td align="center">DT</td>
<td align="center">80</td>
<td align="center">648</td>
</tr>
<tr>
<td align="center">KNN</td>
<td align="center">80</td>
<td align="center">648</td>
</tr>
<tr>
<td align="center">ResNet-50</td>
<td align="center">80</td>
<td align="center">648</td>
</tr>
<tr>
<td align="center">VGG-19</td>
<td align="center">80</td>
<td align="center">648</td>
</tr>
<tr>
<td align="center">aiGeneR 1.0</td>
<td align="center">80</td>
<td align="center">648</td>
</tr>
<tr>
<td align="center">1D-CNN</td>
<td align="center">80</td>
<td align="center">648</td>
</tr>
<tr>
<td align="center">CNN Ensemble</td>
<td align="center">80</td>
<td align="center">648</td>
</tr>
<tr>
<td align="center">CNN TL</td>
<td align="center">80</td>
<td align="center">648</td>
</tr>
<tr>
<td align="center">aiGeneR 3.0</td>
<td align="center">
<bold>70</bold>
</td>
<td align="center">
<bold>567</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values show the best performance result in our study.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>During our experiments, we observed that the studied AI models require more training data for generalization compared to our proposed aiGeneR 3.0 model. If we set the performance threshold as classification accuracy, then our aiGeneR 3.0 takes only 70% (567 unseen cases) of the data to achieve this trademark. Similarly, all of the implemented ML and DL models take 80% of the unseen data to obtain the generalization standard. The generalization of our aiGeneR 3.0 requires 10% less unseen data to obtain the best classification accuracy, proving that our model can be generalized by utilizing fewer strains than all other studied models.</p>
</sec>
<sec id="s7-2">
<title>7.2 Confusion matrix</title>
<p>The matrix shown in <xref ref-type="fig" rid="F9">Figure 9</xref> has significant diagonal dominance, indicating that the model predicted the proper class with few misclassifications. Most GEN-resistant strains (142 of 154) were correctly classified, with only a few CTZ and CTX misassignments. The model also predicted CIP, a smaller class, with great accuracy (148 out of 156 properly classified), demonstrating its class imbalance resilience. The confusion matrix yielded class-wise measurements. Each class has good precision, sensitivity (recall), and specificity, indicating that the model minimized false positives and negatives. Minority class CIP had good sensitivity and specificity, showing that the model did not underperform on underrepresented categories, a major antimicrobial resistance prediction difficulty.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Confusion matrix of aiGeneR 3.0 on all four antibiotics.</p>
</caption>
<graphic xlink:href="fgene-16-1651917-g009.tif">
<alt-text content-type="machine-generated">Confusion matrix showing true versus predicted labels for four categories: GEN, CTZ, CTX, and CIP. The diagonal shows high accuracy with numbers 142, 143, 146, and 148. Off-diagonal values represent misclassifications. A color gradient indicates frequency intensity, with darker blue signifying higher values.</alt-text>
</graphic>
</fig>
<p>The model for GEN has lower specificity (76%) than sensitivity (93%), suggesting reliable identification of susceptible strains but a little probability of under-detection of resistant isolates. CTZ had 91% sensitivity and 85% specificity, recognizing resistant bacteria with minimal false-positive rates. CTX and CIP had a stable finding, with sensitivity and specificity exceeding 92%, indicating robust class classification. CIP&#x2019;s sensitivity (93%) and specificity (94%) were the best, detecting resistant bacteria and identifying vulnerable ones. These results show that aiGeneR 3.0 consistently supports better specificity while maintaining excellent sensitivity, ensuring reliable detection of minority resistant strains without inflating misleading resistance predictions. The confusion matrix confirms that aiGeneR 3.0 demonstrated balanced predictive performance among antibiotics, with &#x223c;92% accuracy, 92% precision, 91% sensitivity, 95% specificity, 91% F1-score, and 87% MCC.</p>
</sec>
<sec id="s7-3">
<title>7.3 Receiver operating curves</title>
<p>The Receiver Operating Characteristic (ROC) curve is an essential metric for evaluating the effectiveness of a classification model. In this study, we conduct a performance analysis of our suggested aiGeneR 3.0 model in comparison to other studied models, with a significance level of p &#x3d; 0.001. K-5 cross-validation is employed to determine the variation in the accuracy of each model as the quantity of training data changes.</p>
<p>The ROC performance of all the studied models is shown in <xref ref-type="fig" rid="F10">Figure 10</xref>. The proposed model, aiGeneR 3.0, has achieved a significant milestone by achieving a strong Area Under the Curve (AUC) value of 98.48%. Nevertheless, when compared to other classification models, the ROC value of the 1-D CNN is the lowest (93.47%). Compared to the previously developed CNN TL model and despite the hurdles posed by the imbalance and small dataset, our proposed aiGeneR 3.0 achieves the highest AUC value in the identification of the resistant strains.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>ROC-AUC of all the studied models with p-value &#x3c;0.001.</p>
</caption>
<graphic xlink:href="fgene-16-1651917-g010.tif">
<alt-text content-type="machine-generated">ROC curves comparing various models with the x-axis as the False Positive Rate and the y-axis as the True Positive Rate. The legend lists models with their AUC values and p-values: 1-D CNN, aiGeneR 1.0, DT, CNN Ensemble, ResNet-50, VGG-19, SVM-RBF, CNN-TL, KNN, and aiGeneR 3.0, which has the highest AUC of 0.9848.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s7-4">
<title>7.4 Model generalization</title>
<p>In the validation phase of our aiGeneR 3.0 model, we utilize an openly available and highly imbalanced dataset (<xref ref-type="bibr" rid="B34">Moradigaravand et al., 2018</xref>). The detailed characteristics of the dataset are summarized in <xref ref-type="table" rid="T9">Table 9</xref>. There is a high imbalance in the susceptible-to-resistant ratio in all four antibiotics taken for validation of our aiGeneR 3.0 model. It can be seen from the table that the ratio is very high in the case of CTZ and GEN (&#x2248;1:7) there is a slight increase in the ratio for CTX and CIP (&#x2248;1:4). We perform the model validation in two different phases first, we have considered four different datasets based on four individual antibiotics and secondly, prepare the dataset by combining all the four antibiotics into one dataset.</p>
<table-wrap id="T9" position="float">
<label>TABLE 9</label>
<caption>
<p>Characteristics of the validation dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Antibiotics</th>
<th align="center">GEN</th>
<th align="center">CTZ</th>
<th align="center">CTX</th>
<th align="center">CIP</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">&#x23; Susceptible</td>
<td align="center">1,651</td>
<td align="center">1,670</td>
<td align="center">1,476</td>
<td align="center">1,508</td>
</tr>
<tr>
<td align="center">&#x23; Resistance</td>
<td align="center">284</td>
<td align="center">265</td>
<td align="center">459</td>
<td align="center">427</td>
</tr>
<tr>
<td align="center">Total</td>
<td align="center">1,935</td>
<td align="center">1,935</td>
<td align="center">1,935</td>
<td align="center">1,935</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We tested the efficacy of our proposed model on the four individual antibiotics considered for our experiments in the publicly available data, and aiGeneR 3.0 holds the consistency and remains the best performer in terms of classification accuracy, specificity, and sensitivity. In <xref ref-type="table" rid="T10">Table 10</xref>, we summarize the performance of aiGeneR 3.0 on individual datasets.</p>
<table-wrap id="T10" position="float">
<label>TABLE 10</label>
<caption>
<p>Validation model metrics of aiGeneR 3.0.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Dataset</th>
<th align="center">Acc (%)</th>
<th align="center">Pre (%)</th>
<th align="center">Sen (%)</th>
<th align="center">Spe (%)</th>
<th align="center">F1 (%)</th>
<th align="center">MCC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">CIP</td>
<td align="center">89</td>
<td align="center">94</td>
<td align="center">92</td>
<td align="center">70</td>
<td align="center">93</td>
<td align="center">0.60</td>
</tr>
<tr>
<td align="center">CTX</td>
<td align="center">93</td>
<td align="center">98</td>
<td align="center">94</td>
<td align="center">88</td>
<td align="center">96</td>
<td align="center">0.74</td>
</tr>
<tr>
<td align="center">CTZ</td>
<td align="center">90</td>
<td align="center">97</td>
<td align="center">91</td>
<td align="center">79</td>
<td align="center">94</td>
<td align="center">0.60</td>
</tr>
<tr>
<td align="center">GEN</td>
<td align="center">89</td>
<td align="center">96</td>
<td align="center">91</td>
<td align="center">77</td>
<td align="center">94</td>
<td align="center">0.59</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>It is observed from the above table that, during validation of aiGeneR 3.0 with individual antibiotics data, we obtained a higher classification accuracy in the case of CTX, and this is due to the higher strain ratio compared to the other three antibiotics datasets. aiGeneR 3.0 achieves the second-highest classification accuracy in the case of CTZ (90%), followed by CIP and GEN (89%).</p>
<p>Similarly, while we tested aiGeneR 3.0 along with other studied models on the dataset that combines all four antibiotics, we observed that aiGeneR 3.0 achieves the highest classification accuracy (90%), as shown in <xref ref-type="table" rid="T11">Table 11</xref>. The sensitivity and specificity of aiGeneR 3.0 are 97% and 76%, respectively, which is the highest among all the studied models, and this is due to the one-hot encoding we adopt in our study. In addition to this, the SVM, aiGeneR 1.0, CNN TL, CNN ensemble, and ResNet-50 achieve a remarkable sensitivity of more than 90%, whereas CNN TL and ResNet-50 achieve a specificity just higher than 70%. The studied model metrics of the validation phase are shown in <xref ref-type="fig" rid="F11">Figure 11</xref>. However, in the validation phase, we observed that the ResNet-50, CNN ensembled, and TL performed better than aiGeneR 1.0. This is due to the effectiveness and automated feature extraction techniques with these models compared to our previously developed aiGeneR 1.0, which is based on traditional feature selection techniques. This validation outcome may provide insight into the use of automated and effective feature selection techniques, especially DL, for future resistant strain identification.</p>
<table-wrap id="T11" position="float">
<label>TABLE 11</label>
<caption>
<p>Model metrics of all the studied models on the validation dataset (all four antibiotics).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Models</th>
<th align="center">Acc (%)</th>
<th align="center">Pre (%)</th>
<th align="center">Sen (%)</th>
<th align="center">Spe (%)</th>
<th align="center">F1 (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">KNN</td>
<td align="center">68</td>
<td align="center">68</td>
<td align="center">84</td>
<td align="center">45</td>
<td align="center">75</td>
</tr>
<tr>
<td align="center">DT</td>
<td align="center">73</td>
<td align="center">72</td>
<td align="center">88</td>
<td align="center">51</td>
<td align="center">79</td>
</tr>
<tr>
<td align="center">SVM</td>
<td align="center">75</td>
<td align="center">72</td>
<td align="center">90</td>
<td align="center">53</td>
<td align="center">80</td>
</tr>
<tr>
<td align="center">VGG-19</td>
<td align="center">75</td>
<td align="center">75</td>
<td align="center">87</td>
<td align="center">83</td>
<td align="center">81</td>
</tr>
<tr>
<td align="center">aiGeneR 1.0</td>
<td align="center">76</td>
<td align="center">75</td>
<td align="center">91</td>
<td align="center">55</td>
<td align="center">80</td>
</tr>
<tr>
<td align="center">1D-CNN</td>
<td align="center">77</td>
<td align="center">79</td>
<td align="center">88</td>
<td align="center">57</td>
<td align="center">83</td>
</tr>
<tr>
<td align="center">CNN Ensemble</td>
<td align="center">77</td>
<td align="center">75</td>
<td align="center">92</td>
<td align="center">56</td>
<td align="center">82</td>
</tr>
<tr>
<td align="center">CNN TL</td>
<td align="center">87</td>
<td align="center">85</td>
<td align="center">96</td>
<td align="center">71</td>
<td align="center">90</td>
</tr>
<tr>
<td align="center">ResNet-50</td>
<td align="center">88</td>
<td align="center">87</td>
<td align="center">95</td>
<td align="center">72</td>
<td align="center">91</td>
</tr>
<tr>
<td align="center">aiGeneR 3.0</td>
<td align="center">
<bold>90</bold>
</td>
<td align="center">
<bold>89</bold>
</td>
<td align="center">
<bold>97</bold>
</td>
<td align="center">
<bold>76</bold>
</td>
<td align="center">
<bold>92</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values show the best performance result in our study.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>Model metrics of all the studied models during the model validation phase.</p>
</caption>
<graphic xlink:href="fgene-16-1651917-g011.tif">
<alt-text content-type="machine-generated">Bar chart comparing performance metrics of various AI models: KNN, DT, SVM, VGG-19, aiGeneR 1.0, 1D-CNN, CNN Ensemble, CNN TL, ResNet-50, and aiGeneR 3.0. Metrics include Accuracy (Acc), Precision (Pre), Sensitivity (Sen), Specificity (Spe), and F1 Score (F1), depicted by different colored bars. Performance percentage values vary, with aiGeneR 3.0 generally showing high scores across all metrics.</alt-text>
</graphic>
</fig>
<p>The performance of the designed model on every conceivable train-test split and the comparison of classification accuracy on test data were also explored in this study. The learning model is impacted by the amount of training data, which also helps the model generalize effectively to new data. Using a dataset with various train-test splits, we assess our suggested model, aiGeneR 3.0, and the four other classifiers employed in this investigation. It has been noted that while other models require a more significant number of cases for generalization, aiGeneR 3.0 requires a small number of cases. This section thoroughly explains how data size affects our suggested model. <xref ref-type="fig" rid="F12">Figure 12</xref> displays the comparison of classification accuracy on test data as well as all conceivable train-test splits on the utilized dataset.</p>
<fig id="F12" position="float">
<label>FIGURE 12</label>
<caption>
<p>Performance of all the studied models with different train-test splits.</p>
</caption>
<graphic xlink:href="fgene-16-1651917-g012.tif">
<alt-text content-type="machine-generated">Line graph showing model performance with varying sample sizes. The x-axis represents training data size, and the y-axis shows accuracy in percent. Multiple models are compared, including aiGeneR 3.0, which outperforms others, achieving over 90% accuracy with higher sample sizes. Other models like KNN, DT, SVM, and CNN variants show varying performance increases as sample size grows.</alt-text>
</graphic>
</fig>
<p>It can be observed from the figure that the ML models require more training samples to obtain generalization in classification accuracy than the DL models. While we compare the top three ML (SVM &#x2b; DT &#x2b; KNN) with the top three DL (CNN TL &#x2b; aiGeneR 1.0 &#x2b; aiGeneR 3.0) models, there is a significant difference in the train-test split for learning models to achieve their best results. Compared to the top three DL models, the top three ML models take 65.9% more data to be generalized. The other DL models studied in this work take a range of 55%&#x2013;75% unseen cases to obtain their generalization. In addition to this, our proposed aiGeneR 3.0 requires 70% (567 cases) of data to generalize and obtain a stable classification accuracy. This performance outcome of aiGeneR 3.0 showcases the model&#x2019;s generalization ability with a very small number of unseen data, which leads to its chances of better performance with real-time data.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s8">
<title>8 Discussion</title>
<p>The results show that the aiGeneR 3.0 model effectively detects resistance strains without using any known resistance strains during model training. However, there are some limitations to be aware of due to variations in dataset sizes and methods. We implemented our suggested aiGeneR 3.0 model using a basic model architecture and then applied it to a publicly available, imbalanced, and noisy dataset. When given balanced antibiotic data, learning models perform much better in terms of accuracy, and we fine-tuned aiGeneR 3.0 to consistently classify each drug. In comparison to other conventional ML models used in our study, aiGeneR 3.0&#x2019;s computational time is much lower.</p>
<p>We observed that, because typical one-hot encoding introduces a relative scale with numbers like 1, 2, 3, and 4, higher numerical values may inadvertently dominate or introduce bias during the learning process in certain deep learning models. Using bigger numerical representations may result in learning disparities in certain deep learning models, especially those that are sensitive to input magnitudes (models that ineffectively normalize weights), even if one-hot encoding is categorical and theoretically scale-invariant. By limiting the range to 0.25&#x2013;1, biases resulting from magnitude are less likely to occur, and a more consistent expression is assured. We found that scaling to a smaller range improved training convergence and made the gradient updates of our models more reliable.</p>
<p>The deployed aiGeneR 3.0 model has a straightforward design that can deliver good classification accuracy. Among the most advanced ML and DL models we tested, aiGeneR 3.0 yielded the best classification accuracy. The following is a list of some of the major study findings we came across while doing this work: a simple and effective model architecture can achieve better classification accuracy, minimal computational cost, antimicrobial resistance (AMR) analysis, and antibiotic resistance strain identification.<list list-type="simple">
<list-item>
<p>&#x2022; The proposed aiGeneR 3.0 has a simple deep network architecture and has the potential of a good learning model by providing relatively higher classification accuracy to identify the resistant strains.</p>
</list-item>
<list-item>
<p>&#x2022; The aiGeneR 3.0 requires less computational time compared to all the studied models in this work.</p>
</list-item>
<list-item>
<p>&#x2022; The multi-drug prediction ability with significant minute errors is a major contribution of our proposed aiGeneR 3.0 model.</p>
</list-item>
<list-item>
<p>&#x2022; The aiGeneR 3.0 can effectively identify the resistant strains with a classification accuracy of 92% which is the highest among all the studied models.</p>
</list-item>
<list-item>
<p>&#x2022; Model generalization of aiGeneR 3.0 persists in its classification potential and proves the ability of our proposed model to handle imbalanced and unseen data.</p>
</list-item>
</list>
</p>
<sec id="s8-1">
<title>8.1 Claim</title>
<p>Our cutting-edge study reveals that aiGeneR 3.0 is an excellent resource for identifying strains of antibiotic resistance; it can handle imbalanced and constrained datasets with ease. Through the utilization of sophisticated DL algorithms, aiGeneR 3.0 achieves better classification accuracy, as shown in <xref ref-type="table" rid="T12">Table 12</xref>.</p>
<table-wrap id="T12" position="float">
<label>TABLE 12</label>
<caption>
<p>Benchmarking parameters of the studied state-of-the-art ML and DL techniques for AMR analysis.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Authors</th>
<th align="left">Objective</th>
<th align="left">Dataset</th>
<th align="left">Techniques</th>
<th align="left">Performance evaluation</th>
<th align="left">Limitations</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<xref ref-type="bibr" rid="B40">Pesesky et al. (2016)</xref>
</td>
<td align="left">Predict AR patterns in Bacilli</td>
<td align="left">Genome Sequence</td>
<td align="left">ML and Rule-based approaches</td>
<td align="left">Acc (Resfams &#x3d; 94.9%, Resfinder &#x3d; 85.9%, CARD &#x3d; 57.7%</td>
<td align="left">The accuracy and generalizability of estimations are affected by constraints such as small sample sizes</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B34">Moradigaravand et al. (2018)</xref>
</td>
<td align="left">Identify AR in <italic>E. coli</italic> bacteria</td>
<td align="left">
<italic>E. coli</italic> strains</td>
<td align="left">LR, RF, Gradient Boosting</td>
<td align="left">Acc &#x3d; 97%, Precision &#x3d; 93%, Recall &#x3d; 83%</td>
<td align="left">A high false negative rate caused by undetected SNPs in specific locations and the accessory genome</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B24">Kavvas et al. (2018)</xref>
</td>
<td align="left">Detecting microbial tuberculosis ARG.</td>
<td align="left">Sequence</td>
<td align="left">Pan-genome Analysis, SVM, LR</td>
<td align="left">AUC &#x3d; 0.80</td>
<td align="left">The challenges include reference bias, data selection bias, and the necessity for experimental validation</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B28">Liu et al. (2020)</xref>
</td>
<td align="left">Predicting the AR in A. pleuropneumonia</td>
<td align="left">Genome sequence</td>
<td align="left">SVM, SCM</td>
<td align="left">Tet (Acc &#x3d; 97%), Amp(Acc &#x3d; 100%), Sul (Acc &#x3d; 100%)</td>
<td align="left">Problems with the size of the dataset, biases, and the generalizability of the model</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B16">Green et al. (2022)</xref>
</td>
<td align="left">Prediction of M.tuberculosis ARGs</td>
<td align="left">WGS</td>
<td align="left">CNN</td>
<td align="left">AUC for MD-CNN &#x3d; 91.2%<break/>AUC for SD-CNN &#x3d; 93.8%</td>
<td align="left">A higher computational cost is required to validate the results</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B27">Li et al. (2022)</xref>
</td>
<td align="left">Identification of new antimicrobial peptides</td>
<td align="left">Sequence</td>
<td align="left">AMPlify, Bi-LSTM, RNN, CNN</td>
<td align="left">Acc &#x3d; 93.71%, F1-Score &#x3d; 93.66%, AUROC &#x3d; 98.37%</td>
<td align="left">Inadequate training data is the reason for the difficulties in training AMP models</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B13">Florensa et al. (2022)</xref>
</td>
<td align="left">Identifying ARGs in NGS data</td>
<td align="left">NGS data</td>
<td align="left">ML, NGS</td>
<td align="left">-</td>
<td align="left">An issue with the study&#x2019;s findings is that it lacks any data for the ResFinder tool&#x2019;s performance parameters, such as sensitivity, specificity, and precision</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B43">Ren et al. (2022b)</xref>
</td>
<td align="left">Predicting AMR.</td>
<td align="left">WGS (<italic>E. coli</italic>)</td>
<td align="left">CNN</td>
<td align="left">CIP (Acc &#x3d; 91%), CTX (Acc &#x3d; 78%), GEN (Acc &#x3d; 78%)</td>
<td align="left">Issues with computing resources, interpretability, external validation, and dataset size</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B17">Grey et al. (2023)</xref>
</td>
<td align="left">UTI current condition review</td>
<td align="left">-</td>
<td align="left">Culture, AI</td>
<td align="left">Acc &#x3d; 93.22%&#x2013;98.80%</td>
<td align="left">This study fails to determine treatment accuracy and ARG identification</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B1">Almaghrabi et al. (2024)</xref>
</td>
<td align="left">Analyze resistance genes in <italic>P. aeruginosa</italic>
</td>
<td align="left">WGS</td>
<td align="left">Web-based tools</td>
<td align="left">MDR &#x3d; 77.1%</td>
<td align="left">This study lacks the ability to draw clear epidemiological connections between environmental and clinical isolates</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B23">Jin et al. (2024)</xref>
</td>
<td align="left">AMR prediction in <italic>E. coli</italic>
</td>
<td align="left">WGS</td>
<td align="left">RF, SVM, LR, CNN</td>
<td align="left">F1-Sc:82%, MCC:48%, AUC:77%</td>
<td align="left">The learning rate of the model is very low</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B14">Gao et al. (2024)</xref>
</td>
<td align="left">AMR in A. baumannii</td>
<td align="left">Sequence</td>
<td align="left">ML model</td>
<td align="left">Acc &#x3d; 96%</td>
<td align="left">The analysis limits may affect the model&#x2019;s generalizability, misbalancing, etc.</td>
</tr>
<tr>
<td align="left">aiGeneR 3.0 [Proposed]</td>
<td align="left">Predict MDR and antimicrobial-resistant strains</td>
<td align="left">WGS (<italic>E. coli</italic>)</td>
<td align="left">ML/DL</td>
<td align="left">Acc &#x3d; 93%, Sen &#x3d; 90%, Spe &#x3d; 95%, ROC &#x3d; 99%</td>
<td align="left">Data augmentation and advanced computational techniques, such as a transformer, may play a crucial role in increasing classification accuracy</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>By minimizing variance and keeping feature scaling consistent, this method stabilizes the training process, which in turn produces smoother gradients and avoids problems like bursting gradients (<xref ref-type="bibr" rid="B11">dos Santos and Papa, 2022</xref>). We found that our studied DL models performed much better when we used a one-hot encoding range of 0.25-1 rather than 1&#x2013;4. With a 2% increase in specificity and a 1% improvement in precision, our empirical data demonstrated better accuracy and generalizability on both the validation and test sets. In addition, our model was able to generalize well to a different dataset (<xref ref-type="bibr" rid="B34">Moradigaravand et al., 2018</xref>), which further proves how effective and resilient this proposed normalized feature range is for the learning of the deployed DL models.</p>
<p>The validation confirms the edge of aiGeneR 3.0, demonstrating its capacity to surpass rivals with small input data. Its processing cost is minimized, and its simple design gives it the ability to run on typical personal computers and laptops, ensuring better classification accuracy. Furthermore, a comprehensive power analysis reveals aiGeneR 3.0&#x2019;s capacity to surpass the desired number of training cases, underscoring its potential for further refinement and expansion. Additionally, the significant AUC value of 98.48% shows the potential of our aiGeneR 3.0 toward its adaptability and learning capacity with imbalanced datasets. Overall, our research shows that aiGeneR 3.0 is an innovative breakthrough that will change how we diagnose diseases and, more generally, not just when identifying strains of antibiotic resistance.</p>
</sec>
<sec id="s8-2">
<title>8.2 Special notes</title>
<p>We designed the cutting-edge aiGeneR 3.0 model, a DL-based AMR analysis tool, to use double-mutated gene data to predict multi-drug resistance and detect antibiotic resistance strains without prior knowledge of known ARGs. The powerful DL model, LSTM, and LR combination in aiGeneR 3.0 advances AMR analysis, particularly multi-drug prediction. The primary notable accomplishments of our aiGeneR 3.0 framework are as follows:<list list-type="simple">
<list-item>
<p>&#x2022; We proposed aiGeneR 3.0, an AI model with a simple and robust architecture that can handle imbalances and small genomics data.</p>
</list-item>
<list-item>
<p>&#x2022; The aiGeneR 3.0 can predict the resistant strain from a double-mutated gene sequence with higher classification accuracy compared to previous studies.</p>
</list-item>
<list-item>
<p>&#x2022; aiGeneR 3.0 offers an ultimate ability to predict the multi-drug resistance in strains with 98% prediction accuracy.</p>
</list-item>
<list-item>
<p>&#x2022; The generalization and scientific validation of aiGeneR 3.0 prove its potential to handle small and imbalanced (curse of dimensionality) gene data.</p>
</list-item>
<list-item>
<p>&#x2022; The benchmarking of aiGeneR 3.0 with other state-of-the-art- AI models enhance its adaptability for real-time implementation.</p>
</list-item>
</list>
</p>
<p>This proposed aiGeneR 3.0 model has the ability to identify <italic>E. coli</italic> bacteria that are resistant to antibiotics, which could be useful for antimicrobial stewardship initiatives. The approach has the potential to lower the usage of inefficient medicines and limit the use of broad-spectrum antibiotics by offering early insights about resistance profiles, which could support antibiotic selection that is tailored to individual patients. Subject to medical validation, the fast prediction capacity suggests real-time clinical decision support. Hospital monitoring systems can benefit from aiGeneR 3.0, which could have applications beyond individual patient care and bolster initiatives to combat new resistance tendencies. However, before these applications are widely used in ordinary practice, more multicenter clinical trials and prospective validation must be conducted.</p>
</sec>
<sec id="s8-3">
<title>8.3 Limitations</title>
<p>This highly motivated study focuses on identifying resistant strains utilizing imbalances and small datasets. Data augmentation can be used to further this study and potentially improve model performance. However, because it is medically incorrect, experts do not advise using this approach (the augmentation of medical data). Better model metrics might be obtained if the model were trained using synthetic data. Further study could address a few biases in our model, such as (i) smaller studies are found related to our work, (ii) SNP filtering threshold applied during preprocessing, which may have influenced the set of variants included for model training (iii) the use of data augmentation, (iv) comparisons with other ML and recently trending DL models like deep network with an attention mechanism. (v) a summary of the benchmarking studies and (vi) no remarks regarding the clinical validation (<xref ref-type="bibr" rid="B39">Paul et al., 2022</xref>; <xref ref-type="bibr" rid="B12">Eskofier et al., 2016</xref>; <xref ref-type="bibr" rid="B19">Hu et al., 2021</xref>).</p>
<p>In addition to the above, AiGeneR 3.0 has a few limitations despite its better predictive accuracy. First, despite balancing and preprocessing, the datasets had class imbalances that potentially bias predictions toward the majority class. The model may learn dataset-specific artifacts instead of generalizable biological patterns when training on short or noisy datasets, increasing the risk of overfitting.</p>
</sec>
<sec id="s8-4">
<title>8.4 Extension</title>
<p>This work focuses on applying DL and AI models to resistant strain identification and classification. The proposed aiGeneR 3.0 is now considered a benchmark in the field of AMR analysis due to its great improvement in detecting resistant strains. The aiGeneR 3.0 model performs highly compared to earlier research (<xref ref-type="bibr" rid="B43">Ren Y. et al., 2022</xref>) on resistant strain identification. Furthermore, cross-validation and unseen implementations show the system&#x2019;s endurance, domain flexibility, and capacity to function well in domains other than the one on which it was trained. In extension, the application of other DL models, especially transformer architecture with attention mechanism, and Dl model hyper-parameter optimization may be adopted to validate their efficacy in identifying multi-drug resistance in double-mutated genome sequences.</p>
<p>Despite aiGeneR 3.0&#x2019;s capabilities, small or imbalanced datasets risk overfitting, when the model learns training data-specific patterns or noise rather than generalizable correlations. Even high accuracy on the training set may not guarantee accurate predictions on unseen data. Imbalanced class distributions bias the model toward majority classes, making unusual antibiotic-resistant organisms harder to detect. Additional variability can worsen model performance in real-world deployment. Different laboratory techniques, sample preparation methods, sequencing platforms, and batch effects introduce systematic variances that may not be in the training data. Sequencing errors or missing data can skew input features, and variable sample distributions across populations may produce patterns the model has not learnt, raising misclassification risk. Parameters and techniques like Cross-validation, data augmentation, and regularization (dropout, weight decay, and early stopping) need to be tested in a wider range. Finally, ongoing retraining with new datasets adapts the model to changing data distributions, ensuring robustness and reliability in clinical or laboratory contexts. Future research should use explainable AI methods like feature attribution or pathway-level analysis to identify resistance-predicting genes or biological markers. For clinical implementation, the model needs to be verified across larger, multi-center cohorts to account for sequencing procedures, sample handling, and patient demographics. To maintain accuracy and dependability in clinical operations, rigorous benchmarking, seamless software interfaces, real-time processing, and constant retraining with new resistance data are needed.</p>
</sec>
</sec>
<sec id="s9">
<title>9 Benchmarking</title>
<p>At its core, our study centered on identifying resistant strains using a DL model that combined advanced techniques with a simple design. Along with this, we also want to make sure that our pipeline does not lose consistency when applied to small or imbalanced datasets. Research shows that few studies have used DL models to identify resistance strains in double-mutated WGS. This set of AI models was constructed by merging several deep networks. Hence, it is essential to evaluate our method for previous AI models. In light of this, we choose to tackle the benchmarking efforts head-on by comparing our proposed models to earlier DL/ML models used in AMR for resistant strain classification and other disease investigations.</p>
<p>Our suggested aiGeneR 3.0 has a higher classification accuracy of 93% and can manage imbalanced data because of its streamlined model architecture. Furthermore, the computing time required for both the learning phase and the prediction of resistant strains in aiGeneR 3.0 is significantly lower compared to previously examined DL models. Furthermore, the validation of aiGeneR 3.0 establishes it as a strong and versatile model for classifying antibiotic-resistant strains and predicting multidrug resistance.</p>
</sec>
<sec sec-type="conclusion" id="s10">
<title>10 Conclusion</title>
<p>In this work, we used double-mutated <italic>E. coli</italic> NGS WGS data to show how effective aiGeneR 3.0 is in identifying bacteria that are resistant to antibiotics. Additionally, it presents the multi-drug-resistance patterns identified in the resistant strains. aiGeneR 3.0 is a hybrid computational method that employs advanced LSTM architecture and NGS data to discern resistant and susceptible strains within small and highly unbalanced datasets. Primarily, our aiGeneR 3.0 model enhances the accuracy of classification and prediction compared to earlier investigated models. The remarkable performance of our proposed pipeline is evidenced by aiGeneR 3.0, achieving a 92% accuracy, with a sensitivity of 91% and a specificity of 95% in finding resistant strains. aiGeneR 3.0 attains a 98% prediction accuracy in multi-drug identification, accompanied by a minimal MSE of 0.00054 and an RMSE of 0.02327.</p>
<p>Our study emphasizes the promise of predictive modelling utilizing NGS data and DL techniques to tackle the escalating problem of antibiotic resistance, perhaps leading to the creation of novel therapies. The ability of aiGeneR 3.0 to consistently and extensively generate models indicates its prospective usefulness in AMR research moving forward. Antibiotic resistance is emerging as a critical concern in the field of infectious diseases. Our research enhances our comprehension of the issue, enables us to predict its future trajectories, and eventually aids in addressing it. Due to the numerous constraints in identifying resistance patterns across different strains resulting from the limited number of strains, we want to employ deep learning models on whole genome sequencing with various augmentation techniques in our future research to find resistant strains.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s11">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec sec-type="author-contributions" id="s12">
<title>Author contributions</title>
<p>DN: Conceptualization, Data curation, Formal Analysis, Methodology, Project administration, Validation, Writing &#x2013; original draft. AbP: Conceptualization, Data curation, Investigation, Software, Validation, Writing &#x2013; original draft. AmP: Conceptualization, Data curation, Project administration, Resources, Supervision, Writing &#x2013; original draft, Validation. MK: Methodology, Writing &#x2013; review and editing, Formal Analysis, Supervision. BA: Data curation, Methodology, Validation, Writing &#x2013; original draft, Writing &#x2013; review and editing, Conceptualization, Investigation. SS: Data curation, Formal Analysis, Project administration, Writing &#x2013; original draft, Writing &#x2013; review and editing, Methodology, Validation. BS: Conceptualization, Data curation, Formal Analysis, Investigation, Writing &#x2013; original draft, Writing &#x2013; review and editing, Project administration, Software. AA: Conceptualization, Data curation, Formal Analysis, Supervision, Writing &#x2013; review and editing, Investigation, Validation, Writing &#x2013; original draft. SM: Data curation, Formal Analysis, Funding acquisition, Methodology, Project administration, Supervision, Visualization, Writing &#x2013; original draft, Writing &#x2013; review and editing. TS: Conceptualization, Data curation, Formal Analysis, Methodology, Supervision, Visualization, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s13">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. Princess Nourah bint Abdulrahman University Researchers Supporting Project number (PNURSP2025R440), Princess Nourah bint Abdulrahman University, Riyadh, Saudi Arabia.</p>
</sec>
<sec sec-type="COI-statement" id="s14">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The author(s) declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec sec-type="ai-statement" id="s15">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s16">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Almaghrabi</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Macori</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Sheridan</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>McCarthy</surname>
<given-names>S. C.</given-names>
</name>
<name>
<surname>Floss-Jones</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Fanning</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Whole genome sequencing of resistance and virulence genes in multi-drug resistant Pseudomonas Aeruginosa</article-title>. <source>J. Infect. Public Health</source> <volume>17</volume> (<issue>2</issue>), <fpage>299</fpage>&#x2013;<lpage>307</lpage>. <pub-id pub-id-type="doi">10.1016/j.jiph.2023.12.012</pub-id>
<pub-id pub-id-type="pmid">38154433</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arango-Argoty</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Garner</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Pruden</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Heath</surname>
<given-names>L. S.</given-names>
</name>
<name>
<surname>Vikesland</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Deeparg: a deep learning approach for predicting antibiotic resistance genes from metagenomic data</article-title>. <source>Microbiome</source> <volume>6</volume>, <fpage>23</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1186/s40168-018-0401-z</pub-id>
<pub-id pub-id-type="pmid">29391044</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bellazzi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zupan</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Predictive data mining in clinical medicine: current issues and guidelines</article-title>. <source>Int. J. Med. Inf.</source> <volume>77</volume> (<issue>2</issue>), <fpage>81</fpage>&#x2013;<lpage>97</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2006.11.006</pub-id>
<pub-id pub-id-type="pmid">17188928</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boolchandani</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>D&#x2019;Souza</surname>
<given-names>A. W.</given-names>
</name>
<name>
<surname>Dantas</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Sequencing-based methods and resources to study antimicrobial resistance</article-title>. <source>Nat. Rev. Genet.</source> <volume>20</volume> (<issue>6</issue>), <fpage>356</fpage>&#x2013;<lpage>370</lpage>. <pub-id pub-id-type="doi">10.1038/s41576-019-0108-4</pub-id>
<pub-id pub-id-type="pmid">30886350</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bryce</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hay</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Lane</surname>
<given-names>I. F.</given-names>
</name>
<name>
<surname>Thornton</surname>
<given-names>H. V.</given-names>
</name>
<name>
<surname>Wootton</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Costelloe</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Global prevalence of antibiotic resistance in paediatric urinary tract infections caused by <italic>Escherichia coli</italic> and association with routine use of antibiotics in primary care: systematic review and meta-analysis</article-title>. <source>BMJ</source> <volume>352</volume>, <fpage>i939</fpage>. <pub-id pub-id-type="doi">10.1136/bmj.i939</pub-id>
<pub-id pub-id-type="pmid">26980184</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chandra</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mk</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Ke</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Mukhopadhyay</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dinesh Acharya</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Rajesh</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Antimicrobial resistance and the post antibiotic era: better late than never effort</article-title>. <source>Expert Opin. Drug Saf.</source> <volume>20</volume> (<issue>11</issue>), <fpage>1375</fpage>&#x2013;<lpage>1390</lpage>. <pub-id pub-id-type="doi">10.1080/14740338.2021.1928633</pub-id>
<pub-id pub-id-type="pmid">33999733</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Fastp: an ultra-fast all-in-one fastq preprocessor</article-title>. <source>Bioinformatics</source> <volume>34</volume> (<issue>17</issue>), <fpage>i884</fpage>&#x2013;<lpage>i890</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty560</pub-id>
<pub-id pub-id-type="pmid">30423086</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dahouda</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Joe</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A deep-learned embedding technique for categorical features encoding</article-title>. <source>IEEE Access</source> <volume>9</volume>, <fpage>114381</fpage>&#x2013;<lpage>114391</lpage>. <pub-id pub-id-type="doi">10.1109/access.2021.3104357</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Danecek</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Auton</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abecasis</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Albers</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Banks</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>DePristo</surname>
<given-names>M. A.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>The variant call format and vcftools</article-title>. <source>Bioinformatics</source> <volume>27</volume> (<issue>15</issue>), <fpage>2156</fpage>&#x2013;<lpage>2158</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btr330</pub-id>
<pub-id pub-id-type="pmid">21653522</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Das</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Mittal</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Goswami</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Adhana</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Rathore</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Prevalence of multidrug resistance (Mdr) and extended spectrum beta-lactamases (Esbls) among uropathogenic <italic>Escherichia coli</italic> isolates from female patients in a tertiary care hospital in North India</article-title>. <source>Int. J. Reproduction, Contracept. Obstetrics Gynecol.</source> <volume>7</volume> (<issue>12</issue>), <fpage>5031</fpage>&#x2013;<lpage>5037</lpage>. <pub-id pub-id-type="doi">10.18203/2320-1770.ijrcog20184961</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>dos Santos</surname>
<given-names>C. F. G.</given-names>
</name>
<name>
<surname>Papa</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Avoiding overfitting: a survey on regularization methods for convolutional neural networks</article-title>. <source>ACM Comput. Surv. (CSUR)</source> <volume>54</volume> (<issue>10s</issue>), <fpage>1</fpage>&#x2013;<lpage>25</lpage>. <pub-id pub-id-type="doi">10.1145/3510413</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Eskofier</surname>
<given-names>B. M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S. I.</given-names>
</name>
<name>
<surname>Daneault</surname>
<given-names>J.-F.</given-names>
</name>
<name>
<surname>Golabchi</surname>
<given-names>F. N.</given-names>
</name>
<name>
<surname>Ferreira-Carvalho</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Vergara-Diaz</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). &#x201c;<article-title>Recent machine learning advancements in sensor-based mobility analysis: deep learning for Parkinson&#x27;s disease assessment</article-title>,&#x201d; in <conf-name>Paper presented at the 2016 38th annual international conference of the IEEE Engineering in Medicine and Biology Society (EMBC)</conf-name>, <fpage>655</fpage>&#x2013;<lpage>658</lpage>. <pub-id pub-id-type="doi">10.1109/EMBC.2016.7590787</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Florensa</surname>
<given-names>A. F.</given-names>
</name>
<name>
<surname>Sommer Kaas</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Clausen</surname>
<given-names>P. T. L. C.</given-names>
</name>
<name>
<surname>Aytan-Aktug</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Aarestrup</surname>
<given-names>F. M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>ResFinder&#x2013;an open online resource for identification of antimicrobial resistance genes in next-generation sequencing data and prediction of phenotypes from genotypes</article-title>. <source>Microb. genomics</source> <volume>8</volume> (<issue>1</issue>), <fpage>000748</fpage>. <pub-id pub-id-type="doi">10.1099/mgen.0.000748</pub-id>
<pub-id pub-id-type="pmid">35072601</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Machine learning and feature extraction for rapid antimicrobial resistance prediction of Acinetobacter Baumannii from whole-genome sequencing data</article-title>. <source>Front. Microbiol.</source> <volume>14</volume>, <fpage>1320312</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2023.1320312</pub-id>
<pub-id pub-id-type="pmid">38274740</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<collab>GitHub</collab> (<year>2025</year>). <source>Deep transfer learning enables robust prediction of antimicrobial resistance for novel antibiotics</source>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://github.com/YunxiaoRen/deep_transfer_learning_AMR">https://github.com/YunxiaoRen/deep_transfer_learning_AMR</ext-link> (Accessed December 12, 2025)</comment>.</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Green</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Yoon</surname>
<given-names>C.Ho</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Ektefaie</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fina</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Freschi</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>A convolutional neural network highlights mutations relevant to antimicrobial resistance in Mycobacterium tuberculosis</article-title>. <source>Nat. Commun.</source> <volume>13</volume> (<issue>1</issue>), <fpage>3817</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-022-31236-0</pub-id>
<pub-id pub-id-type="pmid">35780211</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grey</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Upton</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Joshi</surname>
<given-names>L. T.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Urinary tract infections: a review of the current diagnostics landscape</article-title>. <source>J. Med. Microbiol.</source> <volume>72</volume> (<issue>11</issue>), <fpage>001780</fpage>. <pub-id pub-id-type="doi">10.1099/jmm.0.001780</pub-id>
<pub-id pub-id-type="pmid">37966174</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gunasekaran</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ramalakshmi</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Rex Macedo Arokiaraj</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Deepa Kanmani</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Venkatesan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dhas</surname>
<given-names>C. S. G.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Analysis of DNA sequence classification using CNN and hybrid models</article-title>. <source>Comput. Math. Methods Med.</source> <volume>2021</volume>, <fpage>1835056</fpage>. <pub-id pub-id-type="doi">10.1155/2021/1835056</pub-id>
<pub-id pub-id-type="pmid">34306171</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>V&#xe4;lim&#xe4;ki</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A risk prediction model based on machine learning for cognitive impairment among Chinese community-dwelling elderly people with normal cognition: development and validation study</article-title>. <source>J. Med. Internet Res.</source> <volume>23</volume> (<issue>2</issue>), <fpage>e20298</fpage>. <pub-id pub-id-type="doi">10.2196/20298</pub-id>
<pub-id pub-id-type="pmid">33625369</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jafri</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Qasim</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Masoud</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Izhar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kazmi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kazmi</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Antibiotic resistance of <italic>E. coli</italic> isolates from urine samples of urinary tract infection (UTI) patients in Pakistan</article-title>. <source>Bioinformation</source> <volume>10</volume> (<issue>7</issue>), <fpage>419</fpage>&#x2013;<lpage>422</lpage>. <pub-id pub-id-type="doi">10.6026/97320630010419</pub-id>
<pub-id pub-id-type="pmid">25187681</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jamthikar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Khanna</surname>
<given-names>N. N.</given-names>
</name>
<name>
<surname>Saba</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Araki</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Viskovic</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>A low-cost machine learning-based cardiovascular/stroke risk assessment system: integration of conventional factors with image phenotypes</article-title>. <source>Cardiovasc. diagnosis Ther.</source> <volume>9</volume> (<issue>5</issue>), <fpage>420</fpage>&#x2013;<lpage>430</lpage>. <pub-id pub-id-type="doi">10.21037/cdt.2019.09.03</pub-id>
<pub-id pub-id-type="pmid">31737514</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jamthikar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Khanna</surname>
<given-names>N. N.</given-names>
</name>
<name>
<surname>Saba</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Laird</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Suri</surname>
<given-names>J. S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Cardiovascular/stroke risk prevention: a new machine learning framework integrating carotid ultrasound image-based phenotypes and its harmonics with conventional risk factors</article-title>. <source>Indian Heart J.</source> <volume>72</volume> (<issue>4</issue>), <fpage>258</fpage>&#x2013;<lpage>264</lpage>. <pub-id pub-id-type="doi">10.1016/j.ihj.2020.06.004</pub-id>
<pub-id pub-id-type="pmid">32861380</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jin</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yue</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Predicting antimicrobial resistance in <italic>E. coli</italic> with discriminative position fused deep learning classifier</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>23</volume>, <fpage>559</fpage>&#x2013;<lpage>565</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2023.12.041</pub-id>
<pub-id pub-id-type="pmid">38274998</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kavvas</surname>
<given-names>E. S.</given-names>
</name>
<name>
<surname>Catoiu</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Mih</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Yurkovich</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Seif</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Dillon</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Machine learning and structural analysis of Mycobacterium tuberculosis pan-genome identifies genetic signatures of antibiotic resistance</article-title>. <source>Nat. Commun.</source> <volume>9</volume> (<issue>1</issue>), <fpage>4306</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-018-06634-y</pub-id>
<pub-id pub-id-type="pmid">30333483</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kuang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>A practical Gpu based Knn algorithm</article-title>,&#x201d; in <conf-name>Paper presented at the Proceedings. The 2009 International Symposium on Computer Science and Computational Technology (ISCSCI 2009)</conf-name>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Durbin</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Fast and accurate short read alignment with burrows&#x2013;wheeler transform</article-title>. <source>Bioinformatics</source> <volume>25</volume> (<issue>14</issue>), <fpage>1754</fpage>&#x2013;<lpage>1760</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp324</pub-id>
<pub-id pub-id-type="pmid">19451168</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sutherland</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Austin Hammond</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Taho</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Bergman</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Amplify: attentive deep learning model for discovery of novel antimicrobial peptides effective against who priority pathogens</article-title>. <source>BMC Genomics</source> <volume>23</volume> (<issue>1</issue>), <fpage>77</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-022-08310-4</pub-id>
<pub-id pub-id-type="pmid">35078402</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Rong</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Evaluation of machine learning models for predicting antimicrobial resistance of actinobacillus pleuropneumoniae from whole genome sequences</article-title>. <source>Front. Microbiol.</source> <volume>11</volume>, <fpage>474876</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2020.00048</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lueftinger</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Majek</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Beisken</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rattei</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Posch</surname>
<given-names>A. E.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Learning from limited data: towards best practice techniques for antimicrobial resistance prediction from whole genome sequencing data</article-title>. <source>Front. Cell. Infect. Microbiol.</source> <volume>11</volume>, <fpage>610348</fpage>. <pub-id pub-id-type="doi">10.3389/fcimb.2021.610348</pub-id>
<pub-id pub-id-type="pmid">33659219</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Malekian</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Agrawal</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Berendonk</surname>
<given-names>T. U.</given-names>
</name>
<name>
<surname>Al-Fatlawi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schroeder</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A genome-wide scan of wastewater <italic>E. coli</italic> for genes under positive selection: focusing on mechanisms of antibiotic resistance</article-title>. <source>Sci. Rep.</source> <volume>12</volume> (<issue>1</issue>), <fpage>8037</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-022-11432-0</pub-id>
<pub-id pub-id-type="pmid">35577863</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McArthur</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Waglechner</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Nizam</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Azad</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Baylay</surname>
<given-names>A. J.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>The comprehensive antibiotic resistance database</article-title>. <source>Antimicrob. Agents Chemother.</source> <volume>57</volume> (<issue>7</issue>), <fpage>3348</fpage>&#x2013;<lpage>3357</lpage>. <pub-id pub-id-type="doi">10.1128/AAC.00419-13</pub-id>
<pub-id pub-id-type="pmid">23650175</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="web">
<collab>Medcalc</collab> (<year>2025</year>). <article-title>Medcalc statistical software version</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.medcalc.org/">https://www.medcalc.org/</ext-link> (Accessed December 14, 2025)</comment>.</citation>
</ref>
<ref id="B33">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Mohanty</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nayak</surname>
<given-names>D. S. K.</given-names>
</name>
<name>
<surname>Swarnkar</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>A neural network framework for predicting adenocarcinoma cancer using high-throughput gene expression data</article-title>,&#x201d; in <conf-name>Paper presented at the AIP Conference Proceedings</conf-name>. <pub-id pub-id-type="doi">10.1063/5.0137033</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moradigaravand</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Palm</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Farewell</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mustonen</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Warringer</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Parts</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Prediction of antibiotic resistance in <italic>Escherichia coli</italic> from large-scale pan-genome data</article-title>. <source>PLoS Comput. Biol.</source> <volume>14</volume> (<issue>12</issue>), <fpage>e1006258</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1006258</pub-id>
<pub-id pub-id-type="pmid">30550564</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Nayak</surname>
<given-names>D. S. K.</given-names>
</name>
<name>
<surname>Das</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Swarnkar</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Quality control pipeline for next generation sequencing data analysis</article-title>,&#x201d; in <source>Intelligent and cloud computing: proceedings of Icicc 2021</source> (<publisher-name>Springer</publisher-name>), <fpage>215</fpage>&#x2013;<lpage>225</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Nayak</surname>
<given-names>D. S. K.</given-names>
</name>
<name>
<surname>Mohapatra</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Al-Dabass</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Swarnkar</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Deep learning approaches for high dimension cancer microarray data feature prediction: a review</article-title>,&#x201d; in <source>Computational intelligence in cancer diagnosis</source>, <fpage>13</fpage>&#x2013;<lpage>41</lpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nayak</surname>
<given-names>D. S. K.</given-names>
</name>
<name>
<surname>Mahapatra</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Routray</surname>
<given-names>S. P.</given-names>
</name>
<name>
<surname>Sahoo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sahoo</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Fouda</surname>
<given-names>M. M.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Aigener 1.0: an artificial intelligence technique for the revelation of informative and antibiotic resistant genes in <italic>Escherichia coli</italic>
</article-title>. <source>Front. Bioscience-Landmark</source> <volume>29</volume> (<issue>2</issue>), <fpage>82</fpage>. <pub-id pub-id-type="doi">10.31083/j.fbl2902082</pub-id>
<pub-id pub-id-type="pmid">38420832</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Niranjan</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Malini</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Antimicrobial resistance pattern in <italic>Escherichia coli</italic> causing urinary tract infection among inpatients</article-title>. <source>Indian J. Med. Res.</source> <volume>139</volume> (<issue>6</issue>), <fpage>945</fpage>&#x2013;<lpage>948</lpage>.<pub-id pub-id-type="pmid">25109731</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paul</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Maindarkar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Saxena</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Saba</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Turk</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kalra</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Bias investigation in artificial intelligence systems for early detection of parkinson&#x2019;s disease: a narrative review</article-title>. <source>Diagnostics</source> <volume>12</volume> (<issue>1</issue>), <fpage>166</fpage>. <pub-id pub-id-type="doi">10.3390/diagnostics12010166</pub-id>
<pub-id pub-id-type="pmid">35054333</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pesesky</surname>
<given-names>M. W.</given-names>
</name>
<name>
<surname>Tahir</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Dantas</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Andleeb</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Burnham</surname>
<given-names>C. A. D.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Evaluation of machine learning and rules-based approaches for predicting antimicrobial resistance profiles in gram-negative Bacilli from whole genome sequence data</article-title>. <source>Front. Microbiol.</source> <volume>7</volume>, <fpage>223089</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2016.01887</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="book">
<collab>Python</collab> (<year>2025</year>). <source>Python</source>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.python.org/">https://www.python.org/</ext-link> (Accessed January 11, 2025)</comment>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chakraborty</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Doijad</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Falgenhauer</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Falgenhauer</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Goesmann</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2022a</year>). <article-title>Multi-label classification for multi-drug resistance prediction of <italic>Escherichia coli</italic>
</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>20</volume>, <fpage>1264</fpage>&#x2013;<lpage>1270</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2022.03.007</pub-id>
<pub-id pub-id-type="pmid">35317240</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chakraborty</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Doijad</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Falgenhauer</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Falgenhauer</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Goesmann</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2022b</year>). <article-title>Deep transfer learning enables robust prediction of antimicrobial resistance for novel antibiotics</article-title>. <source>Antibiotics</source> <volume>11</volume> (<issue>11</issue>), <fpage>1611</fpage>. <pub-id pub-id-type="doi">10.3390/antibiotics11111611</pub-id>
<pub-id pub-id-type="pmid">36421255</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reza Asadi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Habibi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bouzari</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Urinary tract infection: pathogenicity, antibiotic resistance and development of effective vaccines against uropathogenic <italic>Escherichia coli</italic>
</article-title>. <source>Mol. Immunol.</source> <volume>108</volume>, <fpage>56</fpage>&#x2013;<lpage>67</lpage>. <pub-id pub-id-type="doi">10.1016/j.molimm.2019.02.007</pub-id>
<pub-id pub-id-type="pmid">30784763</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Satam</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Joshi</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Mangrolia</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Waghoo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zaidi</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Rawool</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Next-generation sequencing technology: current trends and advancements</article-title>. <source>Biology</source> <volume>12</volume> (<issue>7</issue>), <fpage>997</fpage>. <pub-id pub-id-type="doi">10.3390/biology12070997</pub-id>
<pub-id pub-id-type="pmid">37508427</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sharma</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chauhan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ranjan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mathkor</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Haque</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ramniwas</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Emerging challenges in antimicrobial resistance: implications for pathogenic microorganisms, novel antibiotics, and their impact on sustainability</article-title>. <source>Front. Microbiol.</source> <volume>15</volume>, <fpage>1403168</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2024.1403168</pub-id>
<pub-id pub-id-type="pmid">38741745</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Links</surname>
<given-names>M. G.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Dillon</surname>
<given-names>J.-A. R.</given-names>
</name>
<name>
<surname>Horsch</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Antimicrobial resistance genetic factor identification from whole-genome sequence data using deep feature selection</article-title>. <source>BMC Bioinforma.</source> <volume>20</volume>, <fpage>535</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-019-3054-4</pub-id>
<pub-id pub-id-type="pmid">31874612</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Skandha</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Saba</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Koppula</surname>
<given-names>V. K.</given-names>
</name>
<name>
<surname>Johri</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Khanna</surname>
<given-names>N. N.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>3-D optimized classification and characterization artificial intelligence paradigm for cardiovascular/stroke risk stratification using carotid ultrasound-based delineated plaque: atheromatic&#x2122; 2.0</article-title>. <source>Comput. Biol. Med.</source> <volume>125</volume>, <fpage>103958</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2020.103958</pub-id>
<pub-id pub-id-type="pmid">32927257</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stoesser</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Batty</surname>
<given-names>E. M.</given-names>
</name>
<name>
<surname>Eyre</surname>
<given-names>D. W.</given-names>
</name>
<name>
<surname>Morgan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wyllie</surname>
<given-names>D. H.</given-names>
</name>
<name>
<surname>Elias</surname>
<given-names>C. D. O.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Predicting antimicrobial susceptibilities for <italic>Escherichia coli</italic> and Klebsiella Pneumoniae isolates using whole genomic sequence data</article-title>. <source>J. Antimicrob. Chemother.</source> <volume>68</volume> (<issue>10</issue>), <fpage>2234</fpage>&#x2013;<lpage>2244</lpage>. <pub-id pub-id-type="doi">10.1093/jac/dkt180</pub-id>
<pub-id pub-id-type="pmid">23722448</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Swain</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Nayak</surname>
<given-names>D. S. K.</given-names>
</name>
<name>
<surname>Swarnkar</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>A comparative analysis of machine learning models for Colon cancer classification</article-title>,&#x201d; in <conf-name>Paper presented at the 2023 International Conference in Advances in Power, Signal, and Information Technology (APSIT)</conf-name>, <fpage>1</fpage>&#x2013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1109/apsit58554.2023.10201691</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tan</surname>
<given-names>C. W.</given-names>
</name>
<name>
<surname>Chlebicki</surname>
<given-names>M. P.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Urinary tract infections in adults</article-title>. <source>Singap. Med. J.</source> <volume>57</volume> (<issue>9</issue>), <fpage>485</fpage>&#x2013;<lpage>490</lpage>. <pub-id pub-id-type="doi">10.11622/smedj.2016153</pub-id>
<pub-id pub-id-type="pmid">27662890</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Taylor</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Moore</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Cheung</surname>
<given-names>K.-H.</given-names>
</name>
<name>
<surname>Brandt</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Predicting urinary tract infections in the emergency department with machine learning</article-title>. <source>PloS one</source> <volume>13</volume> (<issue>3</issue>), <fpage>e0194085</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0194085</pub-id>
<pub-id pub-id-type="pmid">29513742</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Totsika</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Moriel</surname>
<given-names>D. G.</given-names>
</name>
<name>
<surname>Idris</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rogers</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Wurpel</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Phan</surname>
<given-names>M.-D.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Uropathogenic <italic>Escherichia coli</italic> mediated urinary tract infection</article-title>. <source>Curr. drug targets</source> <volume>13</volume> (<issue>11</issue>), <fpage>1386</fpage>&#x2013;<lpage>1399</lpage>. <pub-id pub-id-type="doi">10.2174/138945012803530206</pub-id>
<pub-id pub-id-type="pmid">22664092</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Truswell</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>Z. Z.</given-names>
</name>
<name>
<surname>Stegger</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Blinco</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Abraham</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Jordan</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Augmented surveillance of antimicrobial resistance with high-throughput robotics detects transnational flow of fluoroquinolone-resistant <italic>Escherichia coli</italic> strain into poultry</article-title>. <source>J. Antimicrob. Chemother.</source> <volume>78</volume> (<issue>12</issue>), <fpage>2878</fpage>&#x2013;<lpage>2885</lpage>. <pub-id pub-id-type="doi">10.1093/jac/dkad323</pub-id>
<pub-id pub-id-type="pmid">37864344</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vasudevan</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Urinary tract infection: an overview of the infection and the associated risk factors</article-title>. <source>J. Microbiol. Exp.</source> <volume>1</volume> (<issue>2</issue>), <fpage>00008</fpage>. <pub-id pub-id-type="doi">10.15406/jmen.2014.01.00008</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vermeiren</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Van Craenenbroeck</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Alen</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bacheler</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Picchio</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Lecocq</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>Prediction of HIV-1 drug susceptibility phenotype from the viral genotype using linear regression modeling</article-title>. <source>J. virological methods</source> <volume>145</volume> (<issue>1</issue>), <fpage>47</fpage>&#x2013;<lpage>55</lpage>. <pub-id pub-id-type="doi">10.1016/j.jviromet.2007.05.009</pub-id>
<pub-id pub-id-type="pmid">17574687</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wilson</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Garud</surname>
<given-names>N. R.</given-names>
</name>
<name>
<surname>Feder</surname>
<given-names>A. F.</given-names>
</name>
<name>
<surname>Assaf</surname>
<given-names>Z. J.</given-names>
</name>
<name>
<surname>Pennings</surname>
<given-names>P. S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>The population genetics of drug resistance evolution in natural populations of viral, bacterial and eukaryotic pathogens</article-title>. <source>Mol. Ecol.</source> <volume>25</volume> (<issue>1</issue>), <fpage>42</fpage>&#x2013;<lpage>66</lpage>. <pub-id pub-id-type="doi">10.1111/mec.13474</pub-id>
<pub-id pub-id-type="pmid">26578204</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zankari</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Hasman</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Cosentino</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Vestergaard</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rasmussen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lund</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Identification of acquired antimicrobial resistance genes</article-title>. <source>J. Antimicrob. Chemother.</source> <volume>67</volume> (<issue>11</issue>), <fpage>2640</fpage>&#x2013;<lpage>2644</lpage>. <pub-id pub-id-type="doi">10.1093/jac/dks261</pub-id>
<pub-id pub-id-type="pmid">22782487</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>The diversified ensemble neural network</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>33</volume>, <fpage>16001</fpage>&#x2013;<lpage>16011</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>