<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Microbiol.</journal-id>
<journal-title>Frontiers in Microbiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Microbiol.</abbrev-journal-title>
<issn pub-type="epub">1664-302X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmicb.2025.1634705</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Microbiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>MoEPH: an adaptive fusion-based LLM for predicting phage-host interactions in health informatics</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Chen</surname> <given-names>Qian</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1995583/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhao</surname> <given-names>Zihang</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2887677/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Min</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Song</surname> <given-names>Wenchen</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Xiao</surname> <given-names>Minfeng</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<xref ref-type="author-notes" rid="fn003"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1758846/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Fang</surname> <given-names>Min</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>School of Artificial Intelligence and National Engineering Laboratory for Big Data System Computing Technology, Shenzhen University, Shenzhen</institution>, <addr-line>Guangdong</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>School of Computer Science and Technology, The University of Hong Kong, Hong Kong</institution>, <addr-line>Hong Kong SAR</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>BGI-Shenzhen</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>University of Chinese Academy of Sciences</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>Education Center of Experiments and Innovations, Harbin Institute of Technology, Shenzhen</institution>, <addr-line>Guangdong</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Gary Antonio Toranzos, University of Puerto Rico, Puerto Rico</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Amjed Alsultan, University of Al-Qadisiyah, Iraq</p>
<p>Yanmei Sun, Northwest University, China</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Minfeng Xiao <email>xiaominfeng&#x00040;genomics.cn</email></corresp>
<corresp id="c002">Min Fang <email>fangmin&#x00040;hit.edu.cn</email></corresp>
<fn fn-type="other" id="fn002"><p>&#x02020;ORCID: Qian Chen <ext-link ext-link-type="uri" xlink:href="https://orcid.org/0000-0002-2341-2118">orcid.org/0000-0002-2341-2118</ext-link></p></fn>
<fn fn-type="other" id="fn003"><p>Minfeng Xiao <ext-link ext-link-type="uri" xlink:href="https://orcid.org/0000-0002-0507-7352">orcid.org/0000-0002-0507-7352</ext-link></p></fn></author-notes>
<pub-date pub-type="epub">
<day>18</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1634705</elocation-id>
<history>
<date date-type="received">
<day>25</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>28</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Chen, Zhao, Li, Song, Xiao and Fang.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Chen, Zhao, Li, Song, Xiao and Fang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Phage-host interaction prediction plays a crucial role in the development of phage therapy, particularly in combating antimicrobial resistance (AMR). Current in silico models often suffer from limited generalizability and low interpretability. To address these gaps, we introduce MoEPH (Mixture-of-Experts for Phage-Host prediction), a novel framework that integrates transformer-based protein embeddings (ProtBERT and ProT5) with domain-specific statistical descriptors. Our model dynamically combines features using a gated fusion mechanism, ensuring robust and adaptive prediction. We evaluate MoEPH on three publicly available phage-host interaction databases: Dataset 1 (101 host strains, 129 phages), Dataset 2 (38 host strains, 176 phages), and Dataset 3 (combined). Experimental results demonstrate that MoEPH outperforms existing methods, achieving an accuracy of 99.6% on balanced datasets and a 31% improvement on highly imbalanced data. The model provides a transparent, dynamic and knowledge-driven fusion solution for phage-host prediction, contributing to more effective phage therapy recommendations. Future work will focus on incorporating structural protein features and exploring alternative neural backbones for further enhancement.</p></abstract>
<kwd-group>
<kwd>trustworthy phage-host prediction</kwd>
<kwd>interpretable</kwd>
<kwd>robustness</kwd>
<kwd>transformer-based protein embeddings</kwd>
<kwd>mixture of experts (MoE)</kwd>
</kwd-group>
<counts>
<fig-count count="8"/>
<table-count count="2"/>
<equation-count count="44"/>
<ref-count count="22"/>
<page-count count="18"/>
<word-count count="8786"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Phage Biology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Antimicrobial resistance (AMR) has emerged as a global health crisis and a &#x0201C;silent pandemic,&#x0201D; threatening effective infection treatment worldwide. Without intervention, AMR is projected to cause up to 10 million deaths annually by 2050 (<xref ref-type="bibr" rid="B21">Walsh et al., 2023</xref>), with millions of lives already impacted each year. This dire situation has reignited interest in alternative therapeutics (<xref ref-type="bibr" rid="B15">Murray et al., 2022</xref>). Bacteriophage (phage) therapy&#x02014;the use of viruses that specifically infect bacteria&#x02014;has re-emerged as a promising strategy to combat drug-resistant infections. Phages can lyse antibiotic-resistant bacteria with high specificity, offering a potential lifeline where antibiotics fail. In this context, developing accurate phage&#x02014;host identification methods is crucial to actualize phage therapy against AMR. Our work lies at the intersection of phage therapy and artificial intelligence, aiming to advance trustworthy medical AI to address this challenge. Moreover, increasing biological data resources facilitate this goal; for instance, we leverage a large phage&#x02013;host interaction dataset from BGI-Shenzhen,<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref> which provides extensive phage genomic and host information to support model training and evaluation.</p>
<p>Existing <italic>in silico</italic> phage&#x02013;host prediction methods, however, face significant limitations that hinder clinical utility. Early approaches rely on genomic sequence similarity or alignment-based heuristics, such as BLAST hits or CRISPR-spacer matches. Alignment-free statistical methods [e.g., WIsH (<xref ref-type="bibr" rid="B8">Galiez et al., 2017</xref>) and RaFAH (<xref ref-type="bibr" rid="B4">Coutinho et al., 2021</xref>)] predict hosts based on k-mer composition or protein content, but often fail on distantly related phages and have limited accuracy. Later methods introduced machine learning on genomic features: for example, logistic matrix factorization models (<xref ref-type="bibr" rid="B11">Leite et al., 2018</xref>) and network-based frameworks like VirHostMatcher-Net (<xref ref-type="bibr" rid="B22">Wang et al., 2020</xref>) integrated multiple genomic similarity measures to improve predictions. These offered moderate performance gains yet still struggle with generalizability and robustness, especially for novel phages or under-represented hosts. Deep learning has also been applied&#x02014;notably PredPHI (<xref ref-type="bibr" rid="B12">Li et al., 2020</xref>), a CNN-based tool utilizing phage and host protein features&#x02014;achieving higher accuracy than classical approaches. However, such deep models act as black boxes with limited interpretability, and their improvements in accuracy remain modest. Even knowledge-integrated approaches encounter challenges: a recent knowledge graph model KGVHI (<xref ref-type="bibr" rid="B16">Pan et al., 2024</xref>) combines multiple data types (genomic, proteomic, and host metadata) to predict microbe&#x02013;host pairs and shows excellent performance on benchmark datasets, yet it requires comprehensive prior knowledge and is not specialized for bacteriophage therapy scenarios. In summary, existing methods tend to overfit to training data and imbalanced distributions, often over-predicting dominant hosts while missing rare interactions. They lack the adaptability to generalize to new phage or host species and provide little biological insight into predictions. Such opacity and instability undermine user trust, which is critical for AI deployment in healthcare. Thus, improvements in generalization, robustness, and explainability are essential before phage&#x02013;host prediction models can be used confidently in clinical practice.</p>
<p>Meanwhile, large pre-trained models have brought transformative advances to bioinformatics. In the protein biology domain, protein language models (pLMs) like ProtBERT and ProT5 (<xref ref-type="bibr" rid="B6">Elnaggar et al., 2021</xref>) leverage Transformer architectures to learn rich protein sequence representations. These models capture structural motifs and achieve state-of-the-art performance in diverse biological tasks. Treating protein sequences as a &#x0201C;language of life,&#x0201D; such models offer superior feature learning for tasks like phage&#x02013;host prediction. Similarly, in biomedical NLP, large domain-specific language models [e.g., PubMedBERT (<xref ref-type="bibr" rid="B9">Gu et al., 2021</xref>)] have demonstrated that pre-training on in-domain data yields powerful representations for downstream tasks. However, current pLM-based phage&#x02013;host prediction approaches are often used only as static feature extractors, lacking dynamic adaptation or integration of domain knowledge. Recent studies have begun to endow language models with biological knowledge or multi-modal data (<xref ref-type="bibr" rid="B3">Chen et al., 2024</xref>), but a cohesive framework that combines pre-trained embeddings with adaptive, knowledge-driven fusion for phage&#x02013;host prediction remains absent.</p>
<p>To address these gaps, we propose MoEPH, a Mixture-of-Experts framework designed to improve both performance and transparency for trustworthy phage&#x02013;host prediction. MoEPH employs multiple expert subnetworks specializing in different feature modalities, with a gating network dynamically selecting experts for each input (<xref ref-type="bibr" rid="B19">Shazeer et al., 2017</xref>). This architecture enables adaptive, context-specific learning, effectively capturing both genomic composition signals and high-level protein patterns. By synergistically integrating pre-trained protein embeddings with interpretable statistical features, MoEPH achieves not only superior predictive performance but also enhanced explainability. Specifically, our model maintains robust accuracy even on highly imbalanced and novel data, mitigating the overfitting to dominant hosts that plagues prior methods. It also provides interpretability through per-sample expert weight analysis, offering biological insight into which features drive a given prediction. These qualities align with key pillars of trustworthy AI&#x02014;reliability, explainability, and adaptability (<xref ref-type="bibr" rid="B2">Aljohani et al., 2025</xref>)&#x02014;making MoEPH particularly suitable for sensitive applications like phage therapy recommendation. In summary, our main contributions are an innovative multi-expert fusion strategy tailored for phage&#x02013;host prediction, a comprehensive evaluation demonstrating state-of-the-art performance across multiple datasets, and an analysis showing improved generalization and interpretability compared to existing models.</p>
<p>In summary, our main contributions are:</p>
<list list-type="bullet">
<list-item><p><bold>Trusted phage-host prediction framework:</bold> We develop MoEPH, combining pre-trained embeddings with dynamic expert selection to enhance robustness and interpretability in clinical AI applications targeting AMR.</p></list-item>
<list-item><p><bold>Enhanced robustness on imbalanced data:</bold> MoEPH achieves stable performance even under highly imbalanced conditions, avoiding overfitting and reliably predicting under-represented bacteria.</p></list-item>
<list-item><p><bold>Interpretability via expert weight visualization:</bold> The gating network output offers interpretable insights into model decisions, supporting clinician trust and biological discovery.</p></list-item>
<list-item><p><bold>Generalization across datasets and novel pairs:</bold> Extensive evaluation shows that MoEPH generalizes well to external datasets and unseen phage-host pairs, maintaining reliability as scientific knowledge evolves.</p></list-item>
</list></sec>
<sec id="s2">
<title>2 Definition and materials</title>
<p>In pursuit of trustworthy AI, this chapter details the feature design and data preparation that form the foundation for MoEPH, with a particular emphasis on explainability and robustness. Specifically, we integrate knowledge-driven statistical descriptors with context-rich transformer-based protein embeddings to balance interpretability and predictive performance. Fundamental features such as amino acid composition (AAC), atomic composition (AC), and molecular weight (MW) provide human-understandable sequence characteristics that complement the deep sequence representations from pretrained models (e.g., ProtBERT, ProT5). By unifying these two classes of features within a Mixture-of-Experts (MoE) framework, our model can more effectively capture protein complexity in an adaptive manner. The synergy between interpretable statistical features and advanced LLM-derived embeddings thus serves as the basis for a more explainable and resilient predictive system. In this section, we describe the designed statistical features and the datasets utilized, which together underpin the trustworthy modeling approach of MoEPH.</p>
<sec>
<title>2.1 Definition of statistical features</title>
<p>We incorporate three interpretable, domain-knowledge-driven statistical descriptors to complement deep embeddings:</p>
<p>(1) <bold>Amino acid composition (AAC):</bold> Measures the frequency of each amino acid <italic>A</italic><sub><italic>i</italic></sub> in sequence <italic>S</italic>:</p>
<disp-formula id="E1"><mml:math id="M1"><mml:mrow><mml:mtext>AA</mml:mtext><mml:msub><mml:mrow><mml:mtext>C</mml:mtext></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:mi>M</mml:mi><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>where <italic>n</italic><sub><italic>i</italic></sub> is the count of <italic>A</italic><sub><italic>i</italic></sub> and <italic>L</italic> is the sequence length. AAC provides coarse but robust information for classification tasks.</p>
<p>(2) <bold>Atomic composition (AC):</bold> Represents the proportion of each element <italic>E</italic><sub><italic>j</italic></sub> (e.g., C, H, N, O, S) in a protein:</p>
<disp-formula id="E2"><mml:math id="M2"><mml:mrow><mml:mtext>A</mml:mtext><mml:msub><mml:mrow><mml:mtext>C</mml:mtext></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msubsup></mml:mstyle><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:mi>E</mml:mi><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>where <italic>a</italic><sub><italic>j</italic></sub> is the number of atoms of <italic>E</italic><sub><italic>j</italic></sub> in <italic>S</italic>. AC captures the protein&#x00027;s fundamental chemical makeup.</p>
<p>(3) <bold>Molecular weight (MW):</bold> Summarizes compositional information into a physicochemical metric:</p>
<disp-formula id="E3"><mml:math id="M3"><mml:mrow><mml:mtext>MW</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>where <italic>M</italic>(<italic>A</italic><sub><italic>i</italic></sub>) denotes the molecular weight of amino acid <italic>A</italic><sub><italic>i</italic></sub>. MW can discriminate proteins based on mass profiles.</p>
<p>These statistical features enhance model explainability by providing intuitive biochemical insights into protein sequences.</p></sec>
<sec>
<title>2.2 Transformer-based LLMs protein representations</title>
<p>We leverage ProtBERT and ProT5, two pre-trained protein LLMs, to generate context-rich sequence embeddings.</p>
<p>(1) ProtBERT: ProtBERT is a BERT-based protein language model that employs a bidirectional Transformer encoder with a masked language modeling (MLM) objective [20]. During pre-training, a subset <italic>M</italic> &#x0003D; {<italic>m</italic><sub>1</sub>, &#x02026;, <italic>m</italic><sub><italic>K</italic></sub>} of positions in the input sequence <italic>S</italic> is randomly masked, and the model learns to predict the masked residues using context. The MLM loss is defined as:</p>
<disp-formula id="E4"><label>(1)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">MLM</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>M</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mo class="qopname">log</mml:mo><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02223;</mml:mo><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mo>\</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>S</italic><sub>\<italic>m</italic></sub> denotes <italic>S</italic> with the residue at position <italic>m</italic> replaced by a mask token. ProtBERT stacks multiple Transformer layers to iteratively refine the sequence representation:</p>
<disp-formula id="E5"><label>(2)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">TransformerLayer</mml:mtext><mml:msup><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:mi>L</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>with <italic>H</italic><sup>(0)</sup> being the input token embeddings. By leveraging contextual cues, ProtBERT learns both local motifs and long-range dependencies, yielding robust sequence features for downstream tasks.</p>
<p>(2) ProT5: ProT5 is a T5-based encoder&#x02013;decoder model that frames protein modeling tasks in a sequence-to-sequence format. It can be pre-trained to reconstruct corrupted sequences, predict functional or structural annotations, or even generate novel protein sequences. Given an input sequence <italic>S</italic> and a target sequence <italic>Y</italic> &#x0003D; {<italic>y</italic><sub>1</sub>, &#x02026;, <italic>y</italic><sub><italic>T</italic></sub>}, ProT5 models the conditional probability:</p>
<disp-formula id="E6"><label>(3)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>Y</mml:mi><mml:mo>&#x02223;</mml:mo><mml:mi>S</mml:mi><mml:mo>;</mml:mo><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x0220F;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02223;</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0003C;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>S</mml:mi><mml:mo>;</mml:mo><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>It learns by minimizing the negative log-likelihood:</p>
<disp-formula id="E7"><label>(4)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">T5</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mo class="qopname">log</mml:mo><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02223;</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0003C;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>S</mml:mi><mml:mo>;</mml:mo><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The encoder first produces a hidden representation:</p>
<disp-formula id="E8"><label>(5)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">enc</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">Encoder</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The decoder then uses <italic>H</italic><sub>enc</sub> (together with previously generated tokens <italic>y</italic><sub>&#x0003C;<italic>t</italic></sub>) to predict the next token <italic>y</italic><sub><italic>t</italic></sub>:</p>
<disp-formula id="E9"><label>(6)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">dec</mml:mtext><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">Decoder</mml:mtext><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">enc</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0003C;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:mi>T</mml:mi><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Through this process, ProT5 captures rich long-range dependencies and excels in generative and multi-task settings. Training on diverse objectives endows ProT5 with a broad latent space of protein sequences, complementing ProtBERT&#x00027;s embeddings.</p></sec>
<sec>
<title>2.3 Dataset of the immersion experiment</title>
<p>The datasets originate from a BGI research project titled <italic>Research on Artificial Phage Model Construction Based on Deep Generative Adversarial Network Learning</italic> (see text footnote <xref ref-type="fn" rid="fn0001"><sup>1</sup></xref>), providing a reliable empirical foundation for our prediction tasks. This study is part of an ongoing initiative at BGI-Shenzhen to explore phage-host prediction under real-world data constraints. The equipment is based on the NVIDIA A100 cloud platform. To evaluate the model under multi-source conditions, we utilize two distinct datasets and a third integrated dataset. To evaluate the model under multi-source conditions, we utilize two distinct datasets and a third integrated dataset.</p>
<p>To provide a clearer picture of the dataset composition, we include a heatmap of the host-phage interaction matrix (<xref ref-type="fig" rid="F1">Figure 1</xref>) in the data description. This figure offers an overview of which phages infect which host strains in Dataset 1, highlighting the data&#x00027;s structure before any modeling.</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>Host-phage matching matrix.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-16-1634705-g0001.tif">
<alt-text>Heatmap titled &#x0201C;Host-Phage Matching Matrix Heatmap&#x0201D; shows a grid with intensity values ranging from zero to eight. Colors range from light yellow to dark blue, indicating varying levels of matches between hosts and phages. The y-axis labels from zero to one hundred and x-axis labels from zero to one hundred twenty-eight denote different host and phage combinations.</alt-text>
</graphic>
</fig>
<p><bold>Dataset 1:</bold> This dataset includes 101 host bacterial strains and 129 phages, collected under well-defined conditions, ensuring clear information on species composition and environmental context.</p>
<p><bold>Dataset 2:</bold> This dataset comprises 38 host strains and 176 phages. Compared to Dataset 1, its sampling conditions (e.g., environment and strain selection) differ, yielding distinct phage&#x02013;host interaction patterns.</p>
<p><bold>Dataset 3:</bold> Dataset 3 combines Dataset 1 and Dataset 2 to enable performance assessment under more diverse conditions, providing a robust test of the model&#x00027;s adaptability to heterogeneous data sources in real-world settings.</p>
<p>(1) <italic>Immersion experiment and labeling strategy:</italic> The phage&#x02013;host interactions were measured via immersion experiments, where phages were exposed to host cultures and infection outcomes recorded in a host&#x02013;phage matrix (<xref ref-type="fig" rid="F1">Figure 1</xref>). We binarized these outcomes by labeling interactions with infection values above 1.5 as 1 (significant infection) and below 1.5 as 0 (no significant infection), ensuring a clear separation between positive and negative samples.</p>
<p>(2) <italic>Sequence information and feature extraction:</italic> For each host and phage, we obtained the protein amino acid sequence and extracted features including traditional descriptors (AAC, AC, MW) and LLM-based embeddings (ProtBERT, ProT5), yielding a rich feature set for learning.</p>
<p>In summary, these curated multi-source datasets and comprehensive feature sets provide a solid foundation to assess the model&#x00027;s generalization and robustness. Their diversity underscores the importance of MoEPH&#x00027;s dynamic fusion mechanism for reliable predictions across heterogeneous conditions.</p>
</sec></sec>
<sec id="s3">
<title>3 Proposed framework: MoEPH</title>
<p>In this section, we present the architecture of the proposed <bold>M</bold>ixture-<bold>o</bold>f-<bold>E</bold>xperts model for <bold>P</bold>hage <bold>H</bold>ost prediction (MoEPH). The MoEPH framework is designed to leverage transformer-based protein representations and an ensemble of expert sub-models to address the complex task of phage-host prediction. Mixture-of-Experts (MoE) architectures have also been used in recent large language models to achieve greater scalability by dividing the model&#x00027;s knowledge among specialized sub-networks; we adopt a similar principle here to effectively handle the heterogeneity of phage-host data. By decomposing the prediction task among multiple expert networks and using an adaptive gating mechanism to fuse their outputs, MoEPH can capture diverse patterns in the data. This design enhances the model&#x00027;s robustness and flexibility, as each expert can specialize in certain features or sub-distributions of the input, while the gating network dynamically selects and combines expert contributions appropriate for each phage query. In what follows, we detail the overall model framework, the structure of the Mixture-of-Experts layer, and the training and inference procedures.</p>
<sec>
<title>3.1 MoEPH model framework</title>
<p>The MoEPH model proposed in this study is designed to integrate multi-source features both statistical descriptors and deep sequence embeddings extracted by large language models (LLMs) and to adaptively weight these features through a Mixture-of-Experts (MoE) mechanism. By doing so, the model produces more robust and expressive representations for subsequent classification tasks. <xref ref-type="fig" rid="F2">Figure 2</xref> provides an overview of the framework.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>Flowchart of MoEPH. This figure illustrates the model&#x00027;s main components, including statistical feature extraction (A1), two Transformer-based LLM feature extraction modules (A2, A3), the MoE layer (B) for adaptive expert weighting, feature concatenation (C), and the final prediction model (D).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-16-1634705-g0002.tif">
<alt-text>Diagram outlining a machine learning model architecture involving multiple components. At the bottom, input data of phage and host undergo preprocessing. A1 shows statistical features extraction with normalization. A2 and A3 illustrate feature embeddings using ProtBERT and ProtT5 models. These features are fed into a mixture of experts (MoE) layer, labeled B, involving two experts and a gating network. Features are concatenated at C with a total of 1,050 features. The final step, D, involves processing by a model like CNN or MLP to generate the output.</alt-text>
</graphic>
</fig>
<p>In <bold>A1 (Statistical feature extraction)</bold>, we derive fundamental statistical descriptors such as amino acid composition (AAC), atomic composition (AC), and molecular weight (MW) from each protein sequence, denoted as <inline-formula><mml:math id="M10"><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>X</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">stats</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mn>26</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> in the example illustration. These features reflect basic physicochemical properties of the protein and serve as an initial numeric reference for subsequent integration.</p>
<p>Concurrently, in <bold>A2</bold> and <bold>A3 (LLM feature extraction)</bold>, we employ two pre-trained protein language models: ProtBERT and ProT5 to capture deep semantic representations of the sequences, resulting in <inline-formula><mml:math id="M11"><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>X</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">ProtBERT</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mn>1024</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M12"><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>X</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">ProT5</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mn>1024</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>. These high-dimensional embeddings encode contextual, local structural, and functional information within the sequences, thereby enriching the predictive power for phage-host interaction tasks.</p>
<p>Next, these three kinds of features are fed into <bold>B: the MoE layer</bold>, which consists of a <bold>Router</bold> and two expert modules (Expert 1 and Expert 2). A gating network processes the statistical features <bold>X</bold><sub>stats</sub> to generate weights &#x003B1;, dictating how <bold>X</bold><sub>ProtBERT</sub> and <bold>X</bold><sub>ProT5</sub> should be combined. This adaptive weighting ensures that the model can dynamically select the most discriminative or best-matched expert features for each sample under varying data distributions and feature patterns. In addition, the resulting weighting coefficients can provide interpretability regarding the relative importance of each LLM for different types of samples.</p>
<p>As depicted in <bold>C (Concatenate of features)</bold>, the fused features obtained from the MoE layer are concatenated with the statistical features, producing <inline-formula><mml:math id="M13"><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>X</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">combined</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mn>1050</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> (illustrative dimensionality). This consolidated vector representation fully integrates both the &#x0201C;protein statistical attributes&#x0201D; and the &#x0201C;LLM-based deep features.&#x0201D;</p>
<p>Finally, <bold>D (Model)</bold> processes <bold>X</bold><sub>combined</sub> using a customizable prediction network (e.g., a CNN, MLP, or RNN) to output the phage-host interaction results. In this study, we employ a CNN classifier to systematically examine how the fused features improve performance (e.g., accuracy, F1-score, AUC) under a fixed network structure.</p>
<p>In summary, MoEPH incorporates statistical features from traditional analyses and advanced embeddings from ProtBERT/ProT5, then uses an adaptive MoE layer to effectively and interpretably combine multi-source information. This integrated solution provides a robust yet flexible approach to phage-host interaction prediction.</p></sec>
<sec>
<title>3.2 MoE layer of the MoEPH model</title>
<p>In protein-related prediction tasks, relying solely on a single pre-trained model (e.g., ProtBERT or ProT5) often fails to fully capture the diverse sequence patterns and structural information inherent in biological data. To address this limitation, we introduce a Mixture-of-Experts (MoE) mechanism into our model. By combining multiple pre-trained experts and adaptively assigning their importance based on sample-specific statistical attributes, the MoE layer flexibly merges various feature advantages, thereby enhancing predictive performance and overall generalization (<xref ref-type="bibr" rid="B19">Shazeer et al., 2017</xref>; <xref ref-type="bibr" rid="B17">Pearce and Zhang, 2021</xref>). The entire fusion procedure is summarized in <xref ref-type="table" rid="T3">Algorithm 1</xref>.</p>
<table-wrap position="float" id="T3">
<label>Algorithm 1</label>
<caption><p>Mixture-of-Experts (MoE) layer feature fusion in MoEPH.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-16-1634705-i0001.tif"/>
</table-wrap>
</sec>
<sec>
<title>3.3 MoE layer structure and gating mechanism</title>
<p>In our Mixture-of-Experts (MoE) layer, a <italic>gating network</italic> dynamically computes sample-specific weights to fuse the outputs of two expert models (ProtBERT and ProT5). Formally, for each sample <italic>i</italic>, let <inline-formula><mml:math id="M14"><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">ProtBERT</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> and <inline-formula><mml:math id="M15"><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">ProT5</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> denote the pre-trained feature vectors from ProtBERT and ProT5 (with <italic>d</italic> &#x0003D; 1024 in our implementation). Each sample also has an associated statistical feature vector <inline-formula><mml:math id="M16"><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">stats</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> (with <italic>s</italic> &#x0003D; 26 descriptors such as physicochemical properties and sequence composition). The gating function <italic>g</italic>(&#x000B7;) is realized as a two-layer feed-forward network that transforms the statistical features into a pair of importance weights (&#x003B1;<sub><italic>i</italic>, 1</sub>, &#x003B1;<sub><italic>i</italic>, 2</sub>) for the two experts. Concretely, the gating network first applies a fully-connected layer to <italic>X</italic><sub>stats, <italic>i</italic></sub> (shared across samples) to produce a hidden representation, then a second fully-connected layer produces two logit scores which are normalized by a softmax activation:</p>
<disp-formula id="E10"><label>(7)</label><mml:math id="M17"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>H</mml:mi></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:mtext>ReLU</mml:mtext><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">stats</mml:mtext></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x02003;&#x000A0;</mml:mtext><mml:mi>H</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mn>64</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E11"><label>(8)</label><mml:math id="M18"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>G</mml:mi></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:mi>H</mml:mi><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x02003;&#x000A0;</mml:mtext><mml:mi>G</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E12"><label>(9)</label><mml:math id="M19"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>&#x003B1;</mml:mi></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:mtext>softmax</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x02003;&#x000A0;</mml:mtext><mml:mi>&#x003B1;</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M20"><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mn>64</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mn>64</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mn>64</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> are trainable parameters. The ReLU activation in <xref ref-type="disp-formula" rid="E10">Equation 7</xref> introduces nonlinearity into the gating function, and the softmax in <xref ref-type="disp-formula" rid="E12">Equation 9</xref> ensures each sample&#x00027;s two gating coefficients (&#x003B1;<sub><italic>i</italic>, 1</sub>, &#x003B1;<sub><italic>i</italic>, 2</sub>) form a probability distribution (i.e., &#x003B1;<sub><italic>i</italic>, 1</sub>, &#x003B1;<sub><italic>i</italic>, 2</sub>&#x02265;0 and &#x003B1;<sub><italic>i</italic>, 1</sub>&#x0002B;&#x003B1;<sub><italic>i</italic>, 2</sub> &#x0003D; 1 for each <italic>i</italic>). Importantly, this gating is <italic>dynamic and sample-specific</italic>: for each sample <italic>i</italic>, the statistical attribute vector <italic>X</italic><sub>stats, <italic>i</italic></sub> yields its own gating weights &#x003B1;<sub><italic>i</italic>, 1</sub> and &#x003B1;<sub><italic>i</italic>, 2</sub>. This design allows the model to adaptively decide how much to rely on each expert&#x00027;s features based on the characteristics of that sample (rather than using fixed static fusion weights).</p>
<p>Using the gating weights &#x003B1;<sub><italic>i</italic>, 1</sub> and &#x003B1;<sub><italic>i</italic>, 2</sub>, the MoE layer modulates and fuses the expert outputs for each sample. Let &#x003B1;<sub><italic>i</italic></sub> &#x0003D; [&#x003B1;<sub><italic>i</italic>, 1</sub>, &#x003B1;<sub><italic>i</italic>, 2</sub>] be the weight vector for sample <italic>i</italic>. We obtain the <italic>fused</italic> feature for sample <italic>i</italic> by an element-wise weighted sum of the two expert feature vectors:</p>
<disp-formula id="E13"><label>(10)</label><mml:math id="M21"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">fused</mml:mtext><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">ProtBERT</mml:mtext><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">ProT5</mml:mtext><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x02003;&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">fused</mml:mtext><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In other words, the ProtBERT embedding is scaled by &#x003B1;<sub><italic>i</italic>, 1</sub> and the ProT5 embedding by &#x003B1;<sub><italic>i</italic>, 2</sub>, and then they are added together to produce a single fused representation for sample <italic>i</italic>. Stacking these results for all <italic>N</italic> samples yields the fused feature matrix:</p>
<disp-formula id="E14"><label>(11)</label><mml:math id="M22"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mtext>fused</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext>&#x02009;</mml:mtext><mml:msub><mml:mi>&#x003B1;</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>&#x02299;</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mtext>ProtBERT</mml:mtext></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>p</mml:mi><mml:mi>h</mml:mi><mml:msub><mml:mi>a</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>&#x02299;</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mtext>ProT5</mml:mtext></mml:mrow></mml:msub><mml:mtext>&#x02009;</mml:mtext><mml:mo>&#x02208;</mml:mo><mml:mtext>&#x02009;</mml:mtext><mml:msup><mml:mi>&#x0211D;</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msup><mml:mtext>&#x0205F;</mml:mtext><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M23"><mml:msub><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mo>:</mml:mo><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mo>:</mml:mo><mml:mo>,</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> denote the two columns of &#x003B1; (broadcasted across the <italic>d</italic>-dimensional feature vectors), and &#x02299; denotes element-wise (Hadamard) product. Through <xref ref-type="disp-formula" rid="E13">Equations 10</xref>&#x02013;<xref ref-type="disp-formula" rid="E15">12</xref>, the gating weights effectively <italic>modulate</italic> the contribution of each expert: if &#x003B1;<sub><italic>i</italic>, 1</sub>&#x0226B;&#x003B1;<sub><italic>i</italic>, 2</sub> for a given sample, the fused representation <italic>X</italic><sub>fused, <italic>i</italic></sub> will be dominated by ProtBERT&#x00027;s features, whereas if &#x003B1;<sub><italic>i</italic>, 2</sub> is larger, ProT5&#x00027;s features are emphasized. This adaptive fusion flexibly leverages the strengths of both experts, allowing the model to favor the expert that is more informative for each particular sample&#x00027;s attributes.</p>
<p>After obtaining the fused LLM-based features <italic>X</italic><sub>fused</sub>, we integrate them with the original statistical features. Specifically, we concatenate each sample&#x00027;s fused vector with its statistical descriptor vector to form the final combined feature:</p>
<disp-formula id="E15"><label>(12)</label><mml:math id="M24"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">combined</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">fused</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">stats</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>d</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>s</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In our implementation <italic>d</italic> &#x0003D; 1024 and <italic>s</italic> &#x0003D; 26, so <inline-formula><mml:math id="M25"><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">combined</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mn>1050</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>. This concatenation preserves the original handcrafted features alongside the fused deep features, ensuring that downstream classifiers receive a comprehensive feature set. The entire gated fusion procedure is summarized in <xref ref-type="table" rid="T4">Algorithm 2</xref>.</p>
<table-wrap position="float" id="T4">
<label>Algorithm 2</label>
<caption><p>Pseudocode of MoE gating and expert fusion.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-16-1634705-i0002.tif"/>
</table-wrap>
</sec>
<sec>
<title>3.4 Model training and prediction</title>
<p>After extracting and preprocessing the features, we employ a convolutional neural network (CNN) as the classification backbone for phage-host interaction prediction, integrating the MoE module for feature fusion throughout the training process. While we fix this CNN architecture for consistency, our main objective is to demonstrate how the proposed MoEPH framework leverages multi-source embeddings to boost predictive performance, the advantage would extend similarly to other model architectures. The overall training and evaluation pipeline is summarized in <xref ref-type="table" rid="T5">Algorithm 3</xref>.</p>
<table-wrap position="float" id="T5">
<label>Algorithm 3</label>
<caption><p>Model training and evaluation (CNN with MoE fusion).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-16-1634705-i0003.tif"/>
</table-wrap>
<sec>
<title>3.4.1 Network architecture</title>
<p>Let <italic>X</italic>&#x02208;&#x0211D;<sup><italic>N</italic>&#x000D7;<italic>d</italic></sup> be the combined feature matrix obtained via the MoE fusion layer, where <italic>N</italic> is the number of samples and <italic>d</italic> is the dimensionality of the fused features (including both Transformer-based protein embeddings and statistical descriptors). Each sample <inline-formula><mml:math id="M26"><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> is reshaped into a (<italic>d</italic>, 1) array, which is then fed into a series of 1D-CNN layers. Specifically, these layers consist of:</p>
<list list-type="bullet">
<list-item><p>One-dimensional convolutional layers (e.g., <monospace>kernel_size</monospace> = 3, varying channel widths), each followed by a nonlinear activation function (ReLU), batch normalization, and max-pooling;</p></list-item>
<list-item><p>Global pooling operation to reduce the final convolution outputs to a fixed-size feature vector;</p></list-item>
<list-item><p>Fully connected (FC) layers that project the pooled feature vector into logits for the binary classification task (interaction vs. non-interaction).</p></list-item>
</list>
<p>Additional regularization elements (e.g., dropout) may be introduced in the FC layers to mitigate overfitting.</p></sec>
<sec>
<title>3.4.2 Loss function</title>
<p>We adopt the cross-entropy loss for binary classification:</p>
<disp-formula id="E16"><label>(13)</label><mml:math id="M27"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo class="qopname">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo class="qopname">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>y</italic><sub><italic>i</italic></sub>&#x02208;{0, 1} is the true label for sample <italic>i</italic>, and <inline-formula><mml:math id="M28"><mml:mrow><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> is the predicted probability that sample <italic>i</italic> is positive (i.e., displays phage-host interaction). Concretely, let <italic>z</italic><sub><italic>i</italic>, 0</sub> and <italic>z</italic><sub><italic>i</italic>, 1</sub> be the logits for the negative and positive classes, respectively, so that</p>
<disp-formula id="E17"><label>(14)</label><mml:math id="M29"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Averaging over the entire training set of size <italic>N</italic> yields</p>
<disp-formula id="E18"><label>(15)</label><mml:math id="M30"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo class="qopname">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo class="qopname">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Minimizing <italic>L</italic>(&#x003B8;) with respect to the model parameters &#x003B8; steers the CNN toward more accurate probability estimates.</p></sec>
<sec>
<title>3.4.3 Training procedure</title>
<p>We initialize the CNN parameters &#x003B8; randomly and adopt the Adam optimizer with a selected learning rate (e.g., 1 &#x000D7; 10<sup>&#x02212;3</sup>). Training proceeds in mini-batches over a fixed number of epochs. For each mini-batch:</p>
<list list-type="order">
<list-item><p>We obtain the fused features <italic>X</italic><sub>batch_combined</sub> via the MoE layer.</p></list-item>
<list-item><p>Reshape them to (<italic>B, d</italic>, 1) for the 1D-CNN, where <italic>B</italic> is the mini-batch size.</p></list-item>
<list-item><p>Perform a forward pass through the CNN to compute logits and subsequently derive predicted probabilities.</p></list-item>
<list-item><p>Compute the cross-entropy loss using the predicted probabilities and the ground truth labels.</p></list-item>
<list-item><p>Backpropagate to compute gradients &#x02207;<sub>&#x003B8;</sub><italic>L</italic>(&#x003B8;) and update all relevant parameters (MoE and CNN) in one unified step.</p></list-item>
</list>
<p>This joint optimization ensures that the CNN learns discriminative features while the MoE layer continues to refine the expert-selection gating.</p></sec>
<sec>
<title>3.4.4 Prediction</title>
<p>After convergence, we apply the trained model to the test set. Specifically, the MoE layer fuses the expert embeddings for each test sample, the fused feature vectors are reshaped for the CNN, and the final probabilities <inline-formula><mml:math id="M31"><mml:mrow><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> are obtained via a Softmax layer. We use a threshold of 0.5 to determine the predicted class:</p>
<disp-formula id="E19"><label>(16)</label><mml:math id="M32"><mml:mrow><mml:msub><mml:mover accent='true'><mml:mi>y</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mi>j</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo></mml:mrow></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mtext>if&#x000A0;</mml:mtext><mml:msub><mml:mover accent='true'><mml:mi>p</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mi>j</mml:mi></mml:msub><mml:mo>&#x02265;</mml:mo><mml:mn>0.5</mml:mn><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo></mml:mrow></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mtext>otherwise</mml:mtext><mml:mo>.</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>Standard classification metrics (accuracy, F1-score, AUC-ROC, AUPR, etc.) are then computed to evaluate predictive performance.</p></sec></sec></sec>
<sec id="s4">
<title>4 Experimental results</title>
<sec>
<title>4.1 Data description and sampling strategies</title>
<p>This study evaluates the proposed model on three datasets (Dataset1, Dataset2, and a merged Dataset3 which combines the former two). Dataset 1 contains phage&#x02013;host pairs collected under relatively consistent experimental conditions (with well-defined species compositions and environmental factors), whereas Dataset 2 comes from a more complex ecological background, yielding greater heterogeneity in phage&#x02013;host interaction patterns. By merging these two sources into Dataset 3, we impose more stringent demands on the model&#x00027;s adaptability to heterogeneous, multi-source inputs. This strategy was chosen to preserve each dataset&#x00027;s unique characteristics and to assess whether the model generalizes across different sources. Had we merged the datasets from the beginning, any source-specific patterns or performance differences would be hidden. By first testing on each dataset individually, we can demonstrate MoEPH&#x00027;s robust performance under each condition, and then confirm its adaptability on the merged Dataset 3.</p>
<p>To further challenge the model&#x00027;s robustness, each dataset is examined under three class imbalance settings: the original imbalanced distribution (Raw), an Over-sampling variant, and an Under-sampling variant. In Over-sampling, instances from the minority class are replicated to balance the number of positive and negative samples (e.g., if <italic>N</italic><sub><italic>pos</italic></sub> and <italic>N</italic><sub><italic>neg</italic></sub> denote the counts of positive and negative samples with <italic>N</italic><sub><italic>pos</italic></sub>&#x0003C;<italic>N</italic><sub><italic>neg</italic></sub>, additional positive instances are randomly duplicated until <italic>N</italic>&#x02032;<italic>pos</italic>&#x02248;<italic>Nneg</italic>). In Under-sampling, the opposite approach is applied: majority-class instances are randomly removed until <italic>N</italic>&#x02032;<italic>neg</italic>&#x02248;<italic>Npos</italic>, thereby equalizing class counts. These three sampling methods simulate varying degrees of class imbalance encountered in real-world scenarios, enabling a comprehensive evaluation of the model&#x00027;s robustness across different data distributions.</p>
<sec>
<title>4.1.1 Sampling methods overview</title>
<p>To validate our model&#x00027;s performance under different class distributions and assess its generalization capability, we applied three sampling strategies to each dataset (as illustrated in <xref ref-type="fig" rid="F3">Figure 3</xref>):</p>
<list list-type="bullet">
<list-item><p><bold>Raw (imbalanced data):</bold> Directly using the original dataset while preserving its natural ratio of positive and negative samples, without any additional sampling.</p></list-item>
<list-item><p><bold>Over-sampling:</bold> Replicating instances from the minority class to balance the number of positive and negative samples. For instance, if <italic>N</italic><sub><italic>pos</italic></sub> and <italic>N</italic><sub><italic>neg</italic></sub> denote the number of positive and negative samples respectively and <italic>N</italic><sub><italic>pos</italic></sub>&#x0003C;<italic>N</italic><sub><italic>neg</italic></sub>, then over-sampling randomly duplicates some positive samples until <inline-formula><mml:math id="M33"><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02248;</mml:mo><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>.</p></list-item>
<list-item><p><bold>Under-sampling:</bold> The opposite approach, which randomly removes part of the majority class to match the minority class size. If <italic>N</italic><sub><italic>pos</italic></sub>&#x0003C;<italic>N</italic><sub><italic>neg</italic></sub>, we randomly eliminate some negative samples so that <inline-formula><mml:math id="M34"><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02248;</mml:mo><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>.</p></list-item>
</list>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>Data preprocessing pipeline for imbalanced datasets: under sampling, oversampling, and raw data.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-16-1634705-g0003.tif">
<alt-text>Diagram illustrating a data processing flow. Three datasets undergo sampling: positives to a random under sampler, and negatives to a random over sampler. Sampler 3 handles raw unbalanced data. Outputs undergo min-max normalization, adjusting values from original range to zero to one scale, resulting in normalized features.</alt-text>
</graphic>
</fig>
<p>These three sampling methods simulate varying degrees of class imbalance encountered in real-world scenarios, offering a more comprehensive evaluation of the model&#x00027;s adaptability and generalization performance across different data distributions.</p>
<sec>
<title>4.1.1.1 Min-max normalization</title>
<p>In addition to resampling, we apply a uniform preprocessing to all features. Min-Max Normalization: To alleviate discrepancies in feature value ranges, each feature is rescaled to [0, 1] via:</p>
<disp-formula id="E20"><label>(17)</label><mml:math id="M35"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>x</mml:mi><mml:mo>-</mml:mo><mml:mo class="qopname">min</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo class="qopname">max</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mo class="qopname">min</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>x</italic> is an original feature value and <italic>x</italic>&#x02032; is its normalized counterpart. This normalization expedites model convergence and enhances stability, especially when combining features of different scales. Finally, to assess the Generality and Applicability of our approach, we design experiments from multiple perspectives:</p>
<list list-type="bullet">
<list-item><p><bold>Class distribution impact:</bold> We compare model performance on naturally imbalanced data (Raw) versus balanced data (Over-sampled or Under-sampled) to gauge robustness to uneven class distributions.</p></list-item>
<list-item><p><bold>Multi-source backgrounds:</bold> By evaluating the algorithms on Dataset1, Dataset2, and the combined Dataset 3, we examine generalization under different biological settings and mixed conditions. This tests how well the model adapts to multi-source data variability, highlighting its robustness in a heterogeneous scenario.</p></list-item>
<list-item><p><bold>Feature visualization:</bold> Although the heatmaps (e.g., <xref ref-type="fig" rid="F4">Figure 4</xref>) primarily reflect the distribution of final fused features rather than changes before sampling, contrasting the feature patterns generated by different datasets and algorithms still offers useful insights into what aspects the model focuses on and how those relate to potential biological interpretations.</p>
</list-item>
</list>
<fig position="float" id="F4">
<label>Figure 4</label>
<caption><p>Heatmap of features in algorithms on different datasets. <bold>(a)</bold> Statistical-D1. <bold>(b)</bold> Statistical-D2. <bold>(c)</bold> Statistical-D3. <bold>(d)</bold> MoEPH(ProtBERT)-D1. <bold>(e)</bold> MoEPH(ProtBERT)-D2. <bold>(f)</bold> MoEPH(ProtBERT)-D3. <bold>(g)</bold> MoEPH(ProT5)-D1. <bold>(h)</bold> MoEPH(ProT5)-D2. <bold>(i)</bold> MoEPH(ProT5)-D3. <bold>(j)</bold> MoEPH(fused)-D1. <bold>(k)</bold> MoEPH(fused)-D2. <bold>(l)</bold> MoEPH(fused)-D3.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-16-1634705-g0004.tif">
<alt-text>Twelve heatmaps representing different datasets and methods are displayed in a 3&#x000D7;4 grid. Rows are titled Statistical, MoEPH(ProtBERT), MoEPH(ProT5), and MoEPH(fused), while columns are labeled D1, D2, and D3. Each heatmap varies in color intensity, depicting data variations for each method and dataset combination.</alt-text>
</graphic>
</fig>
</sec></sec></sec>
<sec>
<title>4.2 Performance evaluation metrics</title>
<p>In this study, we adopt several widely recognized metrics to evaluate the predictive capability of our model under class-imbalanced conditions (<xref ref-type="bibr" rid="B7">Fawcett, 2006</xref>; <xref ref-type="bibr" rid="B5">Davis and Goadrich, 2006</xref>; <xref ref-type="bibr" rid="B18">Saito and Rehmsmeier, 2015</xref>; <xref ref-type="bibr" rid="B10">Huang and Ling, 2005</xref>). Below, we provide their mathematical definitions and briefly discuss their relevance in the context of phage&#x02013;host interaction prediction.</p>
<disp-formula id="E21"><label>(18)</label><mml:math id="M36"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">ACC</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The proportion of all predictions that are correct.</p>
<disp-formula id="E22"><label>(19)</label><mml:math id="M37"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">Prec</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The proportion of predicted positive cases that are truly positive.</p>
<disp-formula id="E23"><label>(20)</label><mml:math id="M38"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">Spec</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The proportion of actual negative cases correctly identified (true negative rate).</p>
<disp-formula id="E24"><label>(21)</label><mml:math id="M39"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mtext class="textrm" mathvariant="normal">Precision</mml:mtext><mml:mo>&#x000D7;</mml:mo><mml:mtext class="textrm" mathvariant="normal">Recall</mml:mtext></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Precision</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:mtext class="textrm" mathvariant="normal">Recall</mml:mtext></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The harmonic mean of precision and recall.</p>
<disp-formula id="E25"><label>(22)</label><mml:math id="M40"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">AUC</mml:mtext><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mstyle><mml:mtext class="textrm" mathvariant="normal">TPR</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mi>f</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The area under the ROC curve (TPR vs. FPR), summarizing performance across all thresholds.</p>
<disp-formula id="E26"><label>(23)</label><mml:math id="M41"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">AUPR</mml:mtext><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mstyle><mml:mtext class="textrm" mathvariant="normal">Precision</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mi>r</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The area under the precision&#x02013;recall curve, reflecting the precision&#x02013;recall trade-off.</p></sec>
<sec>
<title>4.3 Feature representation analysis</title>
<p><xref ref-type="fig" rid="F4">Figure 4</xref> provides a comparative visualization of the final feature representations produced by different methods across the three datasets. Each vertical column corresponds to a specific feature channel, and the color intensity indicates the magnitude of the feature values (with darker shades representing higher values and lighter shades indicating lower values). These heatmaps are derived from features output by our MoEPH model after the expert fusion layer, just before classification, under various configurations: using only statistical features, using ProtBERT embeddings, using ProT5 embeddings, and using the fused MoE approach (ProtBERT&#x0002B;ProT5). Compared to the traditional statistical baseline, the MoEPH-based models display more distinct block and stripe patterns, indicating that MoEPH captures finer-grained, context-dependent sequence features than conventional methods. Each subfigure corresponds to one method on one dataset (<xref ref-type="fig" rid="F4">Figures 4a</xref>&#x02013;<xref ref-type="fig" rid="F4">l</xref>), allowing side-by-side comparison of how feature distributions vary by method and data complexity. The key observations are as follows:</p>
<sec>
<title>4.3.1 Statistical method vs. MoEPH-based features</title>
<p>Compared to the &#x0201C;Statistical Method&#x0201D; (see <xref ref-type="fig" rid="F4">Figures 4a</xref>&#x02013;<xref ref-type="fig" rid="F4">c</xref>), the MoEPH variants (ProtBERT only, ProT5 only, and the fused approach) typically yield more distinct vertical stripes or block patterns in their heatmaps, with sharper gradients across various feature columns. Because the Statistical Method relies on basic statistical measurements, its derived feature vectors often exhibit relatively homogeneous distribution patterns. By contrast, the MoEPH approaches, having leveraged large pre-trained models, are better able to capture fine-grained, context-dependent sequence representations, leading to more pronounced differences among samples.</p></sec>
<sec>
<title>4.3.2 Differences between ProtBERT and ProT5</title>
<p>As shown in <xref ref-type="fig" rid="F4">Figures 4d</xref>&#x02013;<xref ref-type="fig" rid="F4">i</xref>, using ProtBERT versus ProT5 for feature extraction can produce noticeably different heatmap patterns in certain feature columns. In some dimensions, ProtBERT&#x00027;s response appears more &#x0201C;striped,&#x0201D; whereas ProT5 may display broader regions of lighter or darker shades. This indicates that the two pre-trained models have distinct sensitivities or biases when encoding the same sequence information. Such disparities become even more pronounced for Dataset 2 and Dataset 3, suggesting each model exhibits unique strengths in capturing sequence features under more complex biological conditions.</p></sec>
<sec>
<title>4.3.3 Performance of the dated fusion [MoEPH (fused)]</title>
<p><xref ref-type="fig" rid="F4">Figures 4j</xref>&#x02013;<xref ref-type="fig" rid="F4">l</xref> present the heatmaps of features obtained by gating and fusing the ProtBERT and ProT5 experts. These images reveal that while certain ProtBERT-like or ProT5-like textures remain, some local regions appear smoothed out or amplified. This indicates that the gating mechanism is not merely an averaging step but rather a selective weighting process driven by statistical features and sequence context, yielding a more diversified feature distribution in complex data scenarios.</p></sec>
<sec>
<title>4.3.4 Changes across datasets</title>
<p>A vertical comparison from Dataset 1 through Dataset 3 shows that as the dataset size and heterogeneity increase, the color patterns&#x02014;blocks and stripes&#x02014;also become more pronounced. For instance, in Dataset 3, which includes more varied samples, the color intensity for a given feature column may fluctuate more widely across different instances, suggesting that the large pre-trained models have learned broader sequence distinctions. Conversely, if a method is relatively insensitive to environmental variation, its heatmaps may exhibit similar patterns across datasets, implying limited adaptability in its learned feature space.</p>
<p>Overall, these visual observations suggest that MoEPH-based methods manifest more distinct block structures in their features and are better equipped than traditional statistical approaches to capture deeper sequence-level variations&#x02014;an advantage that can underpin improved classification results in subsequent experiments.</p></sec></sec>
<sec>
<title>4.4 Performance comparison of and analysis</title>
<p>We conducted an in-depth evaluation involving 36 different experimental settings, spanning three datasets, four algorithms, and three data sampling strategies. As shown in <xref ref-type="table" rid="T1">Table 1</xref>, the MoEPH series&#x02014;whether incorporating ProtBERT, ProT5, or both&#x02014;achieved high performance in most scenarios, covering 239 out of 252 metrics (i.e., 94.9%). Its maximum accuracy reached 99.6%, significantly surpassing the current state-of-the-art methods (see <xref ref-type="table" rid="T2">Table 2</xref>). Moreover, in more complex datasets, the accuracy (ACC) was improved by as much as 31%, rising from 0.535 to 0.845 (<xref ref-type="fig" rid="F5">Figure 5</xref>). In some over-sampled scenarios, certain cross-validation folds with very few positive instances yielded a Sensitivity of 1.0. We note that this perfect recall is due to the extremely low number of positives in those folds, reflecting class imbalance.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Performance comparison across three datasets (D1, D2, D3) under three sampling strategies (Raw, Over, Under) with four algorithms [Statistical, MoEPH (ProtBERT), MoEPH (ProT5), MoEPH (fused)].</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Dataset &#x0002B; sampling</bold></th>
<th valign="top" align="left"><bold>Metric</bold></th>
<th valign="top" align="left"><bold>Statistical (baseline)</bold></th>
<th valign="top" align="left"><bold>MoEPH (ProtBERT)</bold></th>
<th valign="top" align="left"><bold>MoEPH (ProT5)</bold></th>
<th valign="top" align="left"><bold>MoEPH (fused)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><bold>D1 - Raw</bold></td>
<td valign="top" align="left">ACC</td>
<td valign="top" align="left">0.966</td>
<td valign="top" align="left">0.968</td>
<td valign="top" align="left"><underline>0.975</underline></td>
<td valign="top" align="left">0.969</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">F1</td>
<td valign="top" align="left">0.81</td>
<td valign="top" align="left">0.81</td>
<td valign="top" align="left"><underline>0.852</underline></td>
<td valign="top" align="left">0.827</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUPR</td>
<td valign="top" align="left">0.837</td>
<td valign="top" align="left"><underline>0.89</underline></td>
<td valign="top" align="left">0.878</td>
<td valign="top" align="left">0.887</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUC</td>
<td valign="top" align="left">0.969</td>
<td valign="top" align="left">0.975</td>
<td valign="top" align="left"><underline>0.977</underline></td>
<td valign="top" align="left">0.972</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Sens</td>
<td valign="top" align="left">0.8</td>
<td valign="top" align="left">0.743</td>
<td valign="top" align="left"><underline>0.804</underline></td>
<td valign="top" align="left">0.801</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Spec</td>
<td valign="top" align="left">0.982</td>
<td valign="top" align="left">0.99</td>
<td valign="top" align="left"><underline>0.992</underline></td>
<td valign="top" align="left">0.986</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Prec</td>
<td valign="top" align="left">0.821</td>
<td valign="top" align="left">0.891</td>
<td valign="top" align="left"><underline>0.906</underline></td>
<td valign="top" align="left">0.854</td>
</tr> <tr>
<td valign="top" align="left"><bold>D1 - Over</bold></td>
<td valign="top" align="left">ACC</td>
<td valign="top" align="left">0.98</td>
<td valign="top" align="left">0.98</td>
<td valign="top" align="left"><underline>0.984</underline></td>
<td valign="top" align="left">0.978</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">F1</td>
<td valign="top" align="left">0.98</td>
<td valign="top" align="left">0.98</td>
<td valign="top" align="left"><underline>0.984</underline></td>
<td valign="top" align="left">0.978</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUPR</td>
<td valign="top" align="left">0.994</td>
<td valign="top" align="left">0.987</td>
<td valign="top" align="left"><underline>0.997</underline></td>
<td valign="top" align="left">0.993</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUC</td>
<td valign="top" align="left">0.996</td>
<td valign="top" align="left">0.994</td>
<td valign="top" align="left"><underline>0.998</underline></td>
<td valign="top" align="left">0.996</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Sens</td>
<td valign="top" align="left"><underline>0.999</underline></td>
<td valign="top" align="left">0.998</td>
<td valign="top" align="left"><underline>0.999</underline></td>
<td valign="top" align="left">0.992</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Spec</td>
<td valign="top" align="left">0.961</td>
<td valign="top" align="left">0.962</td>
<td valign="top" align="left"><underline>0.969</underline></td>
<td valign="top" align="left">0.965</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Prec</td>
<td valign="top" align="left">0.963</td>
<td valign="top" align="left">0.963</td>
<td valign="top" align="left"><underline>0.97</underline></td>
<td valign="top" align="left">0.965</td>
</tr> <tr>
<td valign="top" align="left"><bold>D1 - Under</bold></td>
<td valign="top" align="left">ACC</td>
<td valign="top" align="left">0.917</td>
<td valign="top" align="left"><underline>0.964</underline></td>
<td valign="top" align="left">0.901</td>
<td valign="top" align="left">0.891</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">F1</td>
<td valign="top" align="left"><underline>0.919</underline></td>
<td valign="top" align="left">0.791</td>
<td valign="top" align="left">0.905</td>
<td valign="top" align="left">0.899</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUPR</td>
<td valign="top" align="left"><underline>0.948</underline></td>
<td valign="top" align="left">0.828</td>
<td valign="top" align="left">0.939</td>
<td valign="top" align="left">0.926</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUC</td>
<td valign="top" align="left">0.956</td>
<td valign="top" align="left"><underline>0.966</underline></td>
<td valign="top" align="left">0.952</td>
<td valign="top" align="left">0.946</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Sens</td>
<td valign="top" align="left">0.937</td>
<td valign="top" align="left">0.761</td>
<td valign="top" align="left">0.939</td>
<td valign="top" align="left"><underline>0.957</underline></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Spec</td>
<td valign="top" align="left">0.896</td>
<td valign="top" align="left"><underline>0.984</underline></td>
<td valign="top" align="left">0.863</td>
<td valign="top" align="left">0.825</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Prec</td>
<td valign="top" align="left"><underline>0.901</underline></td>
<td valign="top" align="left">0.824</td>
<td valign="top" align="left">0.874</td>
<td valign="top" align="left">0.847</td>
</tr> <tr>
<td valign="top" align="left"><bold>D2 - Raw</bold></td>
<td valign="top" align="left">ACC</td>
<td valign="top" align="left"><underline>0.988</underline></td>
<td valign="top" align="left"><underline>0.988</underline></td>
<td valign="top" align="left">0.985</td>
<td valign="top" align="left">0.985</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">F1</td>
<td valign="top" align="left"><underline>0.768</underline></td>
<td valign="top" align="left">0.749</td>
<td valign="top" align="left">0.725</td>
<td valign="top" align="left">0.674</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUPR</td>
<td valign="top" align="left"><underline>0.707</underline></td>
<td valign="top" align="left">0.686</td>
<td valign="top" align="left">0.654</td>
<td valign="top" align="left">0.684</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUC</td>
<td valign="top" align="left">0.875</td>
<td valign="top" align="left">0.872</td>
<td valign="top" align="left"><underline>0.903</underline></td>
<td valign="top" align="left">0.864</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Sens</td>
<td valign="top" align="left"><underline>0.716</underline></td>
<td valign="top" align="left">0.642</td>
<td valign="top" align="left">0.679</td>
<td valign="top" align="left">0.541</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Spec</td>
<td valign="top" align="left">0.996</td>
<td valign="top" align="left"><underline>0.998</underline></td>
<td valign="top" align="left">0.994</td>
<td valign="top" align="left"><underline>0.998</underline></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Prec</td>
<td valign="top" align="left">0.83</td>
<td valign="top" align="left"><underline>0.897</underline></td>
<td valign="top" align="left">0.779</td>
<td valign="top" align="left">0.894</td>
</tr> <tr>
<td valign="top" align="left"><bold>D2 - Over</bold></td>
<td valign="top" align="left">ACC</td>
<td valign="top" align="left"><underline>0.997</underline></td>
<td valign="top" align="left">0.996</td>
<td valign="top" align="left">0.996</td>
<td valign="top" align="left">0.996</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">F1</td>
<td valign="top" align="left">0.997</td>
<td valign="top" align="left">0.996</td>
<td valign="top" align="left"><underline>0.999</underline></td>
<td valign="top" align="left">0.996</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUPR</td>
<td valign="top" align="left">0.999</td>
<td valign="top" align="left">0.999</td>
<td valign="top" align="left"><underline>1</underline></td>
<td valign="top" align="left">0.998</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUC</td>
<td valign="top" align="left"><underline>0.999</underline></td>
<td valign="top" align="left"><underline>0.999</underline></td>
<td valign="top" align="left"><underline>0.999</underline></td>
<td valign="top" align="left"><underline>0.999</underline></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Sens</td>
<td valign="top" align="left"><underline>1</underline></td>
<td valign="top" align="left"><underline>1</underline></td>
<td valign="top" align="left"><underline>1</underline></td>
<td valign="top" align="left"><underline>1</underline></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Spec</td>
<td valign="top" align="left"><underline>0.994</underline></td>
<td valign="top" align="left">0.991</td>
<td valign="top" align="left">0.993</td>
<td valign="top" align="left">0.992</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Prec</td>
<td valign="top" align="left"><underline>0.994</underline></td>
<td valign="top" align="left">0.991</td>
<td valign="top" align="left">0.993</td>
<td valign="top" align="left">0.992</td>
</tr> <tr>
<td valign="top" align="left"><bold>D2 - Under</bold></td>
<td valign="top" align="left">ACC</td>
<td valign="top" align="left">0.53</td>
<td valign="top" align="left">0.84</td>
<td valign="top" align="left"><underline>0.845</underline></td>
<td valign="top" align="left"><underline>0.845</underline></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">F1</td>
<td valign="top" align="left">0.44</td>
<td valign="top" align="left">0.826</td>
<td valign="top" align="left"><underline>0.832</underline></td>
<td valign="top" align="left">0.829</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUPR</td>
<td valign="top" align="left">0.577</td>
<td valign="top" align="left">0.732</td>
<td valign="top" align="left">0.814</td>
<td valign="top" align="left"><underline>0.859</underline></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUC</td>
<td valign="top" align="left">0.526</td>
<td valign="top" align="left">0.824</td>
<td valign="top" align="left">0.85</td>
<td valign="top" align="left"><underline>0.899</underline></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Sens</td>
<td valign="top" align="left">0.393</td>
<td valign="top" align="left">0.809</td>
<td valign="top" align="left"><underline>0.819</underline></td>
<td valign="top" align="left">0.798</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Spec</td>
<td valign="top" align="left">0.65</td>
<td valign="top" align="left">0.868</td>
<td valign="top" align="left">0.868</td>
<td valign="top" align="left"><underline>0.887</underline></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Prec</td>
<td valign="top" align="left">0.5</td>
<td valign="top" align="left">0.844</td>
<td valign="top" align="left">0.846</td>
<td valign="top" align="left"><underline>0.862</underline></td>
</tr> <tr>
<td/>
<td valign="top" align="left">ACC</td>
<td valign="top" align="left">0.95</td>
<td valign="top" align="left">0.954</td>
<td valign="top" align="left">0.947</td>
<td valign="top" align="left"><underline>0.96</underline></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">F1</td>
<td valign="top" align="left">0.663</td>
<td valign="top" align="left">0.648</td>
<td valign="top" align="left">0.628</td>
<td valign="top" align="left"><underline>0.7</underline></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUPR</td>
<td valign="top" align="left">0.575</td>
<td valign="top" align="left">0.62</td>
<td valign="top" align="left">0.581</td>
<td valign="top" align="left"><underline>0.724</underline></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUC</td>
<td valign="top" align="left">0.895</td>
<td valign="top" align="left"><underline>0.904</underline></td>
<td valign="top" align="left">0.902</td>
<td valign="top" align="left">0.9</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Sens</td>
<td valign="top" align="left"><underline>0.657</underline></td>
<td valign="top" align="left">0.566</td>
<td valign="top" align="left">0.6</td>
<td valign="top" align="left">0.623</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Spec</td>
<td valign="top" align="left">0.974</td>
<td valign="top" align="left">0.985</td>
<td valign="top" align="left">0.975</td>
<td valign="top" align="left"><underline>0.988</underline></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Prec</td>
<td valign="top" align="left">0.67</td>
<td valign="top" align="left">0.757</td>
<td valign="top" align="left">0.658</td>
<td valign="top" align="left"><underline>0.8</underline></td>
</tr> <tr>
<td valign="top" align="left"><bold>D3 - Over</bold></td>
<td valign="top" align="left">ACC</td>
<td valign="top" align="left">0.969</td>
<td valign="top" align="left">0.961</td>
<td valign="top" align="left"><underline>0.977</underline></td>
<td valign="top" align="left">0.966</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">F1</td>
<td valign="top" align="left">0.97</td>
<td valign="top" align="left">0.962</td>
<td valign="top" align="left"><underline>0.977</underline></td>
<td valign="top" align="left">0.967</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUPR</td>
<td valign="top" align="left">0.985</td>
<td valign="top" align="left">0.976</td>
<td valign="top" align="left"><underline>0.988</underline></td>
<td valign="top" align="left">0.986</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUC</td>
<td valign="top" align="left">0.991</td>
<td valign="top" align="left">0.986</td>
<td valign="top" align="left"><underline>0.993</underline></td>
<td valign="top" align="left">0.991</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Sens</td>
<td valign="top" align="left">0.993</td>
<td valign="top" align="left">0.983</td>
<td valign="top" align="left"><underline>0.996</underline></td>
<td valign="top" align="left">0.992</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Spec</td>
<td valign="top" align="left">0.946</td>
<td valign="top" align="left">0.94</td>
<td valign="top" align="left"><underline>0.958</underline></td>
<td valign="top" align="left">0.941</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Prec</td>
<td valign="top" align="left">0.948</td>
<td valign="top" align="left">0.9415</td>
<td valign="top" align="left"><underline>0.96</underline></td>
<td valign="top" align="left">0.943</td>
</tr> <tr>
<td valign="top" align="left"><bold>D3 - Under</bold></td>
<td valign="top" align="left">ACC</td>
<td valign="top" align="left">0.790</td>
<td valign="top" align="left"><underline>0.793</underline></td>
<td valign="top" align="left">0.782</td>
<td valign="top" align="left">0.74</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">F1</td>
<td valign="top" align="left"><underline>0.796</underline></td>
<td valign="top" align="left">0.793</td>
<td valign="top" align="left">0.789</td>
<td valign="top" align="left">0.745</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUPR</td>
<td valign="top" align="left">0.853</td>
<td valign="top" align="left">0.831</td>
<td valign="top" align="left"><underline>0.855</underline></td>
<td valign="top" align="left">0.85</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">AUC</td>
<td valign="top" align="left"><underline>0.857</underline></td>
<td valign="top" align="left">0.851</td>
<td valign="top" align="left">0.85</td>
<td valign="top" align="left">0.833</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Sens</td>
<td valign="top" align="left"><underline>0.818</underline></td>
<td valign="top" align="left">0.798</td>
<td valign="top" align="left"><underline>0.818</underline></td>
<td valign="top" align="left">0.759</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Spec</td>
<td valign="top" align="left">0.763</td>
<td valign="top" align="left"><underline>0.788</underline></td>
<td valign="top" align="left">0.745</td>
<td valign="top" align="left">0.722</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">Prec</td>
<td valign="top" align="left">0.775</td>
<td valign="top" align="left"><underline>0.789</underline></td>
<td valign="top" align="left">0.762</td>
<td valign="top" align="left">0.73</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold-underlined entries in each row indicate the highest value(s).</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Comparison of best accuracy among state-of-the-art methods in phage&#x02013;host interaction prediction.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Algorithm</bold></th>
<th valign="top" align="left"><bold>Best accuracy (%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><bold>MoEPH (Ours)</bold></td>
<td valign="top" align="left">99.6%</td>
</tr> <tr>
<td valign="top" align="left">PredPHI (<xref ref-type="bibr" rid="B12">Li et al., 2020</xref>)</td>
<td valign="top" align="left">81%</td>
</tr> <tr>
<td valign="top" align="left">Host Phinder (<xref ref-type="bibr" rid="B20">Villarroel et al., 2016</xref>)</td>
<td valign="top" align="left">81%</td>
</tr> <tr>
<td valign="top" align="left">VirHost Matcher (<xref ref-type="bibr" rid="B1">Ahlgren et al., 2017</xref>)</td>
<td valign="top" align="left">64%</td>
</tr> <tr>
<td valign="top" align="left">WIsH (<xref ref-type="bibr" rid="B8">Galiez et al., 2017</xref>)</td>
<td valign="top" align="left">63%</td>
</tr> <tr>
<td valign="top" align="left">LMFH VH (<xref ref-type="bibr" rid="B13">Liu et al., 2018</xref>)</td>
<td valign="top" align="left">63.17%</td>
</tr> <tr>
<td valign="top" align="left">ILMF VH (<xref ref-type="bibr" rid="B14">Liu et al., 2019</xref>)</td>
<td valign="top" align="left">63.66%</td>
</tr>
<tr>
<td valign="top" align="left"><xref ref-type="bibr" rid="B11">Leite et al. (2018)</xref></td>
<td valign="top" align="left">95.7%</td>
</tr></tbody>
</table>
</table-wrap>
<fig position="float" id="F5">
<label>Figure 5</label>
<caption><p>Radar chart of metrics in dataset 2 (under sample).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-16-1634705-g0005.tif">
<alt-text>Radar chart comparing four models: Statistical (Baseline), MoEPH (ProtBERT), MoEPH (ProT5), and MoEPH (fused) across eight metrics: ACC, F1, AUPR, AUC, Sens, Spec, Prec. Baseline is blue, ProtBERT is orange, ProT5 is green, and fused is red. MoEPH models generally outperform the baseline.</alt-text>
</graphic>
</fig>
<p><xref ref-type="fig" rid="F6">Figure 6</xref> illustrates the comparison between our proposed MoEPH variants (ProtBERT, ProT5, and fused) and the Statistical approach, as well as PredPHI (<xref ref-type="bibr" rid="B12">Li et al., 2020</xref>), under the Dataset 2 (Under-sampling) scenario. The metrics presented include Accuracy (ACC), Sensitivity (Sens), and Specificity (Spec). We observe that MoEPH (ProtBERT) and MoEPH (ProT5) attain ACC values of 0.84 and 0.845, respectively, outperforming both the Statistical method (0.53) and PredPHI (0.78) by a notable margin. They also exhibit superior Sensitivity, indicating that even under severe under-sampling, the model can still capture a larger portion of positive samples, thus reducing the rate of missed detections. Furthermore, to comprehensively evaluate performance within the same CNN classification architecture, we expand our discussion to include additional metrics, analyzing the results from three key perspectives: varying protein representation algorithms, different sampling strategies, and multiple datasets.</p>
<fig position="float" id="F6">
<label>Figure 6</label>
<caption><p>Comparison on metrics in Dataset 2 (under sample) with PredPHI (<xref ref-type="bibr" rid="B12">Li et al., 2020</xref>).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-16-1634705-g0006.tif">
<alt-text>Bar chart comparing five models: Statistical (Baseline), MoEPH (ProtBERT), MoEPH (ProtT5), MoEPH (fused), and PredPHI across three metrics: Accuracy (ACC), Sensitivity, and Specificity. PredPHI shows the highest values in all metrics, with 0.845 for ACC, 0.819 for Sensitivity, and 0.900 for Specificity. Statistical (Baseline) has the lowest across metrics.</alt-text>
</graphic>
</fig>
<sec>
<title>4.4.1 Comparison across different algorithms</title>
<sec>
<title>4.4.1.1 Single-expert vs. fusion</title>
<list list-type="bullet">
<list-item><p><bold>MoEPH (ProtBERT) vs. MoEPH (ProT5)</bold>. Each model shows its strengths under different conditions. For instance, in <bold>D1-Raw</bold> and <bold>D1-Over</bold>, MoEPH(ProT5) outperforms MoEPH (ProtBERT) in multiple metrics (ACC, F1, Spec, Prec). Conversely, under <bold>D1-Under</bold>, MoEPH (ProtBERT) achieves higher ACC = 0.964, AUC = 0.966, and Spec = 0.984. This discrepancy indicates that the two pretrained models diverge in their focus on protein sequences, likely attributable to differences in training corpora and model architectures.</p></list-item>
<list-item><p><bold>MoEPH (fused)</bold>. In many scenarios (e.g., D3-Raw), the gating-fusion approach can integrate the merits of both experts and emerge as the best or near-best across multiple metrics (ACC = 0.96, F1 = 0.70, AUPR = 0.724, Prec = 0.80). Nevertheless, there are instances (e.g., D1-Raw) where it slightly lags behind a single expert in some metrics, yet still maintains robust overall performance and often remains on par with or superior to baseline methods. Consequently, gating fusion tends to excel in more complex or highly imbalanced data distributions, whereas a single expert may suffice in simpler scenarios or under near-optimal sampling conditions (e.g., over-sampling).</p></list-item>
</list></sec>
<sec>
<title>4.4.1.2 Surprising strength of the statistical baseline</title>
<p>Although the learned embedding features generally produce superior results, the traditional statistical feature baseline exhibited some surprisingly competitive performances in specific cases. In many comparisons the statistical approach does not achieve the highest scores&#x02014;highlighting its limited capacity to capture deep semantic cues&#x02014;yet under certain dataset and sampling configurations it ties or even outperforms the more complex models on some metrics. For example, in D2-Raw, the statistical method attains leading or tied performance with ACC = 0.988 (tied with ProtBERT), F1 = 0.768, AUPR = 0.707, and Sens = 0.716. Likewise, in D1-Under, it achieves the highest F1 = 0.919, AUPR = 0.948, Sens = 0.937, and Prec = 0.901, while MoEPH (ProtBERT) leads in ACC, AUC, and Spec. This &#x0201C;split across metrics&#x0201D; illustrates that, though the statistical method lacks the contextual awareness derived from large-scale pre-training, its more streamlined features can adapt well to particular data distributions or under-sampling schemes, thereby yielding notably effective positive-class recognition in certain cases.</p></sec></sec>
<sec>
<title>4.4.2 Comparison across sampling methods</title>
<sec>
<title>4.4.2.1 Over-sampling</title>
<p>In the D1-Over, D2-Over, and D3-Over scenarios, most algorithms achieve extremely high Accuracy, AUC, and even Sensitivity = 1, with differences often only discernible at the third decimal place. Over-sampling balances the number of positive and negative classes by substantially amplifying the positive samples, thus making learning more straightforward for most methods and providing enough data to correct any prior bias toward the negative class.</p></sec>
<sec>
<title>4.4.2.2 Under-sampling</title>
<p>Under-sampling also balances class counts but does so by heavily removing majority-class samples, which reduces the total amount of available information. For example, in D2-Under, the Statistical method&#x00027;s Accuracy of 0.53 and F1 of 0.44 are conspicuously lower than the MoEPH variants (all exceeding 0.82 in F1). This indicates that large-scale pre-trained representations can maintain discriminative power even under extreme data reduction. Meanwhile, because D1-Under is intrinsically easier to separate, the Statistical approach outperforms single-model variants in certain metrics (F1, AUPR, Sens, Prec), yet still exhibits a noticeable gap in Accuracy. While the statistical model achieves competitive scores on some datasets, its lack of contextual embedding and limited generalization restricts its utility in more diverse prediction settings (<xref ref-type="fig" rid="F7">Figure 7</xref>).</p>
<fig position="float" id="F7">
<label>Figure 7</label>
<caption><p>Confusion matrices on Dataset2 (Under-sampled) for <bold>(a)</bold> the Statistical baseline and <bold>(b)</bold> MoEPH (fused).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-16-1634705-g0007.tif">
<alt-text>Two confusion matrices compare the performance of models. The left matrix (a) Baseline shows 40% True Positives and 65% True Negatives. The right matrix (b) MoEPH (fused) shows 80% True Positives and 88% True Negatives, indicating improved accuracy.</alt-text>
</graphic>
</fig>
<p>And each cell in <xref ref-type="fig" rid="F8">Figure 8</xref> shows the percentage of instances (on the test set) belonging to the actual class (rows: Positive or Negative) that were predicted as Positive or Negative (columns). The baseline (a) misses a majority of actual positives (only 40% recall) and produces many false positives (35%), whereas MoEPH (fused) (b) correctly identifies the vast majority of positives (80% recall) while keeping false positives low (12%). This demonstrates MoEPH&#x00027;s significantly improved balance between sensitivity and specificity under extreme class imbalance.</p>
<fig position="float" id="F8">
<label>Figure 8</label>
<caption><p>Loss curves for four algorithms.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-16-1634705-g0008.tif">
<alt-text>Line graph showing loss versus epoch for different models. The blue line, representing the statistical baseline, remains around 0.5. The other lines, MoEPH(ProtBERT) in orange, MoEPH(ProtT4) in green, and MoEPH(fused) in red, start around 0.4 and converge below 0.1 over three thousand epochs, indicating better performance.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<title>4.4.2.3 Raw</title>
<p>Retaining the natural distribution of the data (Raw) can cause fluctuations in some metrics. Nonetheless, across D1, D2, and D3, the MoEPH series generally demonstrates sufficiently strong performance, suggesting that, under real-world distributions, these methods&#x00027; differences and applicability become more evident.</p></sec></sec>
<sec>
<title>4.4.3 Comparison across different datasets</title>
<sec>
<title>4.4.3.1 D1 vs. D2</title>
<p>D1 appears easier to separate; when over-sampling is applied, nearly all metrics exceed 0.98, reflecting high performance under any sampling approach. In contrast, D2 can reach near-perfect scores under over-sampling (Sensitivity = 1), yet experiences a drastic decline in some algorithms under under-sampling (e.g., Statistical with F1 = 0.44 vs. MoEPH (fused) with F1 = 0.829). This stark contrast indicates that for D2&#x00027;s more complex distribution, substantially removing majority-class samples significantly impairs methods lacking higher-level sequence semantics and contextual awareness. As shown in <xref ref-type="fig" rid="F8">Figure 8</xref>, the Statistical (Baseline) approach fails to converge, whereas MoEPH-based algorithms rapidly reduce loss within the initial epochs. This highlights how pre-trained models and gating fusion can more effectively capture deep protein-sequence features, thereby converging to superior solutions with fewer training iterations.</p></sec>
<sec>
<title>4.4.3.2 D3</title>
<p>D3 integrates or extends the complexities of the previous distributions, yielding results that are neither &#x0201C;near perfect&#x0201D; (as in D2-Over or D1-Over) nor drastically diminished (as in D2-Under). In D3-Raw, MoEPH (fused) achieves F1 = 0.70, AUPR = 0.724, and Prec = 0.80, clearly exceeding Statistical (F1 = 0.663, AUPR = 0.575, Prec = 0.67), while ProtBERT or ProT5 also excel in selected metrics. This suggests that under broader distributions, large-scale models&#x00027; fine-grained sequence representations can further uncover subtle differences. Likewise, D3-Over yields nearly optimal outcomes, minimizing algorithmic discrepancies; in D3-Under, metrics as a whole decline, yet Statistical remains reasonably competitive (ACC = 0.79, F1 = 0.796, AUPR = 0.853). However, certain MoEPH methods still exhibit minor advantages across other metrics, though the gap is less extreme than in D2-Under. Overall, different datasets impose substantial impact on each algorithm&#x00027;s performance, yet MoEPH consistently outperforms baseline approaches in most metrics.</p></sec></sec></sec>
<sec>
<title>4.5 Clinical innovation</title>
<p>MoEPH holds promising potential to innovate patient care in the context of antibiotic-resistant infections. By rapidly and accurately predicting phage&#x02013;host interactions, MoEPH could assist clinicians in selecting effective phage therapies tailored to a patient&#x00027;s drug-resistant bacterial infection, exemplifying precision medicine in infectious disease treatment. This approach could be integrated into the clinical workflow as a decision-support tool, where its robust and interpretable predictions provide physicians with high-confidence recommendations for alternative treatments when antibiotics fail. The model&#x00027;s emphasis on interpretability and reliability builds the trust necessary for clinical adoption, ensuring that healthcare providers can understand and rely on its suggestions. Ultimately, a trustworthy AI system like MoEPH could streamline the management of AMR cases&#x02014;improving treatment outcomes by offering timely, personalized therapeutic options and potentially integrating into hospital infection control and antibiotic stewardship programs.</p></sec></sec>
<sec sec-type="conclusions" id="s5">
<title>5 Conclusion</title>
<p>In this work, we presented MoEPH&#x02014;a mixture-of-experts model that combines traditional statistical descriptors with deep protein embeddings (ProtBERT and ProT5) to tackle the phage&#x02013;host prediction problem. Experiments on three benchmark datasets with varied sampling regimes demonstrated that MoEPH consistently outperforms both conventional statistical classifiers and single-model LLM baselines. Notably, MoEPH achieved up to 99.6% accuracy on balanced datasets, and improved accuracy by as much as 31 percentage points on highly imbalanced datasets. The model&#x00027;s adaptive fusion of domain-specific features with pre-trained embeddings ensures robust generalization, while its gating mechanism provides transparency by indicating each expert&#x00027;s contribution to a given prediction. Looking ahead, we plan to further enhance MoEPH along several directions. First, we will incorporate structural protein features (e.g., 3D conformational information) to complement the sequence-based embeddings. Second, we aim to explore alternative neural network architectures as backbones for the expert models, which may uncover additional performance gains. These enhancements are expected to broaden MoEPH&#x00027;s applicability to diverse biomedical prediction tasks, while ensuring the model remains a reliable and transparent AI tool for real-world phage&#x02013;host identification challenges.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>QC: Writing &#x02013; original draft, Methodology, Data curation, Investigation, Project administration, Formal analysis. ZZ: Data curation, Investigation, Writing &#x02013; review &#x00026; editing. ML: Data curation, Writing &#x02013; review &#x00026; editing. WS: Investigation, Writing &#x02013; review &#x00026; editing. MX: Validation, Visualization, Funding acquisition, Writing &#x02013; original draft. MF: Writing &#x02013; original draft, Methodology, Visualization.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This research was supported by the Beijing Genomics Institute under grant numbers 62176164 and 62203134. National Nature Science Foundation of China Grant 32100130, the datasets is provided by BGI-Shenzhen (CNSA:CNP0006217) and the Scientific Foundation for Youth Scholars of Shenzhen University (868-000001033385).</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p></sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn0001"><p><sup>1</sup>CNSA accession number: CNP0006217, available at <ext-link ext-link-type="uri" xlink:href="https://db.cngb.org/search/project/CNP0006217/">https://db.cngb.org/search/project/CNP0006217/</ext-link>.</p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ahlgren</surname> <given-names>N. A.</given-names></name> <name><surname>Ren</surname> <given-names>J.</given-names></name> <name><surname>Lu</surname> <given-names>Y. Y.</given-names></name> <name><surname>Fuhrman</surname> <given-names>J. A.</given-names></name> <name><surname>Sun</surname> <given-names>F.</given-names></name></person-group> (<year>2017</year>). <article-title>Alignment-free oligonucleotide frequency dissimilarity measure improves prediction of hosts from metagenomically-derived viral sequences</article-title>. <source>Nucleic Acids Res</source>. <volume>45</volume>, <fpage>39</fpage>&#x02013;<lpage>53</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw1002</pub-id><pub-id pub-id-type="pmid">27899557</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Aljohani</surname> <given-names>M.</given-names></name> <name><surname>Hou</surname> <given-names>J.</given-names></name> <name><surname>Kommu</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name></person-group> (<year>2025</year>). <article-title>A comprehensive survey on the trustworthiness of large language models in healthcare</article-title>. <source>arXiv preprint arXiv:2502.15871</source>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>D.</given-names></name> <name><surname>Hartout</surname> <given-names>P.</given-names></name> <name><surname>Pellizzoni</surname> <given-names>P.</given-names></name> <name><surname>Oliver</surname> <given-names>C.</given-names></name> <name><surname>Borgwardt</surname> <given-names>K.</given-names></name></person-group> (<year>2024</year>). <article-title>Endowing protein language models with structural knowledge</article-title>. <source>arXiv preprint arXiv:2401.14819</source>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Coutinho</surname> <given-names>F. H.</given-names></name> <name><surname>Zaragoza-Solas</surname> <given-names>A.</given-names></name> <name><surname>L&#x000F3;pez-P&#x000E9;rez</surname> <given-names>M.</given-names></name> <name><surname>Barylski</surname> <given-names>J.</given-names></name> <name><surname>Zielezinski</surname> <given-names>A.</given-names></name> <name><surname>Dutilh</surname> <given-names>B. E.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>RaFAH: host prediction for viruses of bacteria and archaea based on protein content</article-title>. <source>Patterns</source> <volume>2</volume>:<fpage>100274</fpage>. <pub-id pub-id-type="doi">10.1016/j.patter.2021.100274</pub-id><pub-id pub-id-type="pmid">34286299</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Davis</surname> <given-names>J.</given-names></name> <name><surname>Goadrich</surname> <given-names>M.</given-names></name></person-group> (<year>2006</year>). <article-title>&#x0201C;The relationship between precision-recall and roc curves,&#x0201D;</article-title> in <source>Proceedings of the 23rd International Conference on Machine Learning</source>, 233&#x02013;240. <pub-id pub-id-type="doi">10.1145/1143844.1143874</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Elnaggar</surname> <given-names>A.</given-names></name> <name><surname>Heinzinger</surname> <given-names>M.</given-names></name> <name><surname>Dallago</surname> <given-names>C.</given-names></name> <name><surname>Rehawi</surname> <given-names>G.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Jones</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Prottrans: toward understanding the language of life through self-supervised learning</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>44</volume>, <fpage>7112</fpage>&#x02013;<lpage>7127</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2021.3095381</pub-id><pub-id pub-id-type="pmid">34232869</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fawcett</surname> <given-names>T.</given-names></name></person-group> (<year>2006</year>). <article-title>An introduction to roc analysis</article-title>. <source>Pattern Recognit. Lett</source>. <volume>27</volume>, <fpage>861</fpage>&#x02013;<lpage>874</lpage>. <pub-id pub-id-type="doi">10.1016/j.patrec.2005.10.010</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Galiez</surname> <given-names>C.</given-names></name> <name><surname>Siebert</surname> <given-names>M.</given-names></name> <name><surname>Enault</surname> <given-names>F.</given-names></name> <name><surname>Vincent</surname> <given-names>J.</given-names></name> <name><surname>S&#x000F6;ding</surname> <given-names>J.</given-names></name></person-group> (<year>2017</year>). <article-title>Wish: who is the host? Predicting prokaryotic hosts from metagenomic phage contigs</article-title>. <source>Bioinformatics</source> <volume>33</volume>, <fpage>3113</fpage>&#x02013;<lpage>3114</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx383</pub-id><pub-id pub-id-type="pmid">28957499</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gu</surname> <given-names>Y.</given-names></name> <name><surname>Tinn</surname> <given-names>R.</given-names></name> <name><surname>Cheng</surname> <given-names>H.</given-names></name> <name><surname>Lucas</surname> <given-names>M.</given-names></name> <name><surname>Usuyama</surname> <given-names>N.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Domain-specific language model pretraining for biomedical natural language processing</article-title>. <source>ACM Trans. Comput. Healthcare</source> <volume>3</volume>, <fpage>1</fpage>&#x02013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1145/3458754</pub-id></citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>J.</given-names></name> <name><surname>Ling</surname> <given-names>C. X.</given-names></name></person-group> (<year>2005</year>). <article-title>Using auc and accuracy in evaluating learning algorithms</article-title>. <source>IEEE Trans. Knowl. Data Eng</source>. <volume>17</volume>, <fpage>299</fpage>&#x02013;<lpage>310</lpage>. <pub-id pub-id-type="doi">10.1109/TKDE.2005.50</pub-id></citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Leite</surname> <given-names>D. M. C.</given-names></name> <name><surname>Brochet</surname> <given-names>X.</given-names></name> <name><surname>Resch</surname> <given-names>G.</given-names></name> <name><surname>Que</surname> <given-names>Y.-A.</given-names></name> <name><surname>Neves</surname> <given-names>A.</given-names></name> <name><surname>Pe na-Reyes</surname> <given-names>C.</given-names></name></person-group> (<year>2018</year>). <article-title>Computational prediction of inter-species relationships through omics data analysis and machine learning</article-title>. <source>BMC Bioinform</source>. <volume>19</volume>, <fpage>151</fpage>&#x02013;<lpage>159</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-018-2388-7</pub-id><pub-id pub-id-type="pmid">30453987</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>M.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>F.</given-names></name> <name><surname>Zhao</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>A deep learning-based method for identification of bacteriophage-host interaction</article-title>. <source>IEEE/ACM Trans. Computat. Biol. Bioinform</source>. <volume>18</volume>, <fpage>1801</fpage>&#x02013;<lpage>1810</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2020.3017386</pub-id><pub-id pub-id-type="pmid">32813660</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>D.</given-names></name> <name><surname>Hu</surname> <given-names>X.</given-names></name> <name><surname>He</surname> <given-names>T.</given-names></name> <name><surname>Jiang</surname> <given-names>X.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Virus-host association prediction by using kernelized logistic matrix factorization on heterogeneous networks,&#x0201D;</article-title> in <source>2018 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>108</fpage>&#x02013;<lpage>113</lpage>. <pub-id pub-id-type="doi">10.1109/BIBM.2018.8621214</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>D.</given-names></name> <name><surname>Ma</surname> <given-names>Y.</given-names></name> <name><surname>Jiang</surname> <given-names>X.</given-names></name> <name><surname>He</surname> <given-names>T.</given-names></name></person-group> (<year>2019</year>). <article-title>Predicting virus-host association by kernelized logistic matrix factorization and similarity network fusion</article-title>. <source>BMC Bioinform</source>. <volume>20</volume>, <fpage>1</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-019-3082-0</pub-id><pub-id pub-id-type="pmid">31787095</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Murray</surname> <given-names>C. J.</given-names></name> <name><surname>Ikuta</surname> <given-names>K. S.</given-names></name> <name><surname>Sharara</surname> <given-names>F.</given-names></name> <name><surname>Swetschinski</surname> <given-names>L.</given-names></name> <name><surname>Aguilar</surname> <given-names>G. R.</given-names></name> <name><surname>Gray</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Global burden of bacterial antimicrobial resistance in 2019: a systematic analysis</article-title>. <source>Lancet</source> <volume>399</volume>, <fpage>629</fpage>&#x02013;<lpage>655</lpage>. <pub-id pub-id-type="doi">10.1016/S0140-6736(21)02724-0</pub-id><pub-id pub-id-type="pmid">35065702</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pan</surname> <given-names>J.</given-names></name> <name><surname>You</surname> <given-names>Z.</given-names></name> <name><surname>You</surname> <given-names>W.</given-names></name> <name><surname>Zhao</surname> <given-names>T.</given-names></name> <name><surname>Feng</surname> <given-names>C.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>A microbial knowledge graph-based deep learning model for predicting candidate microbes for target hosts</article-title>. <source>Brief. Bioinform</source>. <volume>25</volume>:<fpage>bbae119</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbae119</pub-id><pub-id pub-id-type="pmid">38555472</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pearce</surname> <given-names>R.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name></person-group> (<year>2021</year>). <article-title>Deep learning techniques have significantly impacted protein structure prediction and protein design</article-title>. <source>Curr. Opin. Struct. Biol</source>. <volume>68</volume>, <fpage>194</fpage>&#x02013;<lpage>207</lpage>. <pub-id pub-id-type="doi">10.1016/j.sbi.2021.01.007</pub-id><pub-id pub-id-type="pmid">33639355</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saito</surname> <given-names>T.</given-names></name> <name><surname>Rehmsmeier</surname> <given-names>M.</given-names></name></person-group> (<year>2015</year>). <article-title>The precision-recall plot is more informative than the roc plot when evaluating binary classifiers on imbalanced datasets</article-title>. <source>PLoS ONE</source> <volume>10</volume>:<fpage>e0118432</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0118432</pub-id><pub-id pub-id-type="pmid">25738806</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shazeer</surname> <given-names>N.</given-names></name> <name><surname>Mirhoseini</surname> <given-names>A.</given-names></name> <name><surname>Maziarz</surname> <given-names>K.</given-names></name> <name><surname>Davis</surname> <given-names>A.</given-names></name> <name><surname>Le</surname> <given-names>Q.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Outrageously large neural networks: the sparsely-gated mixture-of-experts layer</article-title>. <source>arXiv preprint arXiv:1701.06538</source>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Villarroel</surname> <given-names>J.</given-names></name> <name><surname>Kleinheinz</surname> <given-names>K. A.</given-names></name> <name><surname>Jurtz</surname> <given-names>V. I.</given-names></name> <name><surname>Zschach</surname> <given-names>H.</given-names></name> <name><surname>Lund</surname> <given-names>O.</given-names></name> <name><surname>Nielsen</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Hostphinder: a phage host prediction tool</article-title>. <source>Viruses</source> <volume>8</volume>:<fpage>116</fpage>. <pub-id pub-id-type="doi">10.3390/v8050116</pub-id><pub-id pub-id-type="pmid">27153081</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Walsh</surname> <given-names>T. R.</given-names></name> <name><surname>Gales</surname> <given-names>A. C.</given-names></name> <name><surname>Laxminarayan</surname> <given-names>R.</given-names></name> <name><surname>Dodd</surname> <given-names>P. C.</given-names></name></person-group> (<year>2023</year>). <article-title>Antimicrobial resistance: addressing a global threat to humanity</article-title>. <source>PLoS Med</source>. <volume>20</volume>:<fpage>e1004264</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pmed.1004264</pub-id><pub-id pub-id-type="pmid">37399216</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Ren</surname> <given-names>J.</given-names></name> <name><surname>Tang</surname> <given-names>K.</given-names></name> <name><surname>Dart</surname> <given-names>E.</given-names></name> <name><surname>Ignacio-Espinoza</surname> <given-names>J. C.</given-names></name> <name><surname>Fuhrman</surname> <given-names>J. A.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>A network-based integrated framework for predicting virus-prokaryote interactions</article-title>. <source>NAR Genom. Bioinform</source>. <volume>2</volume>:<fpage>lqaa044</fpage>. <pub-id pub-id-type="doi">10.1093/nargab/lqaa044</pub-id><pub-id pub-id-type="pmid">32626849</pub-id></citation></ref>
</ref-list>
</back>
</article>