<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1616880</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2025.1616880</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>BioSemAF-BiLSTM: a protein sequence feature extraction framework based on semantic and evolutionary information</article-title>
<alt-title alt-title-type="left-running-head">Zhang and Wang</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2025.1616880">10.3389/fgene.2025.1616880</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhang</surname>
<given-names>Zihan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2971296/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Yixuan</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Haide College, Ocean University of China</institution>, <addr-line>Qingdao</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Medicine, Nankai University</institution>, <addr-line>Tianjin</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/30191/overview">Alexandre V. Morozov</ext-link>, Rutgers, The State University of New Jersey, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/47587/overview">Yuriy L. Orlov</ext-link>, I. M. Sechenov First Moscow State Medical University, Russia</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1803617/overview">Showkat A. Dar</ext-link>, National Institutes of Health (NIH), United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Zihan Zhang, <email>zzh7488@stu.ouc.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>17</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1616880</elocation-id>
<history>
<date date-type="received">
<day>30</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>01</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Zhang and Wang.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Zhang and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>S-sulfenylation is a critical post-translational modification that plays an important role in regulating protein function, redox signaling, and maintaining cellular homeostasis. Accurate identification of S-sulfenylation sites is essential for understanding its biological significance and relevance to disease. However, the exclusive detection of S-sulfenylation sites through experimental methods remains challenging, as these approaches are often time-consuming and costly. Motivated by this issue, the present work proposed a deep learning-based computational framework, named BioSemAF-BiLSTM, which integrated evolutionary and semantic features to improve the prediction performance of S-sulfenylation sites. The framework employed fastText to generate subword-based sequence embeddings that captured local contextual information, and employed position-specific scoring matrices (PSSMs) to extract evolutionary conservation features. Importantly, we also quantitatively evaluated feature sufficiency at the protein sequence level using a sequence compression-based measure approximating Kolmogorov complexity, revealing an 11% information loss rate in predictive modeling using these features. These representations were subsequently fed into a bidirectional long short-term memory (BiLSTM) network to model long-range dependencies, and were further refined via an adaptive feature fusion module to enhance feature interaction. Experimental results on a benchmark dataset demonstrated that the model significantly outperformed conventional machine learning methods and current state-of-the-art deep learning approaches, achieving an accuracy of 89.32% on an independent test. It demonstrated improved sensitivity and specificity, effectively bridging the gap between bioinformatics and deep learning, and offered a robust computational tool for predicting post-translational modification sites.</p>
</abstract>
<kwd-group>
<kwd>adaptive fusion</kwd>
<kwd>bidirectional LSTM neural network</kwd>
<kwd>bioinformation encoding</kwd>
<kwd>information loss</kwd>
<kwd>Kolmogorov complexity</kwd>
<kwd>post-translational modification</kwd>
<kwd>protein sequence embedding</kwd>
<kwd>sulfenylation site</kwd>
</kwd-group>
<counts>
<page-count count="16"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Post-translational modifications (PTMs) serve as key mechanisms for regulating protein function, stability, and interactions (<xref ref-type="bibr" rid="B36">Mann and Jensen, 2003</xref>). Recognized as central processes in biological activities, PTMs encompass diverse types, including phosphorylation, acetylation, methylation, and redox modifications. Among these, redox-related modifications have attracted considerable attention, particularly those involving cysteine residues. The thiol group (-SH) of cysteine residues is highly sensitive to reactive oxygen species (ROS), undergoing a variety of oxidative modifications that are crucial for signal transduction and cellular stress responses (<xref ref-type="bibr" rid="B46">Qian et al., 2024</xref>; <xref ref-type="bibr" rid="B55">Walsh and Jefferis, 2006</xref>).</p>
<p>Among cysteine modifications, S-sulfenylation is distinguished by its dynamic and reversible nature, which endows it with unique biological significance. S-sulfenylation entails the reaction of thiol groups with reactive oxygen species (ROS), such as hydrogen peroxide, leading to the formation of -SOH. This modification modulates protein activity, maintains cellular redox homeostasis, and plays a key role in signal transduction (<xref ref-type="bibr" rid="B43">Paulsen and Carroll, 2013</xref>). S-sulfenylation not only protects cysteine residues from irreversible oxidation but also plays critical roles in various pathological conditions, such as cancer, neurodegenerative disorders, and cardiovascular diseases (<xref ref-type="bibr" rid="B3">Anjo et al., 2024</xref>; <xref ref-type="bibr" rid="B8">Chahla et al., 2024</xref>). Therefore, precise identification and localization of S-sulfenylation sites are essential for elucidating their biological functions and understanding disease-associated mechanisms. This necessity forms the basis for ongoing efforts to develop computational approaches for S-sulfenylation site prediction.</p>
<p>Traditionally, the prediction of S-sulfenylation sites has relied on biological experiments that detect specific cysteine residues undergoing modification through direct or indirect approaches. Early studies predominantly used radioactive isotope-based labeling techniques, in which hydrogen peroxide reacts with cysteine to form thiol intermediates detectable using isotope-labeled chemical probes (<xref ref-type="bibr" rid="B32">Leonard et al., 2009</xref>; <xref ref-type="bibr" rid="B58">Yang et al., 2015</xref>). However, these methods suffered from low sensitivity and often altered protein structures, thereby restricting their broader applicability. With technological advancements, chemical probe-based labeling strategies have emerged as mainstream tools in S-sulfenylation research (<xref ref-type="bibr" rid="B43">Paulsen and Carroll, 2013</xref>). For example, in the early 2000s, biochemists developed disulfide-based probes that selectively bind to S-sulfenylation sites, enabling precise localization via mass spectrometry (MS) analysis (<xref ref-type="bibr" rid="B32">Leonard et al., 2009</xref>; <xref ref-type="bibr" rid="B58">Yang et al., 2015</xref>; <xref ref-type="bibr" rid="B57">Yang et al., 2014</xref>). While these approaches significantly improved sensitivity, challenges associated with sample complexity and quantitative analysis still remained. During the 2010s, advanced quantitative proteomics techniques were incorporated into S-sulfenylation research. For instance, Paulsen et al. (<xref ref-type="bibr" rid="B44">Paulsen et al., 2011</xref>) developed the DYn-2 probe, which selectively captures -SOH modification sites and, when combined with high-resolution MS, facilitates large-scale identification of such modifications. In addition, click chemistry-based biotinylated probes were widely adopted, providing powerful tools for high-throughput detection of modified cysteine residues (<xref ref-type="bibr" rid="B32">Leonard et al., 2009</xref>; <xref ref-type="bibr" rid="B58">Yang et al., 2015</xref>). These traditional biochemical methods have established a strong foundation for S-sulfenylation site research and elucidated the dynamic regulation of modifications under various physiological conditions.</p>
<p>Traditional machine learning approaches have long played a central role in the early development of computational tools for S-sulfenylation site prediction. These methods typically fall under the broader category of protein sequence site prediction&#x2013;an important area of bioinformatics that aims to identify functional residues. Accurate sequence representations are also critical for a variety of related prediction tasks in bioinformatics, such as determining protein secondary structures (<xref ref-type="bibr" rid="B24">Jones, 1999</xref>). In this study, however, we focus on PTM site prediction, where a number of traditional machine learning approaches have been developed. For instance, Chou introduced the concept of pseudo-amino acid composition (PseAAC) to encode protein sequence features, and Qiu et al. combined it with random forest (RF) for phosphorylation site prediction (<xref ref-type="bibr" rid="B10">Chou, 2001</xref>; <xref ref-type="bibr" rid="B48">Qiu et al., 2017</xref>). Subsequently, <xref ref-type="bibr" rid="B7">Caragea et al. (2007)</xref> successfully predicted glycosylation sites using a support vector machine (SVM). <xref ref-type="bibr" rid="B6">Bui et al. (2015)</xref> developed MDD-SOH, one of the earliest and most foundational models in this field. MDD-SOH utilizes sequence composition and physicochemical properties of proteins as features, employing an SVM-based maximum dependency decomposition (MDD) for classification. Similarly, Xu et al. proposed the iSulf-Cys model, which utilized physicochemical and distribution properties of amino acids and employed a support vector machine for classification (<xref ref-type="bibr" rid="B56">Xu et al., 2016</xref>). <xref ref-type="bibr" rid="B26">Ju and Wang. (2018)</xref> introduced the Sulf_FSVM model, combining mRMR (maximum relevance minimum redundancy) features and employing a fuzzy support vector machine as the classifier.</p>
<p>Although machine learning approaches have achieved some success, as discussed above, they have relied heavily on complex feature engineering and prior knowledge, with limited adaptability to large-scale datasets, paving the way for deep learning approaches. With the rapid advancement of deep learning, researchers have recognized its superiority in processing sequential data. Deep learning effectively captures intricate patterns within raw data and uncovers latent dependencies in high-dimensional spaces. Consequently, deep learning has been widely adopted for protein sequence analysis and site prediction. Foundational advances in deep learning (<xref ref-type="bibr" rid="B31">LeCun et al., 2015</xref>) and comprehensive reviews on its applications to molecular and protein modeling (<xref ref-type="bibr" rid="B35">Li et al., 2022</xref>; <xref ref-type="bibr" rid="B47">Qin et al., 2024</xref>; <xref ref-type="bibr" rid="B4">Boadu et al., 2024</xref>) have catalyzed this trend. In recent years, several deep learning-based models have been developed specifically for the prediction of S-sulfenylation sites, each adopting distinct strategies for sequence representation and feature extraction.</p>
<p>
<xref ref-type="bibr" rid="B27">Khan and Pi. (2020)</xref> developed nSSPred, which combines nSegmented optimization with a joint feature encoder and applies 2D convolutional neural networks (2D-CNNs) to predict S-sulfenylation sites. <xref ref-type="bibr" rid="B16">Do et al. (2020)</xref> proposed fastSulf-DNN, extracting bio-subword-level features and employing natural language processing techniques to encode protein sequences for deep neural network (DNN) classification. <xref ref-type="bibr" rid="B41">Ning and Li (2022)</xref> presented DLF-Sul, a multi-module framework that integrates binary encoding, BLOSUM62, and amino acid indices, followed by BiLSTM for sequence modeling, and further applies multi-head self-attention and CNNs for feature refinement before classification via fully connected layers. However, these methods share key limitations (<xref ref-type="bibr" rid="B16">Do et al., 2020</xref>): 1. A strong dependence on expert-crafted features restricts the comprehensive capture of biological information; 2. Limited capacity for semantic understanding hampers accurate identification of amino acid functions and interdependencies; 3. Weak adaptability to heterogeneous features constrains dynamic, task-aware feature weighting; 4. A focus on local patterns undermines the modeling of long-range dependencies in sequences. Together, these issues limit the predictive performance of existing approaches for S-sulfenylation sites.</p>
<p>Beyond architectural innovations, recent studies have also highlighted the importance of evaluating how much biological information is preserved in a given sequence representation, for example, through compression- and complexity-based measures (<xref ref-type="bibr" rid="B42">Orlov and Orlov, 2023</xref>). Inspired by these insights, this work further incorporates a Kolmogorov-complexity&#x2013;inspired, compression-based evaluation to quantify feature sufficiency and to guide resampling strategy selection, complementing the predictive modeling rather than replacing it.</p>
<p>We systematically evaluated existing computational tools for S-sulfenylation site prediction. To address identified limitations, a novel predictive framework was proposed by integrating state-of-the-art algorithms. A model named BioSemAF-BiLSTM (Biological Semantic Adaptive Fusion Bidirectional Long Short-Term Memory) was designed to more accurately capture essential features within protein sequences. It combined bioinformatics-derived sequence features and semantic representations of sequence subwords, employing a BiLSTM to extract global contextual information. Position-Specific Scoring Matrices (PSSMs) encode evolutionary information, while fastText-generated embeddings capture local semantics and biologically relevant residue patterns. To enhance feature representation, an attention-based Adaptive Feature Fusion (AF) module was used to integrate fastText embeddings with PSSM-derived numerical features, assigning dynamic weights to emphasize the most informative components. The fused features were then passed through fully connected layers for classification. Computational evaluation demonstrated that the proposed method significantly outperformed existing approaches in terms of sensitivity, specificity, and overall accuracy, highlighting the effectiveness of integrating biological embeddings, deep neural networks, and attention mechanisms for S-sulfenylation site prediction.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<p>The detailed architecture of BioSemAF-BiLSTM is illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>, comprising four main components: Word Embedding, Bio-information Encoding, Bi-LSTM, and the Adaptive Feature Fusion Module. A fully connected dense layer is appended at the end to perform binary classification. This architecture is designed to leverage the complementary strengths of all modules, thereby enhancing prediction accuracy and model robustness.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Flowchart of BioSemAF-BiLSTM. <bold>(a)</bold> Illustration of two types of sequence-based feature representations: semantic embeddings and bioinformational features. <bold>(b)</bold> Overview of the complete prediction pipeline: vectors from PSSM and fastText embeddings are processed by Bi-LSTMs, fused via the Adaptive Feature Fusion Module, and classified by a dense layer to predict S-sulfenylation sites. The illustration in this panel uses PSSM features as an example input. <bold>(c)</bold> Internal architecture of the Bi-LSTM module, where sequence representations are transformed into contextualized hidden features for downstream classification. <bold>(d)</bold> Workflow of the Adaptively Feature Fusion Module, which integrates semantic and evolutionary features into a unified representation; final site prediction is performed after this module as shown in panel <bold>(b)</bold>. Abbreviations: PSSM, Position-Specific Scoring Matrix; Bi-LSTM, Bidirectional Long Short-Term Memory.</p>
</caption>
<graphic xlink:href="fgene-16-1616880-g001.tif">
<alt-text content-type="machine-generated">Diagram illustrating a bioinformatics model for sulfenylation site prediction. Panel a shows word embedding and bioinformatic encoding using a 3-gram split and PSSM. Panel b details the feature fusion module with BiLSTM processing. Panel c depicts the BiLSTM architecture with LSTM layers combining sequences. Panel d explains the attention mechanism with self and cross-attention leading to final prediction. The model predicts sulfenylation and non-sulfenylation sites.</alt-text>
</graphic>
</fig>
<sec id="s2-1">
<title>2.1 Benchmark dataset</title>
<p>We utilized the dataset developed in the iSulf-Cys framework (<xref ref-type="bibr" rid="B56">Xu et al., 2016</xref>), which is based on the work of <xref ref-type="bibr" rid="B57">Yang et al. (2014)</xref>. The dataset comprises 7,124 non-sulfenylation cysteine sites and 1,045 sulfenylation cysteine sites. In total, the data are derived from 778 protein sequences and include 1,105 sulfenylation cysteine sites. The corresponding protein sequences are available at <ext-link ext-link-type="uri" xlink:href="https://ndownloader.figstatic.com/files/5004001">https://ndownloader.figstatic.com/files/5004001</ext-link>.</p>
<p>For each cysteine site subjected to analysis, Xu et al. constructed peptide samples by extracting the 10 amino acids upstream and downstream, resulting in a sequence window of size 21 centered on the target cysteine. If the surrounding sequence contained fewer than 10 residues, the placeholder residue X was used to pad the sequence. The representation of a sample is as <xref ref-type="disp-formula" rid="e1">Equation 1</xref>:<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mtext>Peptide</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>9</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2026;</mml:mo>
<mml:mi>C</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2026;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>where <inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mn>20</mml:mn>
<mml:mtext>&#x2009;amino&#x2009;acids</mml:mtext>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
<mml:mo>&#x222a;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> range from &#x2212;10 to 10, representing the one-dimensional coordinate relative to the central cysteine. A peptide segment with a central cysteine that is a sulfenylation site is defined as a positive sample, while all others are classified as negative samples. To minimize redundancy and reduce homology bias, peptide segments sharing more than 40% sequence similarity were removed from the dataset. Ultimately, 145 positive and 268 negative samples were randomly selected as the test set, while the remaining 900 positive and 6,858 negative samples were allocated to the training set. A summary of the dataset is presented in <xref ref-type="table" rid="T1">Table 1</xref>. This dataset has been widely adopted in sulfenylation site prediction studies (<xref ref-type="bibr" rid="B26">Ju and Wang, 2018</xref>; <xref ref-type="bibr" rid="B27">Khan and Pi, 2020</xref>; <xref ref-type="bibr" rid="B16">Do et al., 2020</xref>), and it exhibits a pronounced imbalance between positive and negative classes.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>The number of positive and negative samples in training and independent test dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Data</th>
<th align="center">Positive</th>
<th align="center">Negative</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Training Set</td>
<td align="center">900</td>
<td align="center">6,856</td>
</tr>
<tr>
<td align="left">Test Set</td>
<td align="center">145</td>
<td align="center">268</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-2">
<title>2.2 Feature extraction</title>
<p>In this study, two strategies were considered for processing peptide segment features. First, bioinformatics-based features were introduced by encoding sequences using the Position-Specific Scoring Matrix (PSSM). Second, natural language processing techniques were applied to generate word embeddings from protein sequences, thereby enabling the model to fully capture the latent semantic information embedded within the sequences. This approach has been validated as effective in previous study (<xref ref-type="bibr" rid="B16">Do et al., 2020</xref>).</p>
<sec id="s2-2-1">
<title>2.2.1 PSSM</title>
<p>PSSM, or Position-Specific Scoring Matrix, is a bioinformatic representation used to characterize evolutionary conservation in biological sequences. Evolutionarily conserved amino acid residues are often associated with functional motifs that play critical roles in protein functions, including post-translational modifications (PTMs) (<xref ref-type="bibr" rid="B33">Li and Dohlman, 2022</xref>). Therefore, PSSM serves as a valuable bioinformatic feature for representing peptide segments in prediction models.</p>
<p>PSSM describes and quantifies the distribution of amino acids (or nucleotides) at each position within a sequence, reflecting both evolutionary conservation and mutational tendencies. It has been widely applied across a range of site prediction tasks. For instance, <xref ref-type="bibr" rid="B59">Yuan et al. (2021)</xref> introduced PSSM as a key feature in characterizing protein&#x2013;protein interaction (PPI) nodes. In PTM site prediction, PSSM has also shown substantial utility. PSSM-Suc employs PSSM to encode succinylation sites for predictive modeling (<xref ref-type="bibr" rid="B13">Dehzangi et al., 2017</xref>), whereas PSSM-Sumo utilizes a variant called PsePSSM to encode SUMOylation sites (<xref ref-type="bibr" rid="B28">Khan et al., 2024</xref>). Both approaches demonstrated strong predictive performance, further validating the effectiveness of PSSM-based representations.</p>
<p>A standard PSSM has a matrix dimension of 20 rows&#x2014;each corresponding to one of the 20 standard amino acids&#x2014;and a number of columns equal to the length of the peptide or sequence. The canonical format of the PSSM is defined as <xref ref-type="disp-formula" rid="e2">Equation 2</xref>:<disp-formula id="e2">
<mml:math id="m4">
<mml:mrow>
<mml:mtext>PSSM</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>11</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>12</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mo>&#x2026;</mml:mo>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>21</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>22</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mo>&#x2026;</mml:mo>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mo>&#x22f1;</mml:mo>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>20,1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>20,2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mo>&#x2026;</mml:mo>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>20</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>Where <inline-formula id="inf3">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the probability of the <italic>i</italic>th amino acid mutating to another amino acid at the <italic>j</italic>th position, with <inline-formula id="inf4">
<mml:math id="m6">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> ranging from 1 to 20 (representing the amino acid types) and <inline-formula id="inf5">
<mml:math id="m7">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> ranging from 1 to L (representing the sequence length, with L being the total length of the sequence).</p>
<p>In this study, PSSMs were generated using Position-Specific Iterated BLAST (PSI-BLAST; <ext-link ext-link-type="uri" xlink:href="https://blast.ncbi.nlm.nih.gov/Blast.cgi">https://blast.ncbi.nlm.nih.gov/Blast.cgi</ext-link>) against the NCBI non-redundant (NR) protein database (<xref ref-type="bibr" rid="B2">Altschul et al., 1997</xref>). Following common practice (<xref ref-type="bibr" rid="B40">Nie et al., 2017</xref>), we used three iterations with an expectation value (E-value) of 0.001. It should be noted that PSI-BLAST was used solely to perform iterative alignment for PSSM generation, not for <italic>de novo</italic> homologous sequence search. Other multiple-sequence-alignment tools such as HHblits (<xref ref-type="bibr" rid="B49">Remmert et al., 2011</xref>) or Clustal Omega (<xref ref-type="bibr" rid="B52">Sievers and Higgins, 2014</xref>) could also be applied for this purpose, but PSI-BLAST remains widely used in related work.</p>
</sec>
</sec>
<sec id="s2-3">
<title>2.2.2 Word embedding</title>
<p>Language is one of the most intuitive forms of representation, and humans have long sought to describe the natural world through linguistic expressions. By assigning specific letters to amino acids, protein sequences can be directly encoded as character strings. These sequences contain rich semantic content that can be mined using appropriate natural language processing (NLP) techniques. This section focuses on local protein-related tasks, particularly on predicting residue-level features from sequence data. Window-based models have proven effective in various PTM site prediction tasks. For instance, <xref ref-type="bibr" rid="B5">Brandes et al. (2015)</xref> introduced a framework that predicts PTM sites by segmenting protein sequences into fixed-size windows, highlighting the potential of semantic features in protein feature engineering.</p>
<p>With the continuous advancement of NLP technologies (<xref ref-type="bibr" rid="B54">Vaswani et al., 2017</xref>; <xref ref-type="bibr" rid="B15">Devlin et al., 2018</xref>), Word2Vec has emerged as a major breakthrough in the field (<xref ref-type="bibr" rid="B37">Mikolov et al., 2013a</xref>). By leveraging contextual information, it generates continuous low-dimensional vector representations for words (or characters), capturing latent semantic relationships embedded in language. This advancement has inspired novel approaches to protein sequence analysis. FastText, an enhanced version of Word2Vec, further expands upon this idea (<xref ref-type="bibr" rid="B25">Joulin et al., 2016</xref>). Compared to Word2Vec, fastText introduces subword modeling, which effectively captures internal morphological and structural features of words. This is particularly suitable for protein sequence modeling, as such sequences frequently contain variable and localized motifs. FastText not only captures local <italic>n</italic>-gram-like information during training but also addresses the Out-of-Vocabulary (OOV) problem&#x2014;a critical advantage when analyzing complex biological sequences.</p>
<p>The core idea of fastText is to enhance traditional word embeddings by incorporating subword modeling. In contrast to the Word2Vec model, which assigns a distinct vector to each word, fastText represents a word as a combination of its constituent subwords, thereby capturing subtle internal variations within words. For example, the word <italic>protein</italic> can be split into subwords such as <italic>pro</italic>, <italic>ote</italic>, and <italic>ein</italic>, all of which participate in the model&#x2019;s training. By encoding and embedding these subwords, fastText learns more nuanced semantic representations, making it especially effective in handling diverse and irregular protein sequences.</p>
<p>From a mathematical standpoint, fastText is built upon the Skip-gram architecture, where the objective is to maximize the conditional probability of predicting a target word given its surrounding context (<xref ref-type="bibr" rid="B38">Mikolov et al., 2013b</xref>). Given a training corpus composed of word sequences <inline-formula id="inf6">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the objective function of fastText is to learn embeddings for both words and their subwords by minimizing the following loss function (<xref ref-type="disp-formula" rid="e3">Equation 3</xref>):<disp-formula id="e3">
<mml:math id="m9">
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>where <inline-formula id="inf7">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the current word, <inline-formula id="inf8">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the context word, <inline-formula id="inf9">
<mml:math id="m12">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the size of the context window, and <inline-formula id="inf10">
<mml:math id="m13">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the conditional probability of predicting the target word <inline-formula id="inf11">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. In fastText, <inline-formula id="inf12">
<mml:math id="m15">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is no longer a simple word&#x2019;s conditional probability but a conditional probability of the subwords. Specifically, for each word <inline-formula id="inf13">
<mml:math id="m16">
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, it is split into several subwords <inline-formula id="inf14">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and the probability of the target word is calculated as <xref ref-type="disp-formula" rid="e4">Equation 4</xref>:<disp-formula id="e4">
<mml:math id="m18">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mo>&#x220f;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:munder>
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>where <inline-formula id="inf15">
<mml:math id="m19">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the set of subwords for the word <inline-formula id="inf16">
<mml:math id="m20">
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf17">
<mml:math id="m21">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the conditional probability of subword <inline-formula id="inf18">
<mml:math id="m22">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> given the context word <inline-formula id="inf19">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>In practical applications, fastText represents the semantic content of a word through both its own embedding and the embeddings of its constituent subwords. For protein sequences, dipeptides or tripeptides derived from amino acid residues can be regarded as subwords. Using this representation, fastText can convert short fragments of protein sequences&#x2014;such as amino acid pairs or triplets&#x2014;into vector embeddings, capturing the underlying contextual relationships between substructures. In this study, each peptide sequence is decomposed into 3-g as subwords. For a sequence consisting of <inline-formula id="inf20">
<mml:math id="m24">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> amino acids, this yields <inline-formula id="inf21">
<mml:math id="m25">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> trigrams, and hence, generates <inline-formula id="inf22">
<mml:math id="m26">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> corresponding word vectors that embed the sequence&#x2019;s implicit order and semantic content. This process constitutes the fastText-based protein sequence embedding used in this study.</p>
</sec>
<sec id="s2-4">
<title>2.3 Resampling</title>
<p>As described in <xref ref-type="sec" rid="s2-1">Section 2.1</xref>, the dataset used in our study exhibits a pronounced imbalance between positive and negative samples, necessitating the application of resampling techniques. Common resampling strategies include down-sampling and up-sampling. Down-sampling involves reducing the number of majority class instances to achieve a balanced class distribution, whereas up-sampling increases the number of minority class samples to the same effect. To minimize information loss from the sample space and retain protein residue-level features, up-sampling is prioritized. Synthetic Minority Over-sampling Technique (SMOTE) is one of the most widely used up-sampling methods (<xref ref-type="bibr" rid="B9">Chawla et al., 2002</xref>). Numerous variants of SMOTE have been proposed in subsequent studies, and this study adopts Support Vector Machine Synthetic Minority Over-sampling Technique (SVMSMOTE) for up-sampling (<xref ref-type="bibr" rid="B14">Demidova and Klyueva, 2017</xref>). The resampling procedure is outlined as follows.<list list-type="simple">
<list-item>
<p>1. Train the SVM model: First, a support vector machine (SVM) classifier is trained using minority class samples to determine the decision boundary. By maximizing the margin between classes, SVM effectively learns to delineate the optimal separation boundary for the training data.</p>
</list-item>
<list-item>
<p>2. Identify support vectors: Support vectors are those training samples situated closest to the decision boundary. These points are critical in defining the classification margin. SVMSMOTE leverages these support vectors to guide the generation of new synthetic instances.</p>
</list-item>
<list-item>
<p>3. Generate synthetic samples: For each minority class support vector, SVMSMOTE generates synthetic samples by interpolating between the support vector and its nearest neighbors. This process is similar to SMOTE, and the interpolation formula is given as <xref ref-type="disp-formula" rid="e5">Equation 5</xref>:</p>
</list-item>
</list>
<disp-formula id="e5">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>new</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>sv</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>neigh</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>sv</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>where <inline-formula id="inf23">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>new</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the generated synthetic sample, <inline-formula id="inf24">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>sv</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the current support vector, <inline-formula id="inf25">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>neigh</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is its nearest minority class neighbor, and <inline-formula id="inf26">
<mml:math id="m31">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a random number in the range [0, 1] that determines the interpolation proportion.<list list-type="simple">
<list-item>
<p>4. Generate the required number of synthetic samples: Based on the method above, SVMSMOTE generates enough synthetic samples to achieve a balanced class distribution in the training dataset.</p>
</list-item>
</list>
</p>
<p>For a schematic illustration of the SMOTE and SVMSMOTE procedures, please refer to <xref ref-type="sec" rid="s11">Supplementary Figure S1</xref>.</p>
</sec>
<sec id="s2-5">
<title>2.4 Bi-LSTM</title>
<p>Considering the sequential nature of the two feature spaces used for protein representation in this study, a Bi-LSTM-based approach was adopted. The strength of LSTM in handling protein sequences lies in its capability to effectively capture long-range dependencies. In protein prediction tasks, especially in the identification of post-translational modification (PTM) sites, long-range contextual dependencies are critical (<xref ref-type="bibr" rid="B51">Shi et al., 2025</xref>; <xref ref-type="bibr" rid="B54">Vaswani et al., 2017</xref>) as relevant information in protein sequences often spans beyond local neighborhoods. Accurate functional prediction therefore, relies on a model&#x2019;s ability to extract such extended context. Traditional recurrent neural networks (RNNs) struggle in these scenarios due to challenges such as vanishing and exploding gradients, which hinder learning across long temporal sequences. In contrast, Bi-LSTM incorporates both forward and backward contextual information along the sequence, allowing it to capture bidirectional dependencies that are essential for accurate PTM site prediction.</p>
<p>A standard Bi-LSTM consists of a forward LSTM and a backward LSTM. Specifically, the LSTM architecture introduces memory cells that determine whether to retain or discard information from past inputs. These memory cells are regulated by three gates: the input gate, forget gate, and output gate. The input gate controls how much of the current input is stored in the memory cell, the forget gate decides which parts of the previous memory to discard, and the output gate determines the information to be passed to the next layer in the network.</p>
<p>In this study, Bi-LSTM is employed as the core model to process protein sequence data. By combining the forward and backward passes, Bi-LSTM effectively captures bidirectional contextual information within the protein sequence, thereby enhancing the accuracy of sequence-based predictions.</p>
<p>Let f denote the forward LSTM process and b denote the backward LSTM process. For each time step <inline-formula id="inf27">
<mml:math id="m32">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the forward LSTM computes the following <xref ref-type="disp-formula" rid="e6">Equations 6</xref>&#x2010;<xref ref-type="disp-formula" rid="e10">10</xref>:<disp-formula id="e6">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="e7">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
<disp-formula id="e8">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
<disp-formula id="e9">
<mml:math id="m36">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>tanh</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
<disp-formula id="e10">
<mml:math id="m37">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>tanh</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
<p>Where <inline-formula id="inf28">
<mml:math id="m38">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the sigmoid activation function, <inline-formula id="inf29">
<mml:math id="m39">
<mml:mrow>
<mml:mi>tanh</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the hyperbolic tangent function, <inline-formula id="inf30">
<mml:math id="m40">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the current input, <inline-formula id="inf31">
<mml:math id="m41">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the previous time step&#x2019;s hidden state, and <inline-formula id="inf32">
<mml:math id="m42">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the previous time step&#x2019;s memory state.</p>
<p>The backward LSTM operates similarly to the forward LSTM but processes the sequence in reverse order. In the standard Bi-LSTM model, the outputs of both the forward and backward LSTMs are combined to calculate the final output for each time step. However, in this study, the initial hidden state of the backward LSTM is modified. Specifically, the initial hidden state of the backward LSTM is set as a weighted sum of the forward LSTM&#x2019;s initial hidden state, with the weights being trainable. This modification is expressed as <xref ref-type="disp-formula" rid="e11">Equation 11</xref>:<disp-formula id="e11">
<mml:math id="m43">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn mathvariant="bold">0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>
</p>
<p>Where <inline-formula id="inf33">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the weight for the <inline-formula id="inf34">
<mml:math id="m45">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th step of the forward LSTM, and <inline-formula id="inf35">
<mml:math id="m46">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the hidden state at the <inline-formula id="inf36">
<mml:math id="m47">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th step. This improvement enables the backward pass to integrate information learned from the forward pass, enhancing backward propagation.</p>
<p>Additionally, the model incorporates residual connections, which can be expressed as <xref ref-type="disp-formula" rid="e12">Equation 12</xref>:<disp-formula id="e12">
<mml:math id="m48">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
</p>
<p>Where <inline-formula id="inf37">
<mml:math id="m49">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a hyperparameter that adjusts the weighting between the current hidden state and the previous hidden state. The residual connection retains information from the previous time step&#x2019;s hidden state, enhancing the model&#x2019;s ability to capture long-range dependencies.</p>
</sec>
<sec id="s2-6">
<title>2.5 Adaptively feature fusion module</title>
<p>Feature extraction in this study is based on both bioinformatics features and semantic features derived from protein sequence fragments. These two types of features emphasize different aspects of information representation: bioinformatics features highlight the structural and functional features of the sequence, whereas semantic features capture contextual relationships at a natural language level. However, due to potential redundancy or inconsistency between these two modalities, an Adaptive Feature Fusion Module is proposed to integrate and utilize them more effectively. This module is designed with two primary objectives: first, to accurately identify and eliminate redundant information to prevent irrelevant or noisy features from impairing model performance; and second, to enhance discriminative features that are crucial to the target task, thereby improving both the learning efficiency and predictive accuracy of the model. The Adaptive Feature Fusion Module not only manages the allocation of importance across different feature types, but also dynamically adjusts the fusion strategy to accommodate variations across different tasks or data distributions.</p>
<p>This study proposed a fusion module leveraging both self-attention and cross-attention mechanisms. The goal of this module is to dynamically assign feature weights using these attention mechanisms, thereby extracting salient information and achieving efficient fusion. A detailed introduction to this module is provided below.</p>
<p>Let the features to be fused be <inline-formula id="inf38">
<mml:math id="m50">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> (size <inline-formula id="inf39">
<mml:math id="m51">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) and <inline-formula id="inf40">
<mml:math id="m52">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> (size <inline-formula id="inf41">
<mml:math id="m53">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>). Since the dimensions of <inline-formula id="inf42">
<mml:math id="m54">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf43">
<mml:math id="m55">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> may differ, projection layers were initially established to map both features into a common dimensionality <inline-formula id="inf44">
<mml:math id="m56">
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. This is done via linear projection followed by a non-linear activation function <inline-formula id="inf45">
<mml:math id="m57">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (<xref ref-type="disp-formula" rid="e13">Equations 13</xref>, <xref ref-type="disp-formula" rid="e14">14</xref>):<disp-formula id="e13">
<mml:math id="m58">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
<disp-formula id="e14">
<mml:math id="m59">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>where <inline-formula id="inf46">
<mml:math id="m60">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf47">
<mml:math id="m61">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Next, self-attention is applied to both <inline-formula id="inf48">
<mml:math id="m62">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf49">
<mml:math id="m63">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. Self-attention is a mechanism used to compute the relevance of each element in the sequence to every other element, allowing it to capture global context. It dynamically assigns higher weights to important elements in the sequence, thus improving the model&#x2019;s ability to understand complex dependencies. The calculation of Self-attention is as <xref ref-type="disp-formula" rid="e15">Equation 15</xref>:<disp-formula id="e15">
<mml:math id="m64">
<mml:mrow>
<mml:mtext>Self</mml:mtext>
<mml:mo>-</mml:mo>
<mml:mtext>Attention</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>softmax</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>
</p>
<p>where <inline-formula id="inf50">
<mml:math id="m65">
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf51">
<mml:math id="m66">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf52">
<mml:math id="m67">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are linear transformations of <inline-formula id="inf53">
<mml:math id="m68">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf54">
<mml:math id="m69">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. Specifically, self-attention is applied to extract their respective key feature patterns (<xref ref-type="disp-formula" rid="e16">Equations 16</xref>, <xref ref-type="disp-formula" rid="e17">17</xref>):<disp-formula id="e16">
<mml:math id="m70">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Self</mml:mtext>
<mml:mo>-</mml:mo>
<mml:mtext>Attention</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
<disp-formula id="e17">
<mml:math id="m71">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Self</mml:mtext>
<mml:mo>-</mml:mo>
<mml:mtext>Attention</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>
</p>
<p>Next, <inline-formula id="inf55">
<mml:math id="m72">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf56">
<mml:math id="m73">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> are combined using the cross-attention mechanism to model the interaction between the two types of features. Cross-attention extracts important information by focusing on the most relevant parts of one feature set relative to another. By doing so, it effectively fuses features from different sources and highlights key associations between them. The computation of Cross-attention is similar to Self-attention, but the design of <inline-formula id="inf57">
<mml:math id="m74">
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf58">
<mml:math id="m75">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf59">
<mml:math id="m76">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> differs (<xref ref-type="disp-formula" rid="e18">Equation 18</xref>).<disp-formula id="e18">
<mml:math id="m77">
<mml:mrow>
<mml:mtext>Cross</mml:mtext>
<mml:mo>-</mml:mo>
<mml:mtext>Attention</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>softmax</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
<label>(18)</label>
</disp-formula>
</p>
<p>Here, <inline-formula id="inf60">
<mml:math id="m78">
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> comes from the linear transformation of one feature set, while <inline-formula id="inf61">
<mml:math id="m79">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf62">
<mml:math id="m80">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> come from the linear transformations of the other feature set. Therefore, relevant semantic feature information can be extracted from the bioinformatics features, as expressed by <xref ref-type="disp-formula" rid="e19">Equations 19</xref>, <xref ref-type="disp-formula" rid="e20">20</xref>:<disp-formula id="e19">
<mml:math id="m81">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Cross</mml:mtext>
<mml:mo>-</mml:mo>
<mml:mtext>Attention</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(19)</label>
</disp-formula>
<disp-formula id="e20">
<mml:math id="m82">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Cross</mml:mtext>
<mml:mo>-</mml:mo>
<mml:mtext>Attention</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(20)</label>
</disp-formula>
</p>
<p>Finally, we perform a weighted fusion of the outputs from both Self-attention and Cross-attention (<xref ref-type="disp-formula" rid="e21">Equation 21</xref>):<disp-formula id="e21">
<mml:math id="m83">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(21)</label>
</disp-formula>
</p>
<p>where <inline-formula id="inf63">
<mml:math id="m84">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a weight parameter. The fused feature representation, <inline-formula id="inf64">
<mml:math id="m85">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, is then passed to the subsequent dense layers.</p>
<p>Because positional biases around modified cysteines are supported by redox chemistry and chemoproteomics (<xref ref-type="bibr" rid="B20">Gupta and Carroll, 2014</xref>; <xref ref-type="bibr" rid="B23">Huang et al., 2019</xref>), the attention weights in the Adaptive Feature Fusion module serve as a proxy for residue-level importance within the <inline-formula id="inf65">
<mml:math id="m86">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> window, encouraging the model to up-weight positions consistent with these biologically plausible contexts.</p>
</sec>
<sec id="s2-7">
<title>2.6 Cross-validation and performance evaluation</title>
<p>To evaluate the robustness and generalization capability of our proposed method, we employed a repeated 10-fold cross-validation strategy. Specifically, the entire dataset was randomly partitioned into 10 folds, of which 9 folds were used for training and one fold for testing. Similar to other studies, we adopted commonly used evaluation metrics, including the Area Under the Receiver Operating Characteristic Curve (auROC), Area Under the Precision-Recall Curve (auPRC), Sensitivity (Sn), Specificity (Sp), Accuracy (ACC), and Matthews Correlation Coefficient (MCC), to comprehensively assess the model&#x2019;s predictive performance. The formulas for the latter four metrics are as <xref ref-type="disp-formula" rid="e22">Equations 22</xref>&#x2013;<xref ref-type="disp-formula" rid="e25">25</xref>:<disp-formula id="e22">
<mml:math id="m87">
<mml:mrow>
<mml:mtext>Sn</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(22)</label>
</disp-formula>
<disp-formula id="e23">
<mml:math id="m88">
<mml:mrow>
<mml:mtext>Sp</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(23)</label>
</disp-formula>
<disp-formula id="e24">
<mml:math id="m89">
<mml:mrow>
<mml:mtext>ACC</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(24)</label>
</disp-formula>
<disp-formula id="e25">
<mml:math id="m90">
<mml:mrow>
<mml:mtext>MCC</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(25)</label>
</disp-formula>
</p>
<p>Where <inline-formula id="inf66">
<mml:math id="m91">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf67">
<mml:math id="m92">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf68">
<mml:math id="m93">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf69">
<mml:math id="m94">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represent the true positives, true negatives, false positives, and false negatives, respectively. Since we use an imbalanced dataset, it is highly appropriate to include auROC and auPRC as evaluation metrics. auPRC better reflects the model&#x2019;s performance on the positive class, especially in situations of class imbalance (<xref ref-type="bibr" rid="B22">Hancock et al., 2023</xref>).</p>
<p>Beyond model performance assessment, we quantitatively evaluate feature adequacy using a sequence compression&#x2013;based approximation of Kolmogorov complexity (KC) (<xref ref-type="bibr" rid="B34">Li and Vit&#xe1;nyi, 1997</xref>). In plain terms, Kolmogorov complexity asks &#x201c;what is the shortest description that can reproduce a given sequence?&#x201c;. If the feature representation preserves most of the regularities and patterns in the raw sequence, then compressing the features (or the features concatenated with the raw sequence) will be nearly as effective as compressing the raw sequence itself. Therefore, a small estimated information-loss rate indicates that the features retain most sequence-level biochemical signals relevant to prediction, whereas a large loss suggests that important patterns may have been discarded. The information-loss rate <inline-formula id="inf70">
<mml:math id="m95">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>loss</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is defined as <xref ref-type="disp-formula" rid="e26">Equation 26</xref>:<disp-formula id="e26">
<mml:math id="m96">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>loss</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="script">K</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">K</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="script">K</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="script">K</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">K</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(26)</label>
</disp-formula>Here <inline-formula id="inf71">
<mml:math id="m97">
<mml:mrow>
<mml:mi mathvariant="script">K</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the (prefix) Kolmogorov complexity of a string <inline-formula id="inf72">
<mml:math id="m98">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, formally (<xref ref-type="disp-formula" rid="e27">Equation 27</xref>)<disp-formula id="e27">
<mml:math id="m99">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mi mathvariant="script">K</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>min</mml:mi>
<mml:mrow>
<mml:mo>{</mml:mo>
</mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mo>:</mml:mo>
<mml:mtext>&#x2009;a&#x2009;program&#x2009;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mtext>&#x2009;outputs&#x2009;</mml:mtext>
<mml:mi>X</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mtext>on&#x2009;a&#x2009;fixed&#x2009;universal&#x2009;</mml:mtext>
<mml:mtext>Turing&#x2009;machine&#x2009;</mml:mtext>
<mml:mi>U</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mo>}</mml:mo>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(27)</label>
</disp-formula>
</p>
<p>In words, <inline-formula id="inf73">
<mml:math id="m100">
<mml:mrow>
<mml:mi mathvariant="script">K</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the length (in bits) of the shortest program that generates <inline-formula id="inf74">
<mml:math id="m101">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> on <inline-formula id="inf75">
<mml:math id="m102">
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. This definition is invariant up to an additive constant (independent of <inline-formula id="inf76">
<mml:math id="m103">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) due to the invariance theorem (<xref ref-type="bibr" rid="B34">Li and Vit&#xe1;nyi, 1997</xref>). Likewise, <inline-formula id="inf77">
<mml:math id="m104">
<mml:mrow>
<mml:mi mathvariant="script">K</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the joint complexity of the concatenated description of <inline-formula id="inf78">
<mml:math id="m105">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf79">
<mml:math id="m106">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf80">
<mml:math id="m107">
<mml:mrow>
<mml:mi mathvariant="script">K</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the conditional complexity (the additional information needed to reconstruct <inline-formula id="inf81">
<mml:math id="m108">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> given <inline-formula id="inf82">
<mml:math id="m109">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>).</p>
<p>In this study, <inline-formula id="inf83">
<mml:math id="m110">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the raw amino-acid sequence fragment and <inline-formula id="inf84">
<mml:math id="m111">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> the feature-extraction mapping (fastText subword embeddings and the PSSM). Since exact Kolmogorov complexity is non-computable (due to the undecidability of the halting problem) (<xref ref-type="bibr" rid="B34">Li and Vit&#xe1;nyi, 1997</xref>), we approximate it using off-the-shelf lossless compressors, which provide practical upper bounds to <inline-formula id="inf85">
<mml:math id="m112">
<mml:mrow>
<mml:mi mathvariant="script">K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Concretely, let <inline-formula id="inf86">
<mml:math id="m113">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> be the compressed length (bytes) under a fixed ZIP/DEFLATE compressor; we estimate (<xref ref-type="disp-formula" rid="e28">Equation 28</xref>)<disp-formula id="e28">
<mml:math id="m114">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="script">K</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="script">K</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="script">K</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2248;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(28)</label>
</disp-formula>where <inline-formula id="inf87">
<mml:math id="m115">
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denotes concatenation with a delimiter to avoid boundary artifacts. This estimator is closely related to the normalized compression distance framework (<xref ref-type="bibr" rid="B11">Cilibrasi and Vit&#xe1;nyi, 2005</xref>). For biological symbol sequences, compression-based complexity estimates&#x2014;often based on Lempel-Ziv variants&#x2014;have a long history and have been successfully applied to genetic texts (<xref ref-type="bibr" rid="B21">Gusev et al., 2000</xref>).</p>
<p>This metric is particularly crucial for our sulfenylation site prediction task where we process 21-residue sequence fragments without structural information. Given that our features convert symbolic sequences to numerical representations for BiLSTM classification, quantifying information preservation ensures the extracted features maintain biochemically relevant patterns while reducing noise (<xref ref-type="bibr" rid="B61">Zvonkin and Levin, 2007</xref>; <xref ref-type="bibr" rid="B29">Kolmogorov, 1983</xref>).</p>
</sec>
</sec>
<sec sec-type="results|discussion" id="s3">
<title>3 Results and discussions</title>
<sec id="s3-1">
<title>3.1 Sample sequence content analysis</title>
<p>To better illustrate the differences between sulfenylation site samples and non-sulfenylation site samples at the residue level, this study employed the Two-Sample Logo technique. This method utilizes statistical analysis to identify significant distinctions between central cysteine residues and their surrounding amino acids in positive and negative samples (<xref ref-type="bibr" rid="B50">Schneider and Stephens, 1990</xref>; <xref ref-type="bibr" rid="B12">Crooks et al., 2004</xref>). The Two-Sample Logo is a commonly used visualization tool in sequence content analysis for various PTMs, offering clear insights into amino acid distribution patterns around the modification site and serving as explanatory support for site prediction.</p>
<p>
<xref ref-type="fig" rid="F2">Figure 2</xref> depicts the Two-Sample Logo comparison between the positive and negative sample sets used in this study, with a significance threshold (P-value) set to 0.5. The figure reveals notable differences between multiple positive and negative samples. For instance, non-central cysteine residues are more prevalent in negative samples, particularly at positions &#x2b;1, &#x2b;4, &#x2b;5, &#x2b;6, &#x2b;7, &#x2b;8, &#x2b;9, &#x2b;10, &#x2b;14, &#x2b;15, &#x2b;17, and &#x2b;21. This observation is consistent with the findings of the Bi-directional Gated Recurrent Unit network with Self-Attention (BiGRUD-SA) model, despite differences in datasets, highlighting the strong generalization capability of the BioSemAF-BiLSTM model proposed in this study (<xref ref-type="bibr" rid="B60">Zhang et al., 2023</xref>). Moreover, other amino acids also exhibit statistically significant differences between the two sample groups, such as.<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf88">
<mml:math id="m116">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Lysine (K): Frequently appears in positive samples at positions &#x2b;1, &#x2b;3, &#x2b;4, &#x2b;5, &#x2b;6, &#x2b;7, &#x2b;9, &#x2b;15, &#x2b;18, &#x2b;19, &#x2b;20;</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf89">
<mml:math id="m117">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Glutamic acid (E): Frequently appears in positive samples at positions &#x2b;1, &#x2b;4, &#x2b;6, &#x2b;7, &#x2b;8, &#x2b;12, &#x2b;13, &#x2b;14, &#x2b;15, &#x2b;16, &#x2b;18;</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf90">
<mml:math id="m118">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Histidine (H): Frequently appears in negative samples at positions &#x2b;3, &#x2b;6, &#x2b;7, &#x2b;8, &#x2b;9, &#x2b;10, &#x2b;12, &#x2b;13, &#x2b;14, &#x2b;15, &#x2b;16, &#x2b;17, &#x2b;18, &#x2b;20.</p>
</list-item>
</list>
</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Two-Sample Logoshowing position-specific enrichment and depletion of amino acids around the central cysteine (Cys) residue. The horizontal axis indicates sequence positions within a 21-residue window, centered on the target Cys at position 11. The vertical axis represents enrichment or depletion levels of amino acids at each position. Enrichment of charged residues proximal to the central Cys aligns with prior mechanistic and chemoproteomic evidence showing that local electrostatics modulate thiol pKa and sulfenic-acid reactivity (<xref ref-type="bibr" rid="B20">Gupta and Carroll, 2014</xref>).</p>
</caption>
<graphic xlink:href="fgene-16-1616880-g002.tif">
<alt-text content-type="machine-generated">A sequence logo representing enrichment and depletion of amino acids at specific positions. Each position has stacked letters indicating amino acids with different heights representing their relative frequencies. The top section shows enriched amino acids, and the bottom section shows depleted amino acids, both with a maximum change of 15.7%.</alt-text>
</graphic>
</fig>
<p>Importantly, the enrichment of charged residues such as Lys (K) and Glu (E) flanking the central cysteine is consistent with established redox chemistry and chemoproteomic observations: local electrostatics and hydrogen-bonding networks modulate cysteine thiol pKa and the lifetime of sulfenic acid, biasing oxidation toward specific sequence microenvironments (<xref ref-type="bibr" rid="B20">Gupta and Carroll, 2014</xref>). Large-scale surveys likewise report redox-sensitive cysteines occurring in characteristic local contexts across proteomes (<xref ref-type="bibr" rid="B23">Huang et al., 2019</xref>; <xref ref-type="bibr" rid="B18">Fu et al., 2023</xref>; <xref ref-type="bibr" rid="B1">Akter et al., 2018</xref>). This concordance provides biological support for the sequence patterns highlighted by our Two-Sample Logo and for subsequent sequence-based prediction.</p>
<p>These features capture significant distinctions between positive and negative samples at the corresponding P-values, thereby confirming the reliability and effectiveness of the sample data and laying a robust foundation for further research.</p>
<p>Additionally, according to <xref ref-type="bibr" rid="B16">Do et al. (2020)</xref>, no definitive correlation has been established between sulfenylation sites and their surrounding amino acid compositions. However, investigating its biological significance remains crucial. Mu et al. (<xref ref-type="bibr" rid="B39">Mu et al., 2024</xref>) further elucidated the mechanisms underlying protein sulfenylation. They highlighted that variations in the generation and distribution of intracellular oxidants (e.g., hydrogen peroxide) lead to differential oxidant levels across organelles and subcellular compartments, thereby influencing sulfenylation occurrence. The pKa of the cysteine thiol group is significantly modulated by protein structure, including its secondary and tertiary structures, hydrogen bonding, macrodipoles, and electrostatic interactions. Lower pKa values enhance the nucleophilicity of the thiol group, rendering it more susceptible to sulfenylation (<xref ref-type="bibr" rid="B30">Kortemme and Creighton, 1995</xref>; <xref ref-type="bibr" rid="B17">Ferrer-Sueta et al., 2011</xref>). Additionally, the polarization of the thiol group and the non-bonding electrons of adjacent atoms directly modulate its nucleophilicity, thereby affecting sulfenic acid formation.</p>
<p>These studies offer critical biological insights into the sequence-based relationship between sulfenylation sites and their surrounding amino acids. They also provide theoretical support for model development and optimization, further advancing our understanding of sulfenylation mechanisms.</p>
</sec>
<sec id="s3-2">
<title>3.2 Parameter tuning experiment</title>
<p>To ensure optimal model performance, we performed an extensive tuning process for the key hyperparameters of the BioSemAF-BiLSTM model. These hyperparameters include <italic>n</italic>-gram subword length (Length), fastText word vector dimensionality (Dim), and the number of attention heads in the adaptive feature fusion module (Head). For each type of parameter, values were progressively adjusted within reasonable ranges, with appropriate step sizes set to precisely identify local optima. Subsequently, by integrating the optimal settings across multiple parameters, the global best configuration for overall model performance was determined.</p>
<p>
<xref ref-type="table" rid="T2">Table 2</xref> illustrates the effects of varying parameter configurations on model performance, including <italic>n</italic>-gram subword lengths (2, 3, 4), fastText embedding dimensions (30, 60, 90), and the number of attention heads in the adaptive feature fusion module (3, 6, 9). The evaluation metrics used in the analysis include ACC, MCC, Sn, and Sp. After individually tuning each hyperparameter and evaluating the overall performance, the optimal settings for these three key parameters were determined as follows.<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf91">
<mml:math id="m119">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <italic>n</italic>-grams Subword Length: Selected as 3, dividing the protein sequence into subwords with three amino acids as one unit;</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf92">
<mml:math id="m120">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> fastText Word Vector Dimensionality: Determined to be 60, while keeping the other two hyperparameters at their optimal values;</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf93">
<mml:math id="m121">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Number of Attention Heads in the Adaptive Feature Fusion Module: Determined to be 3.</p>
</list-item>
</list>
</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Performance of the BioSemAF-BiLSTM model under different hyperparameter settings.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Hyperparameters</th>
<th align="center">Value</th>
<th align="center">Sn</th>
<th align="center">Sp</th>
<th align="center">ACC</th>
<th align="center">MCC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="3" align="left">Length</td>
<td align="center">2</td>
<td align="center">91.44</td>
<td align="center">85.14</td>
<td align="center">88.24</td>
<td align="center">0.7100</td>
</tr>
<tr>
<td align="center">
<bold>3</bold>
</td>
<td align="center">
<bold>93.52</bold>
</td>
<td align="center">
<bold>87.18</bold>
</td>
<td align="center">
<bold>89.26</bold>
</td>
<td align="center">
<bold>0.7000</bold>
</td>
</tr>
<tr>
<td align="center">4</td>
<td align="center">92.61</td>
<td align="center">84.13</td>
<td align="center">86.69</td>
<td align="center">0.6800</td>
</tr>
<tr>
<td rowspan="3" align="left">Dim</td>
<td align="center">30</td>
<td align="center">89.29</td>
<td align="center">82.58</td>
<td align="center">85.33</td>
<td align="center">0.6500</td>
</tr>
<tr>
<td align="center">
<bold>60</bold>
</td>
<td align="center">
<bold>94.34</bold>
</td>
<td align="center">
<bold>84.95</bold>
</td>
<td align="center">
<bold>89.14</bold>
</td>
<td align="center">
<bold>0.7100</bold>
</td>
</tr>
<tr>
<td align="center">90</td>
<td align="center">89.98</td>
<td align="center">82.76</td>
<td align="center">86.67</td>
<td align="center">0.7000</td>
</tr>
<tr>
<td rowspan="3" align="left">Head</td>
<td align="center">
<bold>3</bold>
</td>
<td align="center">
<bold>90.47</bold>
</td>
<td align="center">
<bold>83.28</bold>
</td>
<td align="center">
<bold>88.95</bold>
</td>
<td align="center">
<bold>0.7000</bold>
</td>
</tr>
<tr>
<td align="center">6</td>
<td align="center">87.89</td>
<td align="center">82.61</td>
<td align="center">86.35</td>
<td align="center">0.6800</td>
</tr>
<tr>
<td align="center">9</td>
<td align="center">87.52</td>
<td align="center">81.45</td>
<td align="center">85.82</td>
<td align="center">0.6800</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Through analyzing the impact of different parameters on model performance during the debugging process, the following conclusions were drawn.<list list-type="simple">
<list-item>
<p>1. The length of <italic>n</italic>-grams subwords significantly affects the model&#x2019;s feature representation ability. Too short may lead to insufficient feature information, while too long may introduce irrelevant contextual information, thus affecting model performance.</p>
</list-item>
<list-item>
<p>2. The change in fastText word vector dimensions mainly affects the semantic expression accuracy of input features. A 60-dimensional vector performs best in balancing computational cost and performance improvement.</p>
</list-item>
<list-item>
<p>3. Increasing the number of attention heads can enhance the feature fusion module&#x2019;s ability to capture multimodal features. However, too many heads can increase model complexity and potentially lead to overfitting.</p>
</list-item>
</list>
</p>
<p>The final experimental results validated the effectiveness of these hyperparameter configurations, laying the foundation for future model applications.</p>
</sec>
<sec id="s3-3">
<title>3.3 Comparison of prediction performance of different models</title>
<p>To comprehensively assess the predictive performance of BioSemAF-BiLSTM, we conducted experiments from two perspectives: data resampling strategies and classifier performance evaluation. To mitigate the issue of data imbalance, multiple up-sampling and down-sampling techniques were explored to enhance the model&#x2019;s classification performance. Results indicate that SVMSMOTE significantly improves the model&#x2019;s generalization ability while preserving the original feature distribution, making it the preferred resampling method for this study. <xref ref-type="table" rid="T3">Table 3</xref> summarizes the experimental outcomes of various sampling techniques.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Experimental results of different data resampling methods.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Method</th>
<th align="center">Algorithm</th>
<th align="center">Sn</th>
<th align="center">Sp</th>
<th align="center">ACC</th>
<th align="center">MCC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="4" align="left">Up-sampling</td>
<td align="center">SMOTE</td>
<td align="center">80.96</td>
<td align="center">69.98</td>
<td align="center">77.43</td>
<td align="center">0.5900</td>
</tr>
<tr>
<td align="center">RandomOverSampler</td>
<td align="center">78.06</td>
<td align="center">67.26</td>
<td align="center">77.64</td>
<td align="center">0.6400</td>
</tr>
<tr>
<td align="center">KMeansSMOTE</td>
<td align="center">88.28</td>
<td align="center">82.49</td>
<td align="center">86.38</td>
<td align="center">0.7000</td>
</tr>
<tr>
<td align="center">SVMSMOTE</td>
<td align="center">89.66</td>
<td align="center">84.22</td>
<td align="center">89.02</td>
<td align="center">0.6800</td>
</tr>
<tr>
<td rowspan="2" align="left">Down-sampling</td>
<td align="center">TomekLinks</td>
<td align="center">68.27</td>
<td align="center">61.42</td>
<td align="center">62.88</td>
<td align="center">0.3100</td>
</tr>
<tr>
<td align="center">NearMiss</td>
<td align="center">60.92</td>
<td align="center">63.81</td>
<td align="center">63.2</td>
<td align="center">0.3200</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>It can be seen that there exists a strong gap between the results of the up-sampling and down-sampling methods. To further interpret the significant performance differences observed between them, we conducted a supplementary experiment using a compression-based approximation of Kolmogorov complexity to quantify the information content of the training data under different sampling strategies. Specifically, the original, upsampled, and downsampled training sets were each saved in CSV format and compressed using the gzip algorithm. For practical estimation, we used gzip as a standard lossless compressor to approximate Kolmogorov complexity. Alternative strategies, such as reduced-alphabet encodings or specialized compressors designed for biological sequences (<xref ref-type="bibr" rid="B21">Gusev et al.,&#xa0;2000</xref>), could yield slightly different absolute compression sizes, but the relative differences between sampling strategies remain robust. Since Kolmogorov complexity is incomputable in general, the size of the compressed file provides a practical estimate of the dataset&#x2019;s algorithmic information content. <xref ref-type="fig" rid="F3">Figure 3</xref> shows the results.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Comparison of training data information content under different sampling strategies. Bars represent the compressed file sizes (in KB) of the original, upsampled, and downsampled training datasets after applying gzip compression. A larger compressed size indicates higher algorithmic information content.</p>
</caption>
<graphic xlink:href="fgene-16-1616880-g003.tif">
<alt-text content-type="machine-generated">Bar chart comparing compressed file sizes of training data with different resampling strategies. KMeansSMOTE, SMOTE, and SVMSMOTE have sizes above original at 21.30, 21.20, and 21.10 KB respectively. The original is 20.34 KB. RandomOverSampler is slightly below original at 20.20 KB, while TomekLinks and NearMiss are significantly lower at 17.30 and 16.80 KB.</alt-text>
</graphic>
</fig>
<p>It can be observed that the up-sampling methods produce gzip-compressed files with sizes similar to that of the original dataset. This is because the new samples generated during up-sampling are typically derived from the existing data distribution through interpolation or duplication. Although the number of samples increases, the redundancy in the data also rises, which limits any substantial increase in information complexity. In contrast, down-sampling directly reduces the number of samples, potentially discarding minority class or boundary instances, which results in a significant loss of information. Consequently, the compressed file sizes for down-sampling methods are noticeably smaller than those of the up-sampling methods, indicating a reduction in information complexity.</p>
<p>This finding provides an information-theoretic justification for the performance differences reported in the main experiments: down-sampling reduces the dataset&#x2019;s descriptive richness and diversity, thereby impairing the model&#x2019;s generalization capability, whereas up-sampling maintains the complexity while improving class balance (<xref ref-type="bibr" rid="B19">Gao et al., 2024</xref>; <xref ref-type="bibr" rid="B53">Thabtah et al., 2020</xref>).</p>
<p>In the classifier performance comparison experiment, we selected traditional machine learning algorithms (such as KNN, SVM) and other deep learning models (such as DNN, 2DCNN, BiLSTM) for comparative testing. To ensure fairness, the hyperparameters of each model were adjusted to optimal values. Specifically, the k value of KNN was set to 10, and the penalty coefficient C for SVM was set to 2, consistent with Do&#x2019;s study (<xref ref-type="bibr" rid="B16">Do et al., 2020</xref>). DNN used a four-layer fully connected network, with dimensions of 256, 128, 64, and 32 for each layer. BiLSTM consisted of two layers of bidirectional long short-term memory units, followed by three fully connected layers. 2DCNN included two convolutional layers, two average pooling layers, and one fully connected output layer, with Sigmoid used as the activation function. In these settings, the models were compared according to the auROC and auPRC metrics, with the results shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. The experimental results demonstrate that BioSemAF-BiLSTM performs excellently on both of these key metrics, with an auROC of 0.96, higher than DNN (0.90), BiLSTM (0.93), and 2DCNN (0.87). Its auPRC also reached 0.89, further validating the model&#x2019;s ability to accurately predict approximately 89% sulfenylation sites.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>AuROC and auPRC of different classifiers.</p>
</caption>
<graphic xlink:href="fgene-16-1616880-g004.tif">
<alt-text content-type="machine-generated">Bar chart comparing the auROC and auPRC values for six models: KNN, SVM, DNN, 2DCNN, BiLSTM, and BioSemAF-BiLSTM. Each model&#x27;s performance is shown with red bars for auROC and green bars for auPRC. Values are labeled on top of each bar, with BioSemAF-BiLSTM having the highest auROC at 0.96 and BiLSTM having the highest auPRC at 0.93.</alt-text>
</graphic>
</fig>
<p>Comprehensive analysis indicates that the outstanding performance of BioSemAF-BiLSTM can be primarily ascribed to two key factors. Firstly, SVMSMOTE effectively balances the ratio of positive to negative samples, thereby ensuring more robust input data. Secondly, integrating bidirectional long short-term memory units with the adaptive feature fusion module enables the model to comprehensively capture contextual information within protein sequences, leading to enhanced feature learning and classification performance in comparison with conventional machine learning algorithms and other deep learning architectures. These findings underscore the high reliability and generalization capacity of BioSemAF-BiLSTM in sulfenylation site prediction, offering valuable insights for future research.</p>
</sec>
<sec id="s3-4">
<title>3.4 Comparison of prediction performance with different features</title>
<p>We employed a combination of word embedding techniques and sequence evolutionary conservation information to generate features, aiming to enhance the model&#x2019;s predictive performance. Regarding word embedding techniques, fastText was compared with other feature representation methods, such as one-hot encoding and Global Vectors (GloVe) (<xref ref-type="bibr" rid="B45">Pennington et al., 2014</xref>). When all methods were evaluated under their respective optimal parameter settings, fastText demonstrated a clear advantage. This result suggests that fastText&#x2019;s ability to utilize subwords plays a crucial role in improving the performance of neural networks, which is consistent with the findings reported by <xref ref-type="bibr" rid="B16">Do et al. (2020)</xref>.</p>
<p>In the generation of features based on sequence evolutionary conservation information, this study compared PSSM with BLOSUM62 and amino acid composition (AAC). For BLOSUM62, the substitution scores corresponding to each residue were extracted from the matrix and arranged into per-residue 20-dimensional vectors, resulting in a feature representation of the same shape as the PSSM, which was then fed into the BiLSTM in the same way as the PSSM features. Similarly, under the condition that all parameters were set to optimal values, the PSSM features exhibited superior prediction performance.This finding further suggests a significant correlation between sequence evolutionary conservation information and sulfenylation sites.</p>
<p>The results are illustrated in <xref ref-type="table" rid="T4">Table 4</xref>. In conclusion, fastText&#x2019;s subword technique and PSSM features have demonstrated their superiority in semantic feature extraction and biological information representation, respectively, providing a solid foundation for the model&#x2019;s efficiency and accuracy. This not only validates the importance of multimodal feature generation but also offers strong reference for future research.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Experimental results of different features and feature combinations.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Method</th>
<th align="center">Sn</th>
<th align="center">Sp</th>
<th align="center">ACC</th>
<th align="center">MCC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">FastText</td>
<td align="center">90.96</td>
<td align="center">83.98</td>
<td align="center">88.53</td>
<td align="center">0.7100</td>
</tr>
<tr>
<td align="left">GloVe</td>
<td align="center">81.94</td>
<td align="center">85.74</td>
<td align="center">83.64</td>
<td align="center">0.6300</td>
</tr>
<tr>
<td align="left">One-hot</td>
<td align="center">68.85</td>
<td align="center">71.44</td>
<td align="center">70.39</td>
<td align="center">0.4200</td>
</tr>
<tr>
<td align="left">PSSM</td>
<td align="center">86.1</td>
<td align="center">90.83</td>
<td align="center">89.25</td>
<td align="center">0.7200</td>
</tr>
<tr>
<td align="left">BLOSUM62</td>
<td align="center">77.81</td>
<td align="center">83.87</td>
<td align="center">82.38</td>
<td align="center">0.5400</td>
</tr>
<tr>
<td align="left">AAC</td>
<td align="center">70.43</td>
<td align="center">64.95</td>
<td align="center">69.52</td>
<td align="center">0.4300</td>
</tr>
<tr>
<td align="left">FastText &#x2b; PSSM</td>
<td align="center">91.86</td>
<td align="center">88.29</td>
<td align="center">90.17</td>
<td align="center">0.7500</td>
</tr>
<tr>
<td align="left">GloVe &#x2b; PSSM</td>
<td align="center">87.52</td>
<td align="center">87.40</td>
<td align="center">87.46</td>
<td align="center">0.7100</td>
</tr>
<tr>
<td align="left">One-hot &#x2b; PSSM</td>
<td align="center">85.42</td>
<td align="center">88.71</td>
<td align="center">87.06</td>
<td align="center">0.6300</td>
</tr>
<tr>
<td align="left">FastText &#x2b; BLOSUM62</td>
<td align="center">89.48</td>
<td align="center">86.32</td>
<td align="center">88.00</td>
<td align="center">0.7000</td>
</tr>
<tr>
<td align="left">GloVe &#x2b; BLOSUM62</td>
<td align="center">85.36</td>
<td align="center">86.91</td>
<td align="center">86.18</td>
<td align="center">0.6750</td>
</tr>
<tr>
<td align="left">FastText &#x2b; PSSM &#x2b; GloVe</td>
<td align="center">92.10</td>
<td align="center">88.11</td>
<td align="center">90.12</td>
<td align="center">0.7400</td>
</tr>
<tr>
<td align="left">FastText &#x2b; PSSM &#x2b; BLOSUM62</td>
<td align="center">91.22</td>
<td align="center">87.12</td>
<td align="center">89.26</td>
<td align="center">0.7400</td>
</tr>
<tr>
<td align="left">All</td>
<td align="center">92.34</td>
<td align="center">87.91</td>
<td align="center">90.65</td>
<td align="center">0.7300</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To further explore the potential complementarity between semantic and biological features, several fusion strategies were designed and evaluated. Among them, the combination of fastText and PSSM achieved promising performance, indicating that integrating diverse modalities enhances the model&#x2019;s representation capability.</p>
<p>Notably, further improvements were observed when additional features such as GloVe and BLOSUM62 were included. For instance, the combination of fastText, PSSM, and GloVe achieved the Sn of 92.10, and integrating all six features yielded the best overall performance (ACC: 90.65, Sn: 92.34). These results demonstrate that while deeper feature fusion can be beneficial, the marginal gains may decrease due to redundancy. In addition, it is worth noting that feature fusion increases not only the input dimensionality but also the time cost of feature extraction and model training. In practice, the combination of fastText and PSSM offers a favorable trade-off between performance and computational efficiency.</p>
</sec>
<sec id="s3-5">
<title>3.5 Feature ablation experiment</title>
<p>In order to evaluate the impact of different feature information on model performance, we designed ablation experiments to observe the role of individual features in the prediction task. <xref ref-type="table" rid="T5">Table 5</xref> presents the performance metrics when using only PSSM features and only word embedding features. The experimental results show that when using only PSSM features or only word vector features as inputs, the model&#x2019;s prediction performance decreases. However, when both PSSM and word vector features are used together, the model&#x2019;s metrics reach their optimal values.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Performance comparison among single-feature inputs and fused-feature inputs. All values are reported as mean <inline-formula id="inf94">
<mml:math id="m122">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> standard deviation and expressed in percentage.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Feature</th>
<th align="center">Sn</th>
<th align="center">Sp</th>
<th align="center">ACC</th>
<th align="center">MCC</th>
<th align="center">
<inline-formula id="inf95">
<mml:math id="m123">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">loss</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Only word vector</td>
<td align="center">76.52 <inline-formula id="inf96">
<mml:math id="m124">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 1.34</td>
<td align="center">83.20 <inline-formula id="inf97">
<mml:math id="m125">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 2.49</td>
<td align="center">78.39 <inline-formula id="inf98">
<mml:math id="m126">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 1.49</td>
<td align="center">59.00 <inline-formula id="inf99">
<mml:math id="m127">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 2.41</td>
<td align="center">21.00 <inline-formula id="inf100">
<mml:math id="m128">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.95</td>
</tr>
<tr>
<td align="left">Only PSSM</td>
<td align="center">80.16 <inline-formula id="inf101">
<mml:math id="m129">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 2.91</td>
<td align="center">71.86 <inline-formula id="inf102">
<mml:math id="m130">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 1.37</td>
<td align="center">77.28 <inline-formula id="inf103">
<mml:math id="m131">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.93</td>
<td align="center">71.00 <inline-formula id="inf104">
<mml:math id="m132">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 1.69</td>
<td align="center">28.00 <inline-formula id="inf105">
<mml:math id="m133">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 1.32</td>
</tr>
<tr>
<td align="left">Both</td>
<td align="center">90.74 <inline-formula id="inf106">
<mml:math id="m134">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 2.70</td>
<td align="center">83.74 <inline-formula id="inf107">
<mml:math id="m135">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 2.74</td>
<td align="center">89.32 <inline-formula id="inf108">
<mml:math id="m136">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.81</td>
<td align="center">70.00 <inline-formula id="inf109">
<mml:math id="m137">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 2.49</td>
<td align="center">11.00 <inline-formula id="inf110">
<mml:math id="m138">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.65</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best performance in each column.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>This conclusion indicates that information from a single aspect (such as sequence evolutionary conservation or semantic features) cannot fully represent the global statistical patterns of the data. Relying solely on one feature may overlook some key information. PSSM features capture the biological evolutionary conservation of protein sequences, revealing their potential functional importance, while word vector features, through subword modeling, effectively represent local semantic relationships in sequences. Therefore, combining bioinformatics features (PSSM) with semantic features (word vectors) allows for a more comprehensive integration of the biological significance and statistical properties of the sequence, significantly improving the model&#x2019;s prediction ability. For the optimal combination of fastText and PSSM features, the estimated information loss during feature extraction is approximately 11%, which is considered acceptable for maintaining sequence-level biochemical relevance (<xref ref-type="bibr" rid="B21">Gusev et al., 2000</xref>). This supports the reliability of the extracted features for downstream sulfenylation site prediction. Through the ablation experiment, we further validated the necessity and effectiveness of multidimensional feature fusion, providing theoretical support for building more robust prediction models.</p>
</sec>
<sec id="s3-6">
<title>3.6 Comparison with other prediction tools</title>
<p>We compared the performance of the BioSemAF-BiLSTM model with existing prediction tools. To ensure the accuracy and fairness of the results, three state-of-the-art prediction tools were selected: iSulf-Cys, Sulf_FSVM, and fastSulf-DNN. These tools were trained and tested using the same dataset as in this study. Additionally, to maintain consistency with previous studies, we followed the standard evaluation protocol of the iSulf-Cys benchmark (<xref ref-type="bibr" rid="B56">Xu et al., 2016</xref>), reporting results from both repeated 10-fold cross-validation on the training set and evaluation on the independent gold-standard test set provided with iSulf-Cys. Within each training set, 10% of the data was further set aside as a validation subset for hyperparameter tuning and early stopping, ensuring that model optimization was not influenced by the final test fold. To reduce randomness due to single partitioning, the 10-fold cross-validation process was repeated 10 times with different random seeds.</p>
<p>These three tools use different feature generation methods and model architectures.<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf111">
<mml:math id="m139">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> fastSulf-DNN: Uses only biological subwords as features and adopts a relatively simple deep learning network architecture.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf112">
<mml:math id="m140">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> iSulf-Cys: Uses physicochemical and distributional properties of amino acids as feature inputs.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf113">
<mml:math id="m141">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Sulf_FSVM: Combines mRMR feature selection techniques for feature optimization.</p>
</list-item>
</list>
</p>
<p>
<xref ref-type="table" rid="T6">Table 6</xref> reports the mean and standard deviation (&#xb1;) of evaluation metrics in 10-fold CV and illustrates the results in independent test. The experimental results show that BioSemAF-BiLSTM outperforms all of these tools in all evaluation metrics. Specifically, on the MCC metric, it exceeds the next highest by about 17% on the independent test; on the ACC metric, it outperforms the next highest method by 12%; on the Sn metric, it surpasses the next highest method by 5%; and on the Sp metric, it exceeds the next highest method by 18%. Furthermore, despite integrating multidimensional features, the network architecture of BioSemAF-BiLSTM is not overly complex, indicating that the model design achieves improved performance while maintaining a certain level of simplicity.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Comparison of BioSemAF-BiLSTM with three state-of-the-art prediction tools in repeated 10-fold cross-validation and on the independent gold-standard test set provided with iSulf-Cys.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Evaluation</th>
<th align="left">Predictors</th>
<th align="center">Sn</th>
<th align="center">Sp</th>
<th align="center">ACC</th>
<th align="center">MCC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="4" align="left">10-fold CV</td>
<td align="left">iSulf-Cys</td>
<td align="center">67.31 &#xb1; 1.85</td>
<td align="center">63.89 &#xb1; 2.17</td>
<td align="center">65.59 &#xb1; 1.63</td>
<td align="center">0.3057 &#xb1; 0.028</td>
</tr>
<tr>
<td align="left">Sulf_FSVM</td>
<td align="center">68.54 &#xb1; 2.03</td>
<td align="center">68.03 &#xb1; 1.95</td>
<td align="center">68.29 &#xb1; 1.77</td>
<td align="center">0.3246 &#xb1; 0.030</td>
</tr>
<tr>
<td align="left">fastSulf-DNN</td>
<td align="center">76.20 &#xb1; 1.65</td>
<td align="center">83.20 &#xb1; 1.42</td>
<td align="center">79.90 &#xb1; 1.33</td>
<td align="center">0.6091 &#xb1; 0.022</td>
</tr>
<tr>
<td align="left">BioSemAF-BiLSTM (Proposed)</td>
<td align="center">91.86 &#xb1; 1.07</td>
<td align="center">88.29 &#xb1; 1.15</td>
<td align="center">90.17 &#xb1; 0.98</td>
<td align="center">0.7538 &#xb1; 0.014</td>
</tr>
<tr>
<td rowspan="4" align="left">Independent Test</td>
<td align="left">iSulf-Cys</td>
<td align="center">68.97</td>
<td align="center">65.67</td>
<td align="center">66.83</td>
<td align="center">0.3300</td>
</tr>
<tr>
<td align="left">Sulf_FSVM</td>
<td align="center">80.89</td>
<td align="center">68.66</td>
<td align="center">72.88</td>
<td align="center">0.4700</td>
</tr>
<tr>
<td align="left">fastSulf-DNN</td>
<td align="center">85.71</td>
<td align="center">69.47</td>
<td align="center">77.09</td>
<td align="center">0.5600</td>
</tr>
<tr>
<td align="left">BioSemAF-BiLSTM (Proposed)</td>
<td align="center">90.19</td>
<td align="center">87.78</td>
<td align="center">89.32</td>
<td align="center">0.7362</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>We developed a prediction tool for identifying protein S-sulfenylation sites, aiming to fully extract informative features from protein sequences. Two main categories of features were constructed: one based on semantic biological subword embedding, utilizing fastText to capture local semantic relationships within protein sequences; the other based on sequence evolutionary conservation, using PSSM features to represent the functional conservation of proteins. After processing these two types of features with BiLSTM, they were integrated through an adaptive feature fusion module and subsequently embedded into a dense layer for classification. To validate the effectiveness of the model, various experiments were conducted, including parameter tuning, ablation studies, and comparative analysis with other prediction tools. Extensive experimental results demonstrated that the model exhibited significant performance advantages in ten-fold cross-validation and achieved a prediction accuracy of 89.32% on the independent test set, substantially outperforming existing state-of-the-art algorithms. This outcome strongly supports the robustness and generalization capability of the proposed model. Identifying sulfenylation sites is an essential step in deciphering protein functional regulation networks. This study not only provides a novel perspective for understanding disease mechanisms, but also offers valuable insights for the development of new therapies, the design of antioxidant strategies, and the exploration of fundamental biological questions. Furthermore, the successful development of this tool highlights its potential applications in precision medicine, providing technological support for personalized diagnosis and innovative drug development. In conclusion, this study advances the field of sulfenylation site prediction by integrating multidimensional features with deep learning techniques. It not only offers technical support for basic scientific research, but also opens new avenues for the future of precision medicine.</p>
<p>In future work, incorporating compression- and complexity-based measures of protein sequences may provide additional complementary features for improving predictive performance. Although in this study a Kolmogorov-complexity-inspired, compression-based evaluation was primarily used to assess feature sufficiency and to guide the resampling strategy, these measures could also be systematically explored as predictive features themselves. Integrating such complexity-aware representations with the proposed framework may further enhance its ability to capture subtle sequence-level information.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s11">Supplementary Material</xref>, further inquiries can be directed to the corresponding author. The source code implementing the proposed method is publicly available at <ext-link ext-link-type="uri" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://github.com/zzhhhh666/BioSemAF-BiLSTM">https://github.com/zzhhhh666/BioSemAF-BiLSTM</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>ZZ: Supervision, Software, Methodology, Validation, Data curation, Conceptualization, Funding acquisition, Resources, Formal Analysis, Writing &#x2013; review and editing, Writing &#x2013; original draft, Project administration, Visualization. YW: Writing &#x2013; review and editing, Investigation, Writing &#x2013; original draft.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research and/or publication of this article.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2025.1616880/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2025.1616880/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Supplementaryfile1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn id="fn1">
<label>1</label>
<p>Code will be released in the final version of the paper</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Akter</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Jung</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Conte</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Lawson</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Lowther</surname>
<given-names>W. T.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Chemical proteomics reveals new targets of cysteine sulfinic acid reductase</article-title>. <source>Nat. Chem. Biol.</source> <volume>14</volume>, <fpage>995</fpage>&#x2013;<lpage>1004</lpage>. <pub-id pub-id-type="doi">10.1038/s41589-018-0116-2</pub-id>
<pub-id pub-id-type="pmid">30177848</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altschul</surname>
<given-names>S. F.</given-names>
</name>
<name>
<surname>Madden</surname>
<given-names>T. L.</given-names>
</name>
<name>
<surname>Sch&#xe4;ffer</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>1997</year>). <article-title>Gapped blast and psi-blast: a new generation of protein database search programs</article-title>. <source>Nucleic acids Res.</source> <volume>25 17</volume>, <fpage>3389</fpage>&#x2013;<lpage>3402</lpage>. <pub-id pub-id-type="doi">10.1093/NAR/25.17.3389</pub-id>
<pub-id pub-id-type="pmid">9254694</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Anjo</surname>
<given-names>S. I.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Hussain</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Farooq</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>McIntyre</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Laughton</surname>
<given-names>C. A.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Protein oxidative modifications in neurodegenerative diseases: from advances in detection and modelling to their use as disease biomarkers</article-title>. <source>Antioxidants</source> <volume>13</volume>, <fpage>681</fpage>. <pub-id pub-id-type="doi">10.3390/antiox13060681</pub-id>
<pub-id pub-id-type="pmid">38929122</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boadu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Deep learning methods for protein function prediction</article-title>. <source>PROTEOMICS</source> <volume>25</volume>, <fpage>e2300471</fpage>. <pub-id pub-id-type="doi">10.1002/pmic.202300471</pub-id>
<pub-id pub-id-type="pmid">38996351</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brandes</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ofer</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Linial</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Asap: a machine-learning framework for local protein properties</article-title>. <source>bioRxiv</source>. <pub-id pub-id-type="doi">10.1101/032532</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bui</surname>
<given-names>V.-M.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>C.-T.</given-names>
</name>
<name>
<surname>Ho</surname>
<given-names>T.-T.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>T.-Y.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Mdd&#x2013;soh: exploiting maximal dependence decomposition to identify s-sulfenylation sites with substrate motifs</article-title>. <source>Bioinformatics</source> <volume>32</volume>, <fpage>165</fpage>&#x2013;<lpage>172</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv558</pub-id>
<pub-id pub-id-type="pmid">26411868</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Caragea</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sinapov</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Silvescu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dobbs</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Honavar</surname>
<given-names>V. G.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Glycosylation site prediction using ensembles of support vector machine classifiers</article-title>. <source>BMC Bioinforma.</source> <volume>8</volume>, <fpage>438</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-8-438</pub-id>
<pub-id pub-id-type="pmid">17996106</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chahla</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Kovacic</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ferhat</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Leloup</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Pathological impact of redox post-translational modifications</article-title>. <source>Antioxidants and Redox Signal.</source> <volume>41</volume>, <fpage>152</fpage>&#x2013;<lpage>180</lpage>. <pub-id pub-id-type="doi">10.1089/ars.2023.0252</pub-id>
<pub-id pub-id-type="pmid">38504589</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chawla</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Bowyer</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hall</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Kegelmeyer</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Smote: synthetic minority over-sampling technique</article-title>. <source>J. Artif. Intell. Res. (JAIR)</source> <volume>16</volume>, <fpage>321</fpage>&#x2013;<lpage>357</lpage>. <pub-id pub-id-type="doi">10.1613/jair.953</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chou</surname>
<given-names>K.-C.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Prediction of protein cellular attributes using pseudo-amino acid composition</article-title>. <source>Structure</source> <volume>43</volume>, <fpage>246</fpage>&#x2013;<lpage>255</lpage>. <pub-id pub-id-type="doi">10.1002/prot.1035</pub-id>
<pub-id pub-id-type="pmid">11288174</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cilibrasi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Vit&#xe1;nyi</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Clustering by compression</article-title>. <source>Inf. Theory, IEEE Trans.</source>
<volume>51</volume>, <fpage>1523</fpage>&#x2013;<lpage>1545</lpage>. <pub-id pub-id-type="doi">10.1109/TIT.2005.844059</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Crooks</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Hon</surname>
<given-names>G. C.</given-names>
</name>
<name>
<surname>Chandonia</surname>
<given-names>J.-M.</given-names>
</name>
<name>
<surname>Brenner</surname>
<given-names>S. E.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Weblogo: a sequence logo generator</article-title>. <source>Genome Res.</source> <volume>14</volume> (<issue>6</issue>), <fpage>1188</fpage>&#x2013;<lpage>1190</lpage>. <pub-id pub-id-type="doi">10.1101/GR.849004</pub-id>
<pub-id pub-id-type="pmid">15173120</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dehzangi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>L&#xf3;pez</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lal</surname>
<given-names>S. P.</given-names>
</name>
<name>
<surname>Taherzadeh</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Michaelson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sattar</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Pssm-suc: accurately predicting succinylation using position specific scoring matrix into bigram for feature extraction</article-title>. <source>J. Theor. Biol.</source> <volume>425</volume>, <fpage>97</fpage>&#x2013;<lpage>102</lpage>. <pub-id pub-id-type="doi">10.1016/j.jtbi.2017.05.005</pub-id>
<pub-id pub-id-type="pmid">28483566</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Demidova</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Klyueva</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Svm classification: optimization with the smote algorithm for the class imbalance problem</article-title>. <fpage>1</fpage>&#x2013;<lpage>4</lpage>. <pub-id pub-id-type="doi">10.1109/MECO.2017.7977136</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Devlin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Toutanova</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Bert: pre-training of deep bidirectional transformers for language understanding</article-title>. <source>
<italic>Corr.</italic> abs/1810</source>, <fpage>04805</fpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1810.04805</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Do</surname>
<given-names>D. T.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>T. T.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>N. Q. K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Using deep neural networks and biological subwords to detect protein s-sulfenylation sites</article-title>. <source>Briefings Bioinforma.</source> <volume>22</volume>, <fpage>bbaa128</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa128</pub-id>
<pub-id pub-id-type="pmid">32613242</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ferrer-Sueta</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Manta</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Botti</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Radi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Trujillo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Denicola</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Factors affecting protein thiol reactivity and specificity in peroxide reduction</article-title>. <source>Chem. Res. Toxicol.</source> <volume>24</volume>, <fpage>434</fpage>&#x2013;<lpage>450</lpage>. <pub-id pub-id-type="doi">10.1021/tx100413v</pub-id>
<pub-id pub-id-type="pmid">21391663</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Jung</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>R. B.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Nucleophilic covalent ligand discovery for the cysteine redoxome</article-title>. <source>Nat. Chem. Biol.</source> <volume>19</volume>, <fpage>1309</fpage>&#x2013;<lpage>1319</lpage>. <pub-id pub-id-type="doi">10.1038/s41589-023-01330-5</pub-id>
<pub-id pub-id-type="pmid">37248412</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Mei</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>An enhanced encoder-decoder network architecture for reducing information loss in image semantic segmentation</article-title>, <fpage>116</fpage>, <lpage>120</lpage>. <pub-id pub-id-type="doi">10.1109/ispds62779.2024.10667589</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gupta</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Carroll</surname>
<given-names>K. S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Sulfenic acid chemistry, detection and cellular lifetime</article-title>. <source>Biochimica Biophysica Acta (BBA) - General Subj.</source> <volume>1840</volume>, <fpage>847</fpage>&#x2013;<lpage>875</lpage>. <pub-id pub-id-type="doi">10.1016/j.bbagen.2013.05.040</pub-id>
<pub-id pub-id-type="pmid">23748139</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gusev</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Nemytikova</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chuzhanova</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>On the complexity measures of genetic sequences</article-title>. <source>Bioinforma. Oxf. Engl.</source> <volume>15</volume>, <fpage>994</fpage>&#x2013;<lpage>999</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/15.12.994</pub-id>
<pub-id pub-id-type="pmid">10745989</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hancock</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Khoshgoftaar</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Evaluating classifier performance with highly imbalanced big data</article-title>. <source>J. Big Data</source> <volume>10</volume>, <fpage>42</fpage>. <pub-id pub-id-type="doi">10.1186/s40537-023-00724-5</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Willems</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>R. B.</given-names>
</name>
<name>
<surname>Bodra</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Mining for protein s-sulfenylation in arabidopsis uncovers redox-sensitive sites</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>116</volume>, <fpage>21256</fpage>&#x2013;<lpage>21261</lpage>. <pub-id pub-id-type="doi">10.1073/.pnas.1906768116</pub-id>
<pub-id pub-id-type="pmid">31578252</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jones</surname>
<given-names>D. C.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Protein secondary structure prediction based on position-specific scoring matrices</article-title>. <source>J. Mol. Biol.</source> <volume>292</volume> (<issue>2</issue>), <fpage>195</fpage>&#x2013;<lpage>202</lpage>. <pub-id pub-id-type="doi">10.1006/jmbi.1999.3091</pub-id>
<pub-id pub-id-type="pmid">10493868</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Joulin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Grave</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Bojanowski</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mikolov</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Bag of tricks for efficient text classification</article-title>. <source>
<italic>ArXiv</italic> abs/1607.01759</source>. <pub-id pub-id-type="doi">10.18653/V1/E17-2068</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ju</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.-Y.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Prediction of s-sulfenylation sites using mrmr feature selection and fuzzy support vector machine algorithm</article-title>. <source>J. Theor. Biol.</source> <volume>457</volume>, <fpage>6</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1016/j.jtbi.2018.08.022</pub-id>
<pub-id pub-id-type="pmid">30125576</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khan</surname>
<given-names>Z. U.</given-names>
</name>
<name>
<surname>Pi</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Deepsspred: a deep learning based sulfenylation site predictor <italic>via</italic> a novel n-segmented optimize federated feature encoder</article-title>. <source>Protein peptide Lett.</source> <volume>28</volume>, <fpage>708</fpage>&#x2013;<lpage>721</lpage>. <pub-id pub-id-type="doi">10.2174/0929866527666201202103411</pub-id>
<pub-id pub-id-type="pmid">33267753</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>AlQahtani</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Noor</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ahmad</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Pssm-sumo: deep learning based intelligent model for prediction of sumoylation sites using discriminative features</article-title>. <source>BMC Bioinforma.</source> <volume>25</volume>, <fpage>284</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-024-05917-0</pub-id>
<pub-id pub-id-type="pmid">39215231</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kolmogorov</surname>
<given-names>A. N.</given-names>
</name>
</person-group> (<year>1983</year>). <article-title>Combinatorial foundations of information theory and the calculus of probabilities</article-title>. <source>Russ. Math. Surv.</source> <volume>38</volume>, <fpage>29</fpage>&#x2013;<lpage>40</lpage>. <pub-id pub-id-type="doi">10.1070/RM1983v038n04ABEH004203</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kortemme</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Creighton</surname>
<given-names>T. E.</given-names>
</name>
</person-group> (<year>1995</year>). <article-title>Ionisation of cysteine residues at the termini of model alpha-helical peptides. relevance to unusual thiol pka values in proteins of the thioredoxin family</article-title>. <source>J. Mol. Biol.</source> <volume>253</volume> (<issue>5</issue>), <fpage>799</fpage>&#x2013;<lpage>812</lpage>. <pub-id pub-id-type="doi">10.1006/JMBI.1995.0592</pub-id>
<pub-id pub-id-type="pmid">7473753</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>LeCun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Deep learning</article-title>. <source>Nature</source> <volume>521</volume>, <fpage>436</fpage>&#x2013;<lpage>444</lpage>. <pub-id pub-id-type="doi">10.1038/nature14539</pub-id>
<pub-id pub-id-type="pmid">26017442</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Leonard</surname>
<given-names>S. E.</given-names>
</name>
<name>
<surname>Reddie</surname>
<given-names>K. G.</given-names>
</name>
<name>
<surname>Carroll</surname>
<given-names>K. S.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Mining the thiol proteome for sulfenic acid modifications reveals new targets for oxidation in cells</article-title>. <source>ACS Chem. Biol.</source> <volume>4</volume>, <fpage>783</fpage>&#x2013;<lpage>799</lpage>. <pub-id pub-id-type="doi">10.1021/cb900105q</pub-id>
<pub-id pub-id-type="pmid">19645509</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dohlman</surname>
<given-names>H. G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Evolutionary conservation of sequence motifs at sites of protein modification</article-title>. <source>J. Biol. Chem.</source> <volume>299</volume>, <fpage>104617</fpage>. <pub-id pub-id-type="doi">10.1016/j.jbc.2023.104617</pub-id>
<pub-id pub-id-type="pmid">36933807</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Vit&#xe1;nyi</surname>
<given-names>P. M. B.</given-names>
</name>
</person-group> (<year>1997</year>). &#x201c;<article-title>An introduction to kolmogorov complexity and its applications</article-title>,&#x201d; in <source>Texts in computer science</source>&#x2013;<lpage>edition</lpage>. <pub-id pub-id-type="doi">10.1007/978-0-387-49820-1</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Deep learning methods for molecular representation and property prediction</article-title>. <source>Drug Discov. Today</source> <volume>27</volume>, <fpage>103373</fpage>. <pub-id pub-id-type="doi">10.1016/j.drudis.2022.103373</pub-id>
<pub-id pub-id-type="pmid">36167282</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mann</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jensen</surname>
<given-names>O. N.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Proteomic analysis of post-translational modifications</article-title>. <source>Nat. Biotechnol.</source> <volume>21</volume>, <fpage>255</fpage>&#x2013;<lpage>261</lpage>. <pub-id pub-id-type="doi">10.1038/nbt0303-255</pub-id>
<pub-id pub-id-type="pmid">12610572</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mikolov</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Corrado</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Dean</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2013a</year>). &#x201c;<article-title>Efficient estimation of word representations in vector space</article-title>,&#x201d; in <source>1st international conference on learning representations, ICLR 2013, Scottsdale, Arizona, USA, May 2-4, 2013, workshop track proceedings</source>.</citation>
</ref>
<ref id="B38">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mikolov</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sutskever</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Corrado</surname>
<given-names>G. S.</given-names>
</name>
<name>
<surname>Dean</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2013b</year>). &#x201c;<article-title>Distributed representations of words and phrases and their compositionality</article-title>,&#x201d;. <source>Advances in neural information processing systems</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Burges</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bottou</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Welling</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ghahramani</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Weinberger</surname>
<given-names>K.</given-names>
</name>
</person-group> (<publisher-name>Red Hook, NY, USA: Curran Associates, Inc.</publisher-name>), <volume>26</volume>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Oxidative stress-mediated protein sulfenylation in human diseases: past, present, and future</article-title>. <source>Redox Biol.</source> <volume>76</volume>, <fpage>103332</fpage>. <pub-id pub-id-type="doi">10.1016/j.redox.2024.103332</pub-id>
<pub-id pub-id-type="pmid">39217848</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nie</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Prediction of protein s-sulfenylation sites using a deep belief network</article-title>. <source>Curr. Bioinforma.</source> <volume>13</volume>, <fpage>461</fpage>&#x2013;<lpage>467</lpage>. <pub-id pub-id-type="doi">10.2174/1574893612666171122152208</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ning</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Dlf-sul: a multi-module deep learning framework for prediction of s-sulfinylation sites in proteins</article-title>. <source>Briefings Bioinforma.</source> <volume>23</volume>, <fpage>bbac323</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac323</pub-id>
<pub-id pub-id-type="pmid">35945138</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Orlov</surname>
<given-names>Y. and N. G. O.</given-names>
</name>
<name>
<surname>Orlova</surname>
<given-names>N. G.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Bioinformatics tools for the sequence complexity estimates</article-title>. <source>Biophys. Rev.</source> <volume>15</volume>, <fpage>1367</fpage>&#x2013;<lpage>1378</lpage>. <pub-id pub-id-type="doi">10.1007/S12551-023-01140-Y</pub-id>
<pub-id pub-id-type="pmid">37974990</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paulsen</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Carroll</surname>
<given-names>K. S.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Cysteine-mediated redox signaling: chemistry, biology, and tools for discovery</article-title>. <source>Chem. Rev.</source> <volume>113</volume>, <fpage>4633</fpage>&#x2013;<lpage>4679</lpage>. <pub-id pub-id-type="doi">10.1021/cr300163e</pub-id>
<pub-id pub-id-type="pmid">23514336</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paulsen</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Truong</surname>
<given-names>T. H.</given-names>
</name>
<name>
<surname>Garcia</surname>
<given-names>F. J.</given-names>
</name>
<name>
<surname>Homann</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Leonard</surname>
<given-names>S. E.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Peroxide-dependent sulfenylation of the egfr catalytic site enhances kinase activity</article-title>. <source>Nat. Chem. Biol.</source> <volume>8</volume>, <fpage>57</fpage>&#x2013;<lpage>64</lpage>. <pub-id pub-id-type="doi">10.1038/nchembio.736</pub-id>
<pub-id pub-id-type="pmid">22158416</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Pennington</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Socher</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Manning</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>GloVe: global vectors for word representation</article-title>,&#x201d; in <source>Proceedings of the 2014 conference on empirical methods in natural Language processing (EMNLP)</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Moschitti</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Daelemans</surname>
<given-names>W.</given-names>
</name>
</person-group> (<publisher-loc>Doha, Qatar</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>1532</fpage>&#x2013;<lpage>1543</lpage>. <pub-id pub-id-type="doi">10.3115/v1/D14-1162</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qian</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Disulfide stress and its role in cardiovascular diseases</article-title>. <source>Redox Biol.</source> <volume>75</volume>, <fpage>103297</fpage>. <pub-id pub-id-type="doi">10.1016/j.redox.2024.103297</pub-id>
<pub-id pub-id-type="pmid">39127015</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Deep learning methods for protein structure prediction</article-title>. <source>MedComm &#x2013; Future Med.</source> <volume>3</volume>, <fpage>e96</fpage>. <pub-id pub-id-type="doi">10.1002/mef2.96</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>B.-Q.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Chou</surname>
<given-names>K.-C.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>iphos-pseevo: identifying human phosphorylated proteins by incorporating evolutionary information into general pseaac <italic>via</italic> grey system theory</article-title>. <source>Mol. Inf.</source> <volume>36</volume>, <fpage>1600010</fpage>. <pub-id pub-id-type="doi">10.1002/minf.201600010</pub-id>
<pub-id pub-id-type="pmid">28488814</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Remmert</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Biegert</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hauser</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>S&#xf6;ding</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Hhblits: Lightning-fast iterative protein sequence searching by hmm-hmm alignment</article-title>. <source>Nat. methods</source> <volume>9</volume>, <fpage>173</fpage>&#x2013;<lpage>175</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.1818</pub-id>
<pub-id pub-id-type="pmid">22198341</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schneider</surname>
<given-names>T. D.</given-names>
</name>
<name>
<surname>Stephens</surname>
<given-names>R. M.</given-names>
</name>
</person-group> (<year>1990</year>). <article-title>Sequence logos: a new way to display consensus sequences</article-title>. <source>Nucleic acids Res.</source> <volume>18</volume> (<issue>20</issue>), <fpage>6097</fpage>&#x2013;<lpage>6100</lpage>. <pub-id pub-id-type="doi">10.1093/NAR/18.20.6097</pub-id>
<pub-id pub-id-type="pmid">2172928</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Mocetse: a mixture-of-convolutional experts and transformer-based model for predicting gram-negative bacterial secreted effectors</article-title>. <source>bioRxiv</source>. <pub-id pub-id-type="doi">10.1101/2025.08.06.668857</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sievers</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Higgins</surname>
<given-names>D. G.</given-names>
</name>
</person-group> (<year>2014</year>). <source>Clustal Omega, accurate alignment of very large numbers of sequences</source>. <publisher-loc>Totowa, NJ</publisher-loc>: <publisher-name>Humana Press</publisher-name>, <fpage>105</fpage>&#x2013;<lpage>116</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-62703-646-7_6</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thabtah</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Hammoud</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kamalov</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Gonsalves</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Data imbalance in classification: experimental evaluation</article-title>. <source>Inf. Sci.</source> <volume>513</volume>, <fpage>429</fpage>&#x2013;<lpage>441</lpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2019.11.004</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shazeer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Parmar</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Uszkoreit</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gomez</surname>
<given-names>A. N.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). &#x201c;<article-title>Attention is all you need</article-title>,&#x201d; in <source>Proceedings of the 31st international conference on neural information processing systems</source> (<publisher-loc>Red Hook, NY, USA</publisher-loc>: <publisher-name>Curran Associates Inc.), NIPS&#x2019;</publisher-name>), <volume>17</volume>, <fpage>6000</fpage>&#x2013;<lpage>6010</lpage>.</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Walsh</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Jefferis</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Post-translational modifications in the context of therapeutic proteins</article-title>. <source>Nat. Biotechnol.</source> <volume>24</volume>, <fpage>1241</fpage>&#x2013;<lpage>1252</lpage>. <pub-id pub-id-type="doi">10.1038/nbt1252</pub-id>
<pub-id pub-id-type="pmid">17033665</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>L.-Y.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>isulf-cys: prediction of s-sulfenylation sites in proteins with physicochemical properties of amino acids</article-title>. <source>PLoS ONE</source> <volume>11</volume>, <fpage>e0154237</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0154237</pub-id>
<pub-id pub-id-type="pmid">27104833</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Carroll</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Liebler</surname>
<given-names>D. C.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Site-specific mapping and quantification of protein S-sulphenylation in cells</article-title>. <source>Nat. Commun.</source> <volume>5</volume>, <fpage>4776</fpage>. <pub-id pub-id-type="doi">10.1038/ncomms5776</pub-id>
<pub-id pub-id-type="pmid">25175731</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Tallman</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Porter</surname>
<given-names>N. A.</given-names>
</name>
<name>
<surname>Carroll</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Liebler</surname>
<given-names>D. C.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Global, <italic>in situ,</italic>, site-specific analysis of protein s-sulfenylation</article-title>. <source>Nat. Protoc.</source> <volume>10</volume>, <fpage>1022</fpage>&#x2013;<lpage>1037</lpage>. <pub-id pub-id-type="doi">10.1038/nprot.2015.062</pub-id>
<pub-id pub-id-type="pmid">26086405</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Structure-aware protein&#x2013;protein interaction site prediction using deep graph convolutional network</article-title>. <source>Bioinformatics</source> <volume>38</volume>, <fpage>125</fpage>&#x2013;<lpage>132</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab643</pub-id>
<pub-id pub-id-type="pmid">34498061</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Bigrud-sa: protein s-sulfenylation sites prediction based on bigru and self-attention</article-title>. <source>Comput. Biol. Med.</source> <volume>163</volume>, <fpage>107145</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.107145</pub-id>
<pub-id pub-id-type="pmid">37336062</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zvonkin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Levin</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>The complexity of finite objects and the development of the concepts of information and randomness by means of the theory of algorithms</article-title>. <source>Russ. Math. Surv.</source> <volume>25</volume>, <fpage>83</fpage>&#x2013;<lpage>124</lpage>. <pub-id pub-id-type="doi">10.1070/.RM1970v025n06ABEH001269</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>